program(1.0) [buildInfo = dict, tensor>({{"coremlc-component-MIL", "3520.4.1"}, {"coremlc-version", "3520.5.1"}, {"coremltools-component-torch", "2.7.1"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "9.0"}})] { func main(tensor cur_len, tensor input_ids, tensor kv_k, tensor kv_v) { tensor var_75 = const()[name = tensor("op_75"), val = tensor([1, 1])]; tensor positions = reshape(shape = var_75, x = cur_len)[name = tensor("positions")]; tensor var_83_axes_0 = const()[name = tensor("op_83_axes_0"), val = tensor([-1])]; tensor var_81_to_fp16_dtype_0 = const()[name = tensor("op_81_to_fp16_dtype_0"), val = tensor("fp16")]; tensor positions_to_fp16 = cast(dtype = var_81_to_fp16_dtype_0, x = positions)[name = tensor("cast_690")]; tensor var_83_cast_fp16 = expand_dims(axes = var_83_axes_0, x = positions_to_fp16)[name = tensor("op_83_cast_fp16")]; tensor var_88_to_fp16 = const()[name = tensor("op_88_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(64)))]; tensor freqs_cast_fp16 = mul(x = var_83_cast_fp16, y = var_88_to_fp16)[name = tensor("freqs_cast_fp16")]; tensor var_91 = const()[name = tensor("op_91"), val = tensor(-1)]; tensor emb_interleave_0 = const()[name = tensor("emb_interleave_0"), val = tensor(false)]; tensor emb_cast_fp16 = concat(axis = var_91, interleave = emb_interleave_0, values = (freqs_cast_fp16, freqs_cast_fp16))[name = tensor("emb_cast_fp16")]; tensor cos_1_cast_fp16 = cos(x = emb_cast_fp16)[name = tensor("cos_1_cast_fp16")]; tensor sin_1_cast_fp16 = sin(x = emb_cast_fp16)[name = tensor("sin_1_cast_fp16")]; tensor pj = const()[name = tensor("pj"), val = tensor([[[[0], [1], [2], [3], [4], [5], [6], [7], [8], [9], [10], [11], [12], [13], [14], [15], [16], [17], [18], [19], [20], [21], [22], [23], [24], [25], [26], [27], [28], [29], [30], [31], [32], [33], [34], [35], [36], [37], [38], [39], [40], [41], [42], [43], [44], [45], [46], [47], [48], [49], [50], [51], [52], [53], [54], [55], [56], [57], [58], [59], [60], [61], [62], [63], [64], [65], [66], [67], [68], [69], [70], [71], [72], [73], [74], [75], [76], [77], [78], [79], [80], [81], [82], [83], [84], [85], [86], [87], [88], [89], [90], [91], [92], [93], [94], [95], [96], [97], [98], [99], [100], [101], [102], [103], [104], [105], [106], [107], [108], [109], [110], [111], [112], [113], [114], [115], [116], [117], [118], [119], [120], [121], [122], [123], [124], [125], [126], [127], [128], [129], [130], [131], [132], [133], [134], [135], [136], [137], [138], [139], [140], [141], [142], [143], [144], [145], [146], [147], [148], [149], [150], [151], [152], [153], [154], [155], [156], [157], [158], [159], [160], [161], [162], [163], [164], [165], [166], [167], [168], [169], [170], [171], [172], [173], [174], [175], [176], [177], [178], [179], [180], [181], [182], [183], [184], [185], [186], [187], [188], [189], [190], [191], [192], [193], [194], [195], [196], [197], [198], [199], [200], [201], [202], [203], [204], [205], [206], [207], [208], [209], [210], [211], [212], [213], [214], [215], [216], [217], [218], [219], [220], [221], [222], [223], [224], [225], [226], [227], [228], [229], [230], [231], [232], [233], [234], [235], [236], [237], [238], [239], [240], [241], [242], [243], [244], [245], [246], [247], [248], [249], [250], [251], [252], [253], [254], [255], [256], [257], [258], [259], [260], [261], [262], [263], [264], [265], [266], [267], [268], [269], [270], [271], [272], [273], [274], [275], [276], [277], [278], [279], [280], [281], [282], [283], [284], [285], [286], [287], [288], [289], [290], [291], [292], [293], [294], [295], [296], [297], [298], [299], [300], [301], [302], [303], [304], [305], [306], [307], [308], [309], [310], [311], [312], [313], [314], [315], [316], [317], [318], [319], [320], [321], [322], [323], [324], [325], [326], [327], [328], [329], [330], [331], [332], [333], [334], [335], [336], [337], [338], [339], [340], [341], [342], [343], [344], [345], [346], [347], [348], [349], [350], [351], [352], [353], [354], [355], [356], [357], [358], [359], [360], [361], [362], [363], [364], [365], [366], [367], [368], [369], [370], [371], [372], [373], [374], [375], [376], [377], [378], [379], [380], [381], [382], [383], [384], [385], [386], [387], [388], [389], [390], [391], [392], [393], [394], [395], [396], [397], [398], [399], [400], [401], [402], [403], [404], [405], [406], [407], [408], [409], [410], [411], [412], [413], [414], [415], [416], [417], [418], [419], [420], [421], [422], [423], [424], [425], [426], [427], [428], [429], [430], [431], [432], [433], [434], [435], [436], [437], [438], [439], [440], [441], [442], [443], [444], [445], [446], [447], [448], [449], [450], [451], [452], [453], [454], [455], [456], [457], [458], [459], [460], [461], [462], [463], [464], [465], [466], [467], [468], [469], [470], [471], [472], [473], [474], [475], [476], [477], [478], [479], [480], [481], [482], [483], [484], [485], [486], [487], [488], [489], [490], [491], [492], [493], [494], [495], [496], [497], [498], [499], [500], [501], [502], [503], [504], [505], [506], [507], [508], [509], [510], [511], [512], [513], [514], [515], [516], [517], [518], [519], [520], [521], [522], [523], [524], [525], [526], [527], [528], [529], [530], [531], [532], [533], [534], [535], [536], [537], [538], [539], [540], [541], [542], [543], [544], [545], [546], [547], [548], [549], [550], [551], [552], [553], [554], [555], [556], [557], [558], [559], [560], [561], [562], [563], [564], [565], [566], [567], [568], [569], [570], [571], [572], [573], [574], [575], [576], [577], [578], [579], [580], [581], [582], [583], [584], [585], [586], [587], [588], [589], [590], [591], [592], [593], [594], [595], [596], [597], [598], [599], [600], [601], [602], [603], [604], [605], [606], [607], [608], [609], [610], [611], [612], [613], [614], [615], [616], [617], [618], [619], [620], [621], [622], [623], [624], [625], [626], [627], [628], [629], [630], [631], [632], [633], [634], [635], [636], [637], [638], [639], [640], [641], [642], [643], [644], [645], [646], [647], [648], [649], [650], [651], [652], [653], [654], [655], [656], [657], [658], [659], [660], [661], [662], [663], [664], [665], [666], [667], [668], [669], [670], [671], [672], [673], [674], [675], [676], [677], [678], [679], [680], [681], [682], [683], [684], [685], [686], [687], [688], [689], [690], [691], [692], [693], [694], [695], [696], [697], [698], [699], [700], [701], [702], [703], [704], [705], [706], [707], [708], [709], [710], [711], [712], [713], [714], [715], [716], [717], [718], [719], [720], [721], [722], [723], [724], [725], [726], [727], [728], [729], [730], [731], [732], [733], [734], [735], [736], [737], [738], [739], [740], [741], [742], [743], [744], [745], [746], [747], [748], [749], [750], [751], [752], [753], [754], [755], [756], [757], [758], [759], [760], [761], [762], [763], [764], [765], [766], [767], [768], [769], [770], [771], [772], [773], [774], [775], [776], [777], [778], [779], [780], [781], [782], [783], [784], [785], [786], [787], [788], [789], [790], [791], [792], [793], [794], [795], [796], [797], [798], [799], [800], [801], [802], [803], [804], [805], [806], [807], [808], [809], [810], [811], [812], [813], [814], [815], [816], [817], [818], [819], [820], [821], [822], [823], [824], [825], [826], [827], [828], [829], [830], [831], [832], [833], [834], [835], [836], [837], [838], [839], [840], [841], [842], [843], [844], [845], [846], [847], [848], [849], [850], [851], [852], [853], [854], [855], [856], [857], [858], [859], [860], [861], [862], [863], [864], [865], [866], [867], [868], [869], [870], [871], [872], [873], [874], [875], [876], [877], [878], [879], [880], [881], [882], [883], [884], [885], [886], [887], [888], [889], [890], [891], [892], [893], [894], [895], [896], [897], [898], [899], [900], [901], [902], [903], [904], [905], [906], [907], [908], [909], [910], [911], [912], [913], [914], [915], [916], [917], [918], [919], [920], [921], [922], [923], [924], [925], [926], [927], [928], [929], [930], [931], [932], [933], [934], [935], [936], [937], [938], [939], [940], [941], [942], [943], [944], [945], [946], [947], [948], [949], [950], [951], [952], [953], [954], [955], [956], [957], [958], [959], [960], [961], [962], [963], [964], [965], [966], [967], [968], [969], [970], [971], [972], [973], [974], [975], [976], [977], [978], [979], [980], [981], [982], [983], [984], [985], [986], [987], [988], [989], [990], [991], [992], [993], [994], [995], [996], [997], [998], [999], [1000], [1001], [1002], [1003], [1004], [1005], [1006], [1007], [1008], [1009], [1010], [1011], [1012], [1013], [1014], [1015], [1016], [1017], [1018], [1019], [1020], [1021], [1022], [1023], [1024], [1025], [1026], [1027], [1028], [1029], [1030], [1031], [1032], [1033], [1034], [1035], [1036], [1037], [1038], [1039], [1040], [1041], [1042], [1043], [1044], [1045], [1046], [1047], [1048], [1049], [1050], [1051], [1052], [1053], [1054], [1055], [1056], [1057], [1058], [1059], [1060], [1061], [1062], [1063], [1064], [1065], [1066], [1067], [1068], [1069], [1070], [1071], [1072], [1073], [1074], [1075], [1076], [1077], [1078], [1079], [1080], [1081], [1082], [1083], [1084], [1085], [1086], [1087], [1088], [1089], [1090], [1091], [1092], [1093], [1094], [1095], [1096], [1097], [1098], [1099], [1100], [1101], [1102], [1103], [1104], [1105], [1106], [1107], [1108], [1109], [1110], [1111], [1112], [1113], [1114], [1115], [1116], [1117], [1118], [1119], [1120], [1121], [1122], [1123], [1124], [1125], [1126], [1127], [1128], [1129], [1130], [1131], [1132], [1133], [1134], [1135], [1136], [1137], [1138], [1139], [1140], [1141], [1142], [1143], [1144], [1145], [1146], [1147], [1148], [1149], [1150], [1151], [1152], [1153], [1154], [1155], [1156], [1157], [1158], [1159], [1160], [1161], [1162], [1163], [1164], [1165], [1166], [1167], [1168], [1169], [1170], [1171], [1172], [1173], [1174], [1175], [1176], [1177], [1178], [1179], [1180], [1181], [1182], [1183], [1184], [1185], [1186], [1187], [1188], [1189], [1190], [1191], [1192], [1193], [1194], [1195], [1196], [1197], [1198], [1199], [1200], [1201], [1202], [1203], [1204], [1205], [1206], [1207], [1208], [1209], [1210], [1211], [1212], [1213], [1214], [1215], [1216], [1217], [1218], [1219], [1220], [1221], [1222], [1223], [1224], [1225], [1226], [1227], [1228], [1229], [1230], [1231], [1232], [1233], [1234], [1235], [1236], [1237], [1238], [1239], [1240], [1241], [1242], [1243], [1244], [1245], [1246], [1247], [1248], [1249], [1250], [1251], [1252], [1253], [1254], [1255], [1256], [1257], [1258], [1259], [1260], [1261], [1262], [1263], [1264], [1265], [1266], [1267], [1268], [1269], [1270], [1271], [1272], [1273], [1274], [1275], [1276], [1277], [1278], [1279], [1280], [1281], [1282], [1283], [1284], [1285], [1286], [1287], [1288], [1289], [1290], [1291], [1292], [1293], [1294], [1295], [1296], [1297], [1298], [1299], [1300], [1301], [1302], [1303], [1304], [1305], [1306], [1307], [1308], [1309], [1310], [1311], [1312], [1313], [1314], [1315], [1316], [1317], [1318], [1319], [1320], [1321], [1322], [1323], [1324], [1325], [1326], [1327], [1328], [1329], [1330], [1331], [1332], [1333], [1334], [1335], [1336], [1337], [1338], [1339], [1340], [1341], [1342], [1343], [1344], [1345], [1346], [1347], [1348], [1349], [1350], [1351], [1352], [1353], [1354], [1355], [1356], [1357], [1358], [1359], [1360], [1361], [1362], [1363], [1364], [1365], [1366], [1367], [1368], [1369], [1370], [1371], [1372], [1373], [1374], [1375], [1376], [1377], [1378], [1379], [1380], [1381], [1382], [1383], [1384], [1385], [1386], [1387], [1388], [1389], [1390], [1391], [1392], [1393], [1394], [1395], [1396], [1397], [1398], [1399], [1400], [1401], [1402], [1403], [1404], [1405], [1406], [1407], [1408], [1409], [1410], [1411], [1412], [1413], [1414], [1415], [1416], [1417], [1418], [1419], [1420], [1421], [1422], [1423], [1424], [1425], [1426], [1427], [1428], [1429], [1430], [1431], [1432], [1433], [1434], [1435], [1436], [1437], [1438], [1439], [1440], [1441], [1442], [1443], [1444], [1445], [1446], [1447], [1448], [1449], [1450], [1451], [1452], [1453], [1454], [1455], [1456], [1457], [1458], [1459], [1460], [1461], [1462], [1463], [1464], [1465], [1466], [1467], [1468], [1469], [1470], [1471], [1472], [1473], [1474], [1475], [1476], [1477], [1478], [1479], [1480], [1481], [1482], [1483], [1484], [1485], [1486], [1487], [1488], [1489], [1490], [1491], [1492], [1493], [1494], [1495], [1496], [1497], [1498], [1499], [1500], [1501], [1502], [1503], [1504], [1505], [1506], [1507], [1508], [1509], [1510], [1511], [1512], [1513], [1514], [1515], [1516], [1517], [1518], [1519], [1520], [1521], [1522], [1523], [1524], [1525], [1526], [1527], [1528], [1529], [1530], [1531], [1532], [1533], [1534], [1535], [1536], [1537], [1538], [1539], [1540], [1541], [1542], [1543], [1544], [1545], [1546], [1547], [1548], [1549], [1550], [1551], [1552], [1553], [1554], [1555], [1556], [1557], [1558], [1559], [1560], [1561], [1562], [1563], [1564], [1565], [1566], [1567], [1568], [1569], [1570], [1571], [1572], [1573], [1574], [1575], [1576], [1577], [1578], [1579], [1580], [1581], [1582], [1583], [1584], [1585], [1586], [1587], [1588], [1589], [1590], [1591], [1592], [1593], [1594], [1595], [1596], [1597], [1598], [1599], [1600], [1601], [1602], [1603], [1604], [1605], [1606], [1607], [1608], [1609], [1610], [1611], [1612], [1613], [1614], [1615], [1616], [1617], [1618], [1619], [1620], [1621], [1622], [1623], [1624], [1625], [1626], [1627], [1628], [1629], [1630], [1631], [1632], [1633], [1634], [1635], [1636], [1637], [1638], [1639], [1640], [1641], [1642], [1643], [1644], [1645], [1646], [1647], [1648], [1649], [1650], [1651], [1652], [1653], [1654], [1655], [1656], [1657], [1658], [1659], [1660], [1661], [1662], [1663], [1664], [1665], [1666], [1667], [1668], [1669], [1670], [1671], [1672], [1673], [1674], [1675], [1676], [1677], [1678], [1679], [1680], [1681], [1682], [1683], [1684], [1685], [1686], [1687], [1688], [1689], [1690], [1691], [1692], [1693], [1694], [1695], [1696], [1697], [1698], [1699], [1700], [1701], [1702], [1703], [1704], [1705], [1706], [1707], [1708], [1709], [1710], [1711], [1712], [1713], [1714], [1715], [1716], [1717], [1718], [1719], [1720], [1721], [1722], [1723], [1724], [1725], [1726], [1727], [1728], [1729], [1730], [1731], [1732], [1733], [1734], [1735], [1736], [1737], [1738], [1739], [1740], [1741], [1742], [1743], [1744], [1745], [1746], [1747], [1748], [1749], [1750], [1751], [1752], [1753], [1754], [1755], [1756], [1757], [1758], [1759], [1760], [1761], [1762], [1763], [1764], [1765], [1766], [1767], [1768], [1769], [1770], [1771], [1772], [1773], [1774], [1775], [1776], [1777], [1778], [1779], [1780], [1781], [1782], [1783], [1784], [1785], [1786], [1787], [1788], [1789], [1790], [1791], [1792], [1793], [1794], [1795], [1796], [1797], [1798], [1799], [1800], [1801], [1802], [1803], [1804], [1805], [1806], [1807], [1808], [1809], [1810], [1811], [1812], [1813], [1814], [1815], [1816], [1817], [1818], [1819], [1820], [1821], [1822], [1823], [1824], [1825], [1826], [1827], [1828], [1829], [1830], [1831], [1832], [1833], [1834], [1835], [1836], [1837], [1838], [1839], [1840], [1841], [1842], [1843], [1844], [1845], [1846], [1847], [1848], [1849], [1850], [1851], [1852], [1853], [1854], [1855], [1856], [1857], [1858], [1859], [1860], [1861], [1862], [1863], [1864], [1865], [1866], [1867], [1868], [1869], [1870], [1871], [1872], [1873], [1874], [1875], [1876], [1877], [1878], [1879], [1880], [1881], [1882], [1883], [1884], [1885], [1886], [1887], [1888], [1889], [1890], [1891], [1892], [1893], [1894], [1895], [1896], [1897], [1898], [1899], [1900], [1901], [1902], [1903], [1904], [1905], [1906], [1907], [1908], [1909], [1910], [1911], [1912], [1913], [1914], [1915], [1916], [1917], [1918], [1919], [1920], [1921], [1922], [1923], [1924], [1925], [1926], [1927], [1928], [1929], [1930], [1931], [1932], [1933], [1934], [1935], [1936], [1937], [1938], [1939], [1940], [1941], [1942], [1943], [1944], [1945], [1946], [1947], [1948], [1949], [1950], [1951], [1952], [1953], [1954], [1955], [1956], [1957], [1958], [1959], [1960], [1961], [1962], [1963], [1964], [1965], [1966], [1967], [1968], [1969], [1970], [1971], [1972], [1973], [1974], [1975], [1976], [1977], [1978], [1979], [1980], [1981], [1982], [1983], [1984], [1985], [1986], [1987], [1988], [1989], [1990], [1991], [1992], [1993], [1994], [1995], [1996], [1997], [1998], [1999], [2000], [2001], [2002], [2003], [2004], [2005], [2006], [2007], [2008], [2009], [2010], [2011], [2012], [2013], [2014], [2015], [2016], [2017], [2018], [2019], [2020], [2021], [2022], [2023], [2024], [2025], [2026], [2027], [2028], [2029], [2030], [2031], [2032], [2033], [2034], [2035], [2036], [2037], [2038], [2039], [2040], [2041], [2042], [2043], [2044], [2045], [2046], [2047]]]])]; tensor var_105 = const()[name = tensor("op_105"), val = tensor([1, 1, 1, 1])]; tensor var_106 = reshape(shape = var_105, x = cur_len)[name = tensor("op_106")]; tensor var_107 = equal(x = pj, y = var_106)[name = tensor("op_107")]; tensor var_118 = const()[name = tensor("op_118"), val = tensor([[[[0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173, 174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 209, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226, 227, 228, 229, 230, 231, 232, 233, 234, 235, 236, 237, 238, 239, 240, 241, 242, 243, 244, 245, 246, 247, 248, 249, 250, 251, 252, 253, 254, 255, 256, 257, 258, 259, 260, 261, 262, 263, 264, 265, 266, 267, 268, 269, 270, 271, 272, 273, 274, 275, 276, 277, 278, 279, 280, 281, 282, 283, 284, 285, 286, 287, 288, 289, 290, 291, 292, 293, 294, 295, 296, 297, 298, 299, 300, 301, 302, 303, 304, 305, 306, 307, 308, 309, 310, 311, 312, 313, 314, 315, 316, 317, 318, 319, 320, 321, 322, 323, 324, 325, 326, 327, 328, 329, 330, 331, 332, 333, 334, 335, 336, 337, 338, 339, 340, 341, 342, 343, 344, 345, 346, 347, 348, 349, 350, 351, 352, 353, 354, 355, 356, 357, 358, 359, 360, 361, 362, 363, 364, 365, 366, 367, 368, 369, 370, 371, 372, 373, 374, 375, 376, 377, 378, 379, 380, 381, 382, 383, 384, 385, 386, 387, 388, 389, 390, 391, 392, 393, 394, 395, 396, 397, 398, 399, 400, 401, 402, 403, 404, 405, 406, 407, 408, 409, 410, 411, 412, 413, 414, 415, 416, 417, 418, 419, 420, 421, 422, 423, 424, 425, 426, 427, 428, 429, 430, 431, 432, 433, 434, 435, 436, 437, 438, 439, 440, 441, 442, 443, 444, 445, 446, 447, 448, 449, 450, 451, 452, 453, 454, 455, 456, 457, 458, 459, 460, 461, 462, 463, 464, 465, 466, 467, 468, 469, 470, 471, 472, 473, 474, 475, 476, 477, 478, 479, 480, 481, 482, 483, 484, 485, 486, 487, 488, 489, 490, 491, 492, 493, 494, 495, 496, 497, 498, 499, 500, 501, 502, 503, 504, 505, 506, 507, 508, 509, 510, 511, 512, 513, 514, 515, 516, 517, 518, 519, 520, 521, 522, 523, 524, 525, 526, 527, 528, 529, 530, 531, 532, 533, 534, 535, 536, 537, 538, 539, 540, 541, 542, 543, 544, 545, 546, 547, 548, 549, 550, 551, 552, 553, 554, 555, 556, 557, 558, 559, 560, 561, 562, 563, 564, 565, 566, 567, 568, 569, 570, 571, 572, 573, 574, 575, 576, 577, 578, 579, 580, 581, 582, 583, 584, 585, 586, 587, 588, 589, 590, 591, 592, 593, 594, 595, 596, 597, 598, 599, 600, 601, 602, 603, 604, 605, 606, 607, 608, 609, 610, 611, 612, 613, 614, 615, 616, 617, 618, 619, 620, 621, 622, 623, 624, 625, 626, 627, 628, 629, 630, 631, 632, 633, 634, 635, 636, 637, 638, 639, 640, 641, 642, 643, 644, 645, 646, 647, 648, 649, 650, 651, 652, 653, 654, 655, 656, 657, 658, 659, 660, 661, 662, 663, 664, 665, 666, 667, 668, 669, 670, 671, 672, 673, 674, 675, 676, 677, 678, 679, 680, 681, 682, 683, 684, 685, 686, 687, 688, 689, 690, 691, 692, 693, 694, 695, 696, 697, 698, 699, 700, 701, 702, 703, 704, 705, 706, 707, 708, 709, 710, 711, 712, 713, 714, 715, 716, 717, 718, 719, 720, 721, 722, 723, 724, 725, 726, 727, 728, 729, 730, 731, 732, 733, 734, 735, 736, 737, 738, 739, 740, 741, 742, 743, 744, 745, 746, 747, 748, 749, 750, 751, 752, 753, 754, 755, 756, 757, 758, 759, 760, 761, 762, 763, 764, 765, 766, 767, 768, 769, 770, 771, 772, 773, 774, 775, 776, 777, 778, 779, 780, 781, 782, 783, 784, 785, 786, 787, 788, 789, 790, 791, 792, 793, 794, 795, 796, 797, 798, 799, 800, 801, 802, 803, 804, 805, 806, 807, 808, 809, 810, 811, 812, 813, 814, 815, 816, 817, 818, 819, 820, 821, 822, 823, 824, 825, 826, 827, 828, 829, 830, 831, 832, 833, 834, 835, 836, 837, 838, 839, 840, 841, 842, 843, 844, 845, 846, 847, 848, 849, 850, 851, 852, 853, 854, 855, 856, 857, 858, 859, 860, 861, 862, 863, 864, 865, 866, 867, 868, 869, 870, 871, 872, 873, 874, 875, 876, 877, 878, 879, 880, 881, 882, 883, 884, 885, 886, 887, 888, 889, 890, 891, 892, 893, 894, 895, 896, 897, 898, 899, 900, 901, 902, 903, 904, 905, 906, 907, 908, 909, 910, 911, 912, 913, 914, 915, 916, 917, 918, 919, 920, 921, 922, 923, 924, 925, 926, 927, 928, 929, 930, 931, 932, 933, 934, 935, 936, 937, 938, 939, 940, 941, 942, 943, 944, 945, 946, 947, 948, 949, 950, 951, 952, 953, 954, 955, 956, 957, 958, 959, 960, 961, 962, 963, 964, 965, 966, 967, 968, 969, 970, 971, 972, 973, 974, 975, 976, 977, 978, 979, 980, 981, 982, 983, 984, 985, 986, 987, 988, 989, 990, 991, 992, 993, 994, 995, 996, 997, 998, 999, 1000, 1001, 1002, 1003, 1004, 1005, 1006, 1007, 1008, 1009, 1010, 1011, 1012, 1013, 1014, 1015, 1016, 1017, 1018, 1019, 1020, 1021, 1022, 1023, 1024, 1025, 1026, 1027, 1028, 1029, 1030, 1031, 1032, 1033, 1034, 1035, 1036, 1037, 1038, 1039, 1040, 1041, 1042, 1043, 1044, 1045, 1046, 1047, 1048, 1049, 1050, 1051, 1052, 1053, 1054, 1055, 1056, 1057, 1058, 1059, 1060, 1061, 1062, 1063, 1064, 1065, 1066, 1067, 1068, 1069, 1070, 1071, 1072, 1073, 1074, 1075, 1076, 1077, 1078, 1079, 1080, 1081, 1082, 1083, 1084, 1085, 1086, 1087, 1088, 1089, 1090, 1091, 1092, 1093, 1094, 1095, 1096, 1097, 1098, 1099, 1100, 1101, 1102, 1103, 1104, 1105, 1106, 1107, 1108, 1109, 1110, 1111, 1112, 1113, 1114, 1115, 1116, 1117, 1118, 1119, 1120, 1121, 1122, 1123, 1124, 1125, 1126, 1127, 1128, 1129, 1130, 1131, 1132, 1133, 1134, 1135, 1136, 1137, 1138, 1139, 1140, 1141, 1142, 1143, 1144, 1145, 1146, 1147, 1148, 1149, 1150, 1151, 1152, 1153, 1154, 1155, 1156, 1157, 1158, 1159, 1160, 1161, 1162, 1163, 1164, 1165, 1166, 1167, 1168, 1169, 1170, 1171, 1172, 1173, 1174, 1175, 1176, 1177, 1178, 1179, 1180, 1181, 1182, 1183, 1184, 1185, 1186, 1187, 1188, 1189, 1190, 1191, 1192, 1193, 1194, 1195, 1196, 1197, 1198, 1199, 1200, 1201, 1202, 1203, 1204, 1205, 1206, 1207, 1208, 1209, 1210, 1211, 1212, 1213, 1214, 1215, 1216, 1217, 1218, 1219, 1220, 1221, 1222, 1223, 1224, 1225, 1226, 1227, 1228, 1229, 1230, 1231, 1232, 1233, 1234, 1235, 1236, 1237, 1238, 1239, 1240, 1241, 1242, 1243, 1244, 1245, 1246, 1247, 1248, 1249, 1250, 1251, 1252, 1253, 1254, 1255, 1256, 1257, 1258, 1259, 1260, 1261, 1262, 1263, 1264, 1265, 1266, 1267, 1268, 1269, 1270, 1271, 1272, 1273, 1274, 1275, 1276, 1277, 1278, 1279, 1280, 1281, 1282, 1283, 1284, 1285, 1286, 1287, 1288, 1289, 1290, 1291, 1292, 1293, 1294, 1295, 1296, 1297, 1298, 1299, 1300, 1301, 1302, 1303, 1304, 1305, 1306, 1307, 1308, 1309, 1310, 1311, 1312, 1313, 1314, 1315, 1316, 1317, 1318, 1319, 1320, 1321, 1322, 1323, 1324, 1325, 1326, 1327, 1328, 1329, 1330, 1331, 1332, 1333, 1334, 1335, 1336, 1337, 1338, 1339, 1340, 1341, 1342, 1343, 1344, 1345, 1346, 1347, 1348, 1349, 1350, 1351, 1352, 1353, 1354, 1355, 1356, 1357, 1358, 1359, 1360, 1361, 1362, 1363, 1364, 1365, 1366, 1367, 1368, 1369, 1370, 1371, 1372, 1373, 1374, 1375, 1376, 1377, 1378, 1379, 1380, 1381, 1382, 1383, 1384, 1385, 1386, 1387, 1388, 1389, 1390, 1391, 1392, 1393, 1394, 1395, 1396, 1397, 1398, 1399, 1400, 1401, 1402, 1403, 1404, 1405, 1406, 1407, 1408, 1409, 1410, 1411, 1412, 1413, 1414, 1415, 1416, 1417, 1418, 1419, 1420, 1421, 1422, 1423, 1424, 1425, 1426, 1427, 1428, 1429, 1430, 1431, 1432, 1433, 1434, 1435, 1436, 1437, 1438, 1439, 1440, 1441, 1442, 1443, 1444, 1445, 1446, 1447, 1448, 1449, 1450, 1451, 1452, 1453, 1454, 1455, 1456, 1457, 1458, 1459, 1460, 1461, 1462, 1463, 1464, 1465, 1466, 1467, 1468, 1469, 1470, 1471, 1472, 1473, 1474, 1475, 1476, 1477, 1478, 1479, 1480, 1481, 1482, 1483, 1484, 1485, 1486, 1487, 1488, 1489, 1490, 1491, 1492, 1493, 1494, 1495, 1496, 1497, 1498, 1499, 1500, 1501, 1502, 1503, 1504, 1505, 1506, 1507, 1508, 1509, 1510, 1511, 1512, 1513, 1514, 1515, 1516, 1517, 1518, 1519, 1520, 1521, 1522, 1523, 1524, 1525, 1526, 1527, 1528, 1529, 1530, 1531, 1532, 1533, 1534, 1535, 1536, 1537, 1538, 1539, 1540, 1541, 1542, 1543, 1544, 1545, 1546, 1547, 1548, 1549, 1550, 1551, 1552, 1553, 1554, 1555, 1556, 1557, 1558, 1559, 1560, 1561, 1562, 1563, 1564, 1565, 1566, 1567, 1568, 1569, 1570, 1571, 1572, 1573, 1574, 1575, 1576, 1577, 1578, 1579, 1580, 1581, 1582, 1583, 1584, 1585, 1586, 1587, 1588, 1589, 1590, 1591, 1592, 1593, 1594, 1595, 1596, 1597, 1598, 1599, 1600, 1601, 1602, 1603, 1604, 1605, 1606, 1607, 1608, 1609, 1610, 1611, 1612, 1613, 1614, 1615, 1616, 1617, 1618, 1619, 1620, 1621, 1622, 1623, 1624, 1625, 1626, 1627, 1628, 1629, 1630, 1631, 1632, 1633, 1634, 1635, 1636, 1637, 1638, 1639, 1640, 1641, 1642, 1643, 1644, 1645, 1646, 1647, 1648, 1649, 1650, 1651, 1652, 1653, 1654, 1655, 1656, 1657, 1658, 1659, 1660, 1661, 1662, 1663, 1664, 1665, 1666, 1667, 1668, 1669, 1670, 1671, 1672, 1673, 1674, 1675, 1676, 1677, 1678, 1679, 1680, 1681, 1682, 1683, 1684, 1685, 1686, 1687, 1688, 1689, 1690, 1691, 1692, 1693, 1694, 1695, 1696, 1697, 1698, 1699, 1700, 1701, 1702, 1703, 1704, 1705, 1706, 1707, 1708, 1709, 1710, 1711, 1712, 1713, 1714, 1715, 1716, 1717, 1718, 1719, 1720, 1721, 1722, 1723, 1724, 1725, 1726, 1727, 1728, 1729, 1730, 1731, 1732, 1733, 1734, 1735, 1736, 1737, 1738, 1739, 1740, 1741, 1742, 1743, 1744, 1745, 1746, 1747, 1748, 1749, 1750, 1751, 1752, 1753, 1754, 1755, 1756, 1757, 1758, 1759, 1760, 1761, 1762, 1763, 1764, 1765, 1766, 1767, 1768, 1769, 1770, 1771, 1772, 1773, 1774, 1775, 1776, 1777, 1778, 1779, 1780, 1781, 1782, 1783, 1784, 1785, 1786, 1787, 1788, 1789, 1790, 1791, 1792, 1793, 1794, 1795, 1796, 1797, 1798, 1799, 1800, 1801, 1802, 1803, 1804, 1805, 1806, 1807, 1808, 1809, 1810, 1811, 1812, 1813, 1814, 1815, 1816, 1817, 1818, 1819, 1820, 1821, 1822, 1823, 1824, 1825, 1826, 1827, 1828, 1829, 1830, 1831, 1832, 1833, 1834, 1835, 1836, 1837, 1838, 1839, 1840, 1841, 1842, 1843, 1844, 1845, 1846, 1847, 1848, 1849, 1850, 1851, 1852, 1853, 1854, 1855, 1856, 1857, 1858, 1859, 1860, 1861, 1862, 1863, 1864, 1865, 1866, 1867, 1868, 1869, 1870, 1871, 1872, 1873, 1874, 1875, 1876, 1877, 1878, 1879, 1880, 1881, 1882, 1883, 1884, 1885, 1886, 1887, 1888, 1889, 1890, 1891, 1892, 1893, 1894, 1895, 1896, 1897, 1898, 1899, 1900, 1901, 1902, 1903, 1904, 1905, 1906, 1907, 1908, 1909, 1910, 1911, 1912, 1913, 1914, 1915, 1916, 1917, 1918, 1919, 1920, 1921, 1922, 1923, 1924, 1925, 1926, 1927, 1928, 1929, 1930, 1931, 1932, 1933, 1934, 1935, 1936, 1937, 1938, 1939, 1940, 1941, 1942, 1943, 1944, 1945, 1946, 1947, 1948, 1949, 1950, 1951, 1952, 1953, 1954, 1955, 1956, 1957, 1958, 1959, 1960, 1961, 1962, 1963, 1964, 1965, 1966, 1967, 1968, 1969, 1970, 1971, 1972, 1973, 1974, 1975, 1976, 1977, 1978, 1979, 1980, 1981, 1982, 1983, 1984, 1985, 1986, 1987, 1988, 1989, 1990, 1991, 1992, 1993, 1994, 1995, 1996, 1997, 1998, 1999, 2000, 2001, 2002, 2003, 2004, 2005, 2006, 2007, 2008, 2009, 2010, 2011, 2012, 2013, 2014, 2015, 2016, 2017, 2018, 2019, 2020, 2021, 2022, 2023, 2024, 2025, 2026, 2027, 2028, 2029, 2030, 2031, 2032, 2033, 2034, 2035, 2036, 2037, 2038, 2039, 2040, 2041, 2042, 2043, 2044, 2045, 2046, 2047]]]])]; tensor attendable = less_equal(x = var_118, y = var_106)[name = tensor("attendable")]; tensor var_139_after_broadcast_to_fp16 = const()[name = tensor("op_139_after_broadcast_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(256)))]; tensor const_0_after_broadcast_to_fp16 = const()[name = tensor("const_0_after_broadcast_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(4416)))]; tensor attn_mask_cast_fp16 = select(a = var_139_after_broadcast_to_fp16, b = const_0_after_broadcast_to_fp16, cond = attendable)[name = tensor("attn_mask_cast_fp16")]; tensor x_1_batch_dims_0 = const()[name = tensor("x_1_batch_dims_0"), val = tensor(0)]; tensor x_1_validate_indices_0 = const()[name = tensor("x_1_validate_indices_0"), val = tensor(false)]; tensor tok_weight_to_fp16 = const()[name = tensor("tok_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(8576)))]; tensor greater_equal_0_y_0 = const()[name = tensor("greater_equal_0_y_0"), val = tensor(0)]; tensor greater_equal_0 = greater_equal(x = input_ids, y = greater_equal_0_y_0)[name = tensor("greater_equal_0")]; tensor slice_by_index_0 = const()[name = tensor("slice_by_index_0"), val = tensor(217232)]; tensor add_0 = add(x = input_ids, y = slice_by_index_0)[name = tensor("add_0")]; tensor select_0 = select(a = input_ids, b = add_0, cond = greater_equal_0)[name = tensor("select_0")]; tensor x_1_cast_fp16_axis_0 = const()[name = tensor("x_1_cast_fp16_axis_0"), val = tensor(0)]; tensor x_1_cast_fp16 = gather(axis = x_1_cast_fp16_axis_0, batch_dims = x_1_batch_dims_0, indices = select_0, validate_indices = x_1_validate_indices_0, x = tok_weight_to_fp16)[name = tensor("x_1_cast_fp16")]; tensor x_1_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_1_begin_0 = const()[name = tensor("k_cache_1_begin_0"), val = tensor([0, 0, 0, 0, 0])]; tensor k_cache_1_end_0 = const()[name = tensor("k_cache_1_end_0"), val = tensor([1, 1, 4, 2048, 128])]; tensor k_cache_1_end_mask_0 = const()[name = tensor("k_cache_1_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_1_squeeze_mask_0 = const()[name = tensor("k_cache_1_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor kv_k_to_fp16_dtype_0 = const()[name = tensor("kv_k_to_fp16_dtype_0"), val = tensor("fp16")]; tensor kv_k_to_fp16 = cast(dtype = kv_k_to_fp16_dtype_0, x = kv_k)[name = tensor("cast_688")]; tensor k_cache_1_cast_fp16 = slice_by_index(begin = k_cache_1_begin_0, end = k_cache_1_end_0, end_mask = k_cache_1_end_mask_0, squeeze_mask = k_cache_1_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_1_cast_fp16")]; tensor v_cache_1_begin_0 = const()[name = tensor("v_cache_1_begin_0"), val = tensor([0, 0, 0, 0, 0])]; tensor v_cache_1_end_0 = const()[name = tensor("v_cache_1_end_0"), val = tensor([1, 1, 4, 2048, 128])]; tensor v_cache_1_end_mask_0 = const()[name = tensor("v_cache_1_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_1_squeeze_mask_0 = const()[name = tensor("v_cache_1_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor kv_v_to_fp16_dtype_0 = const()[name = tensor("kv_v_to_fp16_dtype_0"), val = tensor("fp16")]; tensor kv_v_to_fp16 = cast(dtype = kv_v_to_fp16_dtype_0, x = kv_v)[name = tensor("cast_687")]; tensor v_cache_1_cast_fp16 = slice_by_index(begin = v_cache_1_begin_0, end = v_cache_1_end_0, end_mask = v_cache_1_end_mask_0, squeeze_mask = v_cache_1_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_1_cast_fp16")]; tensor var_171 = const()[name = tensor("op_171"), val = tensor(-1)]; tensor var_170_promoted = const()[name = tensor("op_170_promoted"), val = tensor(0x1p+1)]; tensor x_1_cast_fp16_to_fp32 = cast(dtype = x_1_cast_fp16_to_fp32_dtype_0, x = x_1_cast_fp16)[name = tensor("cast_689")]; tensor var_180 = pow(x = x_1_cast_fp16_to_fp32, y = var_170_promoted)[name = tensor("op_180")]; tensor var_1_axes_0 = const()[name = tensor("var_1_axes_0"), val = tensor([-1])]; tensor var_1_keep_dims_0 = const()[name = tensor("var_1_keep_dims_0"), val = tensor(true)]; tensor var_1 = reduce_mean(axes = var_1_axes_0, keep_dims = var_1_keep_dims_0, x = var_180)[name = tensor("var_1")]; tensor var_1_to_fp16_dtype_0 = const()[name = tensor("var_1_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_184_to_fp16 = const()[name = tensor("op_184_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_1_to_fp16 = cast(dtype = var_1_to_fp16_dtype_0, x = var_1)[name = tensor("cast_686")]; tensor var_185_cast_fp16 = add(x = var_1_to_fp16, y = var_184_to_fp16)[name = tensor("op_185_cast_fp16")]; tensor var_185_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_185_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_186_epsilon_0 = const()[name = tensor("op_186_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_185_cast_fp16_to_fp32 = cast(dtype = var_185_cast_fp16_to_fp32_dtype_0, x = var_185_cast_fp16)[name = tensor("cast_685")]; tensor var_186 = rsqrt(epsilon = var_186_epsilon_0, x = var_185_cast_fp16_to_fp32)[name = tensor("op_186")]; tensor var_186_to_fp16_dtype_0 = const()[name = tensor("op_186_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_186_to_fp16 = cast(dtype = var_186_to_fp16_dtype_0, x = var_186)[name = tensor("cast_684")]; tensor x_7_cast_fp16 = mul(x = x_1_cast_fp16, y = var_186_to_fp16)[name = tensor("x_7_cast_fp16")]; tensor layers_0_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_0_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(222454208)))]; tensor x_9_cast_fp16 = mul(x = layers_0_input_layernorm_weight_to_fp16, y = x_7_cast_fp16)[name = tensor("x_9_cast_fp16")]; tensor layers_0_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_0_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(222455296)))]; tensor linear_0_bias_0_to_fp16 = const()[name = tensor("linear_0_bias_0_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(224028224)))]; tensor linear_0_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_0_self_attn_q_proj_weight_to_fp16, x = x_9_cast_fp16)[name = tensor("linear_0_cast_fp16")]; tensor var_200 = const()[name = tensor("op_200"), val = tensor([1, 1, 12, 128])]; tensor x_11_cast_fp16 = reshape(shape = var_200, x = linear_0_cast_fp16)[name = tensor("x_11_cast_fp16")]; tensor x_11_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_11_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_170_promoted_1 = const()[name = tensor("op_170_promoted_1"), val = tensor(0x1p+1)]; tensor x_11_cast_fp16_to_fp32 = cast(dtype = x_11_cast_fp16_to_fp32_dtype_0, x = x_11_cast_fp16)[name = tensor("cast_683")]; tensor var_204 = pow(x = x_11_cast_fp16_to_fp32, y = var_170_promoted_1)[name = tensor("op_204")]; tensor var_3_axes_0 = const()[name = tensor("var_3_axes_0"), val = tensor([-1])]; tensor var_3_keep_dims_0 = const()[name = tensor("var_3_keep_dims_0"), val = tensor(true)]; tensor var_3 = reduce_mean(axes = var_3_axes_0, keep_dims = var_3_keep_dims_0, x = var_204)[name = tensor("var_3")]; tensor var_3_to_fp16_dtype_0 = const()[name = tensor("var_3_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_208_to_fp16 = const()[name = tensor("op_208_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_3_to_fp16 = cast(dtype = var_3_to_fp16_dtype_0, x = var_3)[name = tensor("cast_682")]; tensor var_209_cast_fp16 = add(x = var_3_to_fp16, y = var_208_to_fp16)[name = tensor("op_209_cast_fp16")]; tensor var_209_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_209_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_210_epsilon_0 = const()[name = tensor("op_210_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_209_cast_fp16_to_fp32 = cast(dtype = var_209_cast_fp16_to_fp32_dtype_0, x = var_209_cast_fp16)[name = tensor("cast_681")]; tensor var_210 = rsqrt(epsilon = var_210_epsilon_0, x = var_209_cast_fp16_to_fp32)[name = tensor("op_210")]; tensor var_210_to_fp16_dtype_0 = const()[name = tensor("op_210_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_210_to_fp16 = cast(dtype = var_210_to_fp16_dtype_0, x = var_210)[name = tensor("cast_680")]; tensor x_17_cast_fp16 = mul(x = x_11_cast_fp16, y = var_210_to_fp16)[name = tensor("x_17_cast_fp16")]; tensor layers_0_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_0_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(224031360)))]; tensor var_212_cast_fp16 = mul(x = layers_0_self_attn_q_norm_weight_to_fp16, y = x_17_cast_fp16)[name = tensor("op_212_cast_fp16")]; tensor q_1_perm_0 = const()[name = tensor("q_1_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_0_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_0_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(224031680)))]; tensor linear_1_bias_0_to_fp16 = const()[name = tensor("linear_1_bias_0_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(224556032)))]; tensor linear_1_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_0_self_attn_k_proj_weight_to_fp16, x = x_9_cast_fp16)[name = tensor("linear_1_cast_fp16")]; tensor var_216 = const()[name = tensor("op_216"), val = tensor([1, 1, 4, 128])]; tensor x_19_cast_fp16 = reshape(shape = var_216, x = linear_1_cast_fp16)[name = tensor("x_19_cast_fp16")]; tensor x_19_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_19_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_170_promoted_2 = const()[name = tensor("op_170_promoted_2"), val = tensor(0x1p+1)]; tensor x_19_cast_fp16_to_fp32 = cast(dtype = x_19_cast_fp16_to_fp32_dtype_0, x = x_19_cast_fp16)[name = tensor("cast_679")]; tensor var_220 = pow(x = x_19_cast_fp16_to_fp32, y = var_170_promoted_2)[name = tensor("op_220")]; tensor var_5_axes_0 = const()[name = tensor("var_5_axes_0"), val = tensor([-1])]; tensor var_5_keep_dims_0 = const()[name = tensor("var_5_keep_dims_0"), val = tensor(true)]; tensor var_5 = reduce_mean(axes = var_5_axes_0, keep_dims = var_5_keep_dims_0, x = var_220)[name = tensor("var_5")]; tensor var_5_to_fp16_dtype_0 = const()[name = tensor("var_5_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_224_to_fp16 = const()[name = tensor("op_224_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_5_to_fp16 = cast(dtype = var_5_to_fp16_dtype_0, x = var_5)[name = tensor("cast_678")]; tensor var_225_cast_fp16 = add(x = var_5_to_fp16, y = var_224_to_fp16)[name = tensor("op_225_cast_fp16")]; tensor var_225_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_225_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_226_epsilon_0 = const()[name = tensor("op_226_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_225_cast_fp16_to_fp32 = cast(dtype = var_225_cast_fp16_to_fp32_dtype_0, x = var_225_cast_fp16)[name = tensor("cast_677")]; tensor var_226 = rsqrt(epsilon = var_226_epsilon_0, x = var_225_cast_fp16_to_fp32)[name = tensor("op_226")]; tensor var_226_to_fp16_dtype_0 = const()[name = tensor("op_226_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_226_to_fp16 = cast(dtype = var_226_to_fp16_dtype_0, x = var_226)[name = tensor("cast_676")]; tensor x_25_cast_fp16 = mul(x = x_19_cast_fp16, y = var_226_to_fp16)[name = tensor("x_25_cast_fp16")]; tensor layers_0_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_0_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(224557120)))]; tensor var_228_cast_fp16 = mul(x = layers_0_self_attn_k_norm_weight_to_fp16, y = x_25_cast_fp16)[name = tensor("op_228_cast_fp16")]; tensor k_1_perm_0 = const()[name = tensor("k_1_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_0_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_0_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(224557440)))]; tensor linear_2_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_0_self_attn_v_proj_weight_to_fp16, x = x_9_cast_fp16)[name = tensor("linear_2_cast_fp16")]; tensor var_232 = const()[name = tensor("op_232"), val = tensor([1, 1, 4, 128])]; tensor var_233_cast_fp16 = reshape(shape = var_232, x = linear_2_cast_fp16)[name = tensor("op_233_cast_fp16")]; tensor v_1_perm_0 = const()[name = tensor("v_1_perm_0"), val = tensor([0, 2, 1, 3])]; tensor cos_3_axes_0 = const()[name = tensor("cos_3_axes_0"), val = tensor([1])]; tensor cos_3_cast_fp16 = expand_dims(axes = cos_3_axes_0, x = cos_1_cast_fp16)[name = tensor("cos_3_cast_fp16")]; tensor sin_3_axes_0 = const()[name = tensor("sin_3_axes_0"), val = tensor([1])]; tensor sin_3_cast_fp16 = expand_dims(axes = sin_3_axes_0, x = sin_1_cast_fp16)[name = tensor("sin_3_cast_fp16")]; tensor q_1_cast_fp16 = transpose(perm = q_1_perm_0, x = var_212_cast_fp16)[name = tensor("transpose_111")]; tensor var_237_cast_fp16 = mul(x = q_1_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_237_cast_fp16")]; tensor x1_1_begin_0 = const()[name = tensor("x1_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_1_end_0 = const()[name = tensor("x1_1_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_1_end_mask_0 = const()[name = tensor("x1_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_1_cast_fp16 = slice_by_index(begin = x1_1_begin_0, end = x1_1_end_0, end_mask = x1_1_end_mask_0, x = q_1_cast_fp16)[name = tensor("x1_1_cast_fp16")]; tensor x2_1_begin_0 = const()[name = tensor("x2_1_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_1_end_0 = const()[name = tensor("x2_1_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_1_end_mask_0 = const()[name = tensor("x2_1_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_1_cast_fp16 = slice_by_index(begin = x2_1_begin_0, end = x2_1_end_0, end_mask = x2_1_end_mask_0, x = q_1_cast_fp16)[name = tensor("x2_1_cast_fp16")]; tensor const_1_promoted_to_fp16 = const()[name = tensor("const_1_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_240_cast_fp16 = mul(x = x2_1_cast_fp16, y = const_1_promoted_to_fp16)[name = tensor("op_240_cast_fp16")]; tensor var_242_interleave_0 = const()[name = tensor("op_242_interleave_0"), val = tensor(false)]; tensor var_242_cast_fp16 = concat(axis = var_171, interleave = var_242_interleave_0, values = (var_240_cast_fp16, x1_1_cast_fp16))[name = tensor("op_242_cast_fp16")]; tensor var_243_cast_fp16 = mul(x = var_242_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_243_cast_fp16")]; tensor q_3_cast_fp16 = add(x = var_237_cast_fp16, y = var_243_cast_fp16)[name = tensor("q_3_cast_fp16")]; tensor k_1_cast_fp16 = transpose(perm = k_1_perm_0, x = var_228_cast_fp16)[name = tensor("transpose_110")]; tensor var_245_cast_fp16 = mul(x = k_1_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_245_cast_fp16")]; tensor x1_3_begin_0 = const()[name = tensor("x1_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_3_end_0 = const()[name = tensor("x1_3_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_3_end_mask_0 = const()[name = tensor("x1_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_3_cast_fp16 = slice_by_index(begin = x1_3_begin_0, end = x1_3_end_0, end_mask = x1_3_end_mask_0, x = k_1_cast_fp16)[name = tensor("x1_3_cast_fp16")]; tensor x2_3_begin_0 = const()[name = tensor("x2_3_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_3_end_0 = const()[name = tensor("x2_3_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_3_end_mask_0 = const()[name = tensor("x2_3_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_3_cast_fp16 = slice_by_index(begin = x2_3_begin_0, end = x2_3_end_0, end_mask = x2_3_end_mask_0, x = k_1_cast_fp16)[name = tensor("x2_3_cast_fp16")]; tensor const_2_promoted_to_fp16 = const()[name = tensor("const_2_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_248_cast_fp16 = mul(x = x2_3_cast_fp16, y = const_2_promoted_to_fp16)[name = tensor("op_248_cast_fp16")]; tensor var_250_interleave_0 = const()[name = tensor("op_250_interleave_0"), val = tensor(false)]; tensor var_250_cast_fp16 = concat(axis = var_171, interleave = var_250_interleave_0, values = (var_248_cast_fp16, x1_3_cast_fp16))[name = tensor("op_250_cast_fp16")]; tensor var_251_cast_fp16 = mul(x = var_250_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_251_cast_fp16")]; tensor k_3_cast_fp16 = add(x = var_245_cast_fp16, y = var_251_cast_fp16)[name = tensor("k_3_cast_fp16")]; tensor var_163_to_fp16 = const()[name = tensor("op_163_to_fp16"), val = tensor(0x1p+0)]; tensor update_mask_to_fp16_dtype_0 = const()[name = tensor("update_mask_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_107_to_fp16 = cast(dtype = update_mask_to_fp16_dtype_0, x = var_107)[name = tensor("cast_675")]; tensor var_253_cast_fp16 = sub(x = var_163_to_fp16, y = var_107_to_fp16)[name = tensor("op_253_cast_fp16")]; tensor var_254_cast_fp16 = mul(x = k_cache_1_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_254_cast_fp16")]; tensor var_255_cast_fp16 = mul(x = k_3_cast_fp16, y = var_107_to_fp16)[name = tensor("op_255_cast_fp16")]; tensor k_full_1_cast_fp16 = add(x = var_254_cast_fp16, y = var_255_cast_fp16)[name = tensor("k_full_1_cast_fp16")]; tensor var_258_cast_fp16 = mul(x = v_cache_1_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_258_cast_fp16")]; tensor v_1_cast_fp16 = transpose(perm = v_1_perm_0, x = var_233_cast_fp16)[name = tensor("transpose_109")]; tensor var_259_cast_fp16 = mul(x = v_1_cast_fp16, y = var_107_to_fp16)[name = tensor("op_259_cast_fp16")]; tensor v_full_1_cast_fp16 = add(x = var_258_cast_fp16, y = var_259_cast_fp16)[name = tensor("v_full_1_cast_fp16")]; tensor var_261_axes_0 = const()[name = tensor("op_261_axes_0"), val = tensor([2])]; tensor var_261_cast_fp16 = expand_dims(axes = var_261_axes_0, x = k_full_1_cast_fp16)[name = tensor("op_261_cast_fp16")]; tensor var_263_reps_0 = const()[name = tensor("op_263_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_263_cast_fp16 = tile(reps = var_263_reps_0, x = var_261_cast_fp16)[name = tensor("op_263_cast_fp16")]; tensor var_264 = const()[name = tensor("op_264"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_1_cast_fp16 = reshape(shape = var_264, x = var_263_cast_fp16)[name = tensor("k_rep_1_cast_fp16")]; tensor var_266_axes_0 = const()[name = tensor("op_266_axes_0"), val = tensor([2])]; tensor var_266_cast_fp16 = expand_dims(axes = var_266_axes_0, x = v_full_1_cast_fp16)[name = tensor("op_266_cast_fp16")]; tensor var_268_reps_0 = const()[name = tensor("op_268_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_268_cast_fp16 = tile(reps = var_268_reps_0, x = var_266_cast_fp16)[name = tensor("op_268_cast_fp16")]; tensor var_269 = const()[name = tensor("op_269"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_1_cast_fp16 = reshape(shape = var_269, x = var_268_cast_fp16)[name = tensor("v_rep_1_cast_fp16")]; tensor var_272_transpose_x_1 = const()[name = tensor("op_272_transpose_x_1"), val = tensor(false)]; tensor var_272_transpose_y_1 = const()[name = tensor("op_272_transpose_y_1"), val = tensor(true)]; tensor var_272_cast_fp16 = matmul(transpose_x = var_272_transpose_x_1, transpose_y = var_272_transpose_y_1, x = q_3_cast_fp16, y = k_rep_1_cast_fp16)[name = tensor("op_272_cast_fp16")]; tensor var_273_to_fp16 = const()[name = tensor("op_273_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_1_cast_fp16 = mul(x = var_272_cast_fp16, y = var_273_to_fp16)[name = tensor("attn_1_cast_fp16")]; tensor input_3_cast_fp16 = add(x = attn_1_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_3_cast_fp16")]; tensor input_3_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_3_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_3_cast_fp16_to_fp32 = cast(dtype = input_3_cast_fp16_to_fp32_dtype_0, x = input_3_cast_fp16)[name = tensor("cast_674")]; tensor attn_3 = softmax(axis = var_171, x = input_3_cast_fp16_to_fp32)[name = tensor("attn_3")]; tensor out_1_transpose_x_0 = const()[name = tensor("out_1_transpose_x_0"), val = tensor(false)]; tensor out_1_transpose_y_0 = const()[name = tensor("out_1_transpose_y_0"), val = tensor(false)]; tensor attn_3_to_fp16_dtype_0 = const()[name = tensor("attn_3_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_3_to_fp16 = cast(dtype = attn_3_to_fp16_dtype_0, x = attn_3)[name = tensor("cast_673")]; tensor out_1_cast_fp16 = matmul(transpose_x = out_1_transpose_x_0, transpose_y = out_1_transpose_y_0, x = attn_3_to_fp16, y = v_rep_1_cast_fp16)[name = tensor("out_1_cast_fp16")]; tensor var_278_perm_0 = const()[name = tensor("op_278_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_280 = const()[name = tensor("op_280"), val = tensor([1, 1, 1536])]; tensor var_278_cast_fp16 = transpose(perm = var_278_perm_0, x = out_1_cast_fp16)[name = tensor("transpose_108")]; tensor x_27_cast_fp16 = reshape(shape = var_280, x = var_278_cast_fp16)[name = tensor("x_27_cast_fp16")]; tensor layers_0_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_0_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(225081792)))]; tensor linear_3_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_0_self_attn_o_proj_weight_to_fp16, x = x_27_cast_fp16)[name = tensor("linear_3_cast_fp16")]; tensor x_29_cast_fp16 = add(x = x_1_cast_fp16, y = linear_3_cast_fp16)[name = tensor("x_29_cast_fp16")]; tensor x_29_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_29_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_170_promoted_3 = const()[name = tensor("op_170_promoted_3"), val = tensor(0x1p+1)]; tensor x_29_cast_fp16_to_fp32 = cast(dtype = x_29_cast_fp16_to_fp32_dtype_0, x = x_29_cast_fp16)[name = tensor("cast_672")]; tensor var_291 = pow(x = x_29_cast_fp16_to_fp32, y = var_170_promoted_3)[name = tensor("op_291")]; tensor var_7_axes_0 = const()[name = tensor("var_7_axes_0"), val = tensor([-1])]; tensor var_7_keep_dims_0 = const()[name = tensor("var_7_keep_dims_0"), val = tensor(true)]; tensor var_7 = reduce_mean(axes = var_7_axes_0, keep_dims = var_7_keep_dims_0, x = var_291)[name = tensor("var_7")]; tensor var_7_to_fp16_dtype_0 = const()[name = tensor("var_7_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_295_to_fp16 = const()[name = tensor("op_295_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_7_to_fp16 = cast(dtype = var_7_to_fp16_dtype_0, x = var_7)[name = tensor("cast_671")]; tensor var_296_cast_fp16 = add(x = var_7_to_fp16, y = var_295_to_fp16)[name = tensor("op_296_cast_fp16")]; tensor var_296_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_296_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_297_epsilon_0 = const()[name = tensor("op_297_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_296_cast_fp16_to_fp32 = cast(dtype = var_296_cast_fp16_to_fp32_dtype_0, x = var_296_cast_fp16)[name = tensor("cast_670")]; tensor var_297 = rsqrt(epsilon = var_297_epsilon_0, x = var_296_cast_fp16_to_fp32)[name = tensor("op_297")]; tensor var_297_to_fp16_dtype_0 = const()[name = tensor("op_297_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_297_to_fp16 = cast(dtype = var_297_to_fp16_dtype_0, x = var_297)[name = tensor("cast_669")]; tensor x_35_cast_fp16 = mul(x = x_29_cast_fp16, y = var_297_to_fp16)[name = tensor("x_35_cast_fp16")]; tensor layers_0_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_0_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(226654720)))]; tensor x_37_cast_fp16 = mul(x = layers_0_post_attention_layernorm_weight_to_fp16, y = x_35_cast_fp16)[name = tensor("x_37_cast_fp16")]; tensor layers_0_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_0_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(226655808)))]; tensor linear_4_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_0_mlp_gate_proj_weight_to_fp16, x = x_37_cast_fp16)[name = tensor("linear_4_cast_fp16")]; tensor var_308_cast_fp16 = silu(x = linear_4_cast_fp16)[name = tensor("op_308_cast_fp16")]; tensor layers_0_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_0_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(228228736)))]; tensor linear_5_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_0_mlp_up_proj_weight_to_fp16, x = x_37_cast_fp16)[name = tensor("linear_5_cast_fp16")]; tensor x_39_cast_fp16 = mul(x = var_308_cast_fp16, y = linear_5_cast_fp16)[name = tensor("x_39_cast_fp16")]; tensor layers_0_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_0_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(229801664)))]; tensor linear_6_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_0_mlp_down_proj_weight_to_fp16, x = x_39_cast_fp16)[name = tensor("linear_6_cast_fp16")]; tensor x_41_cast_fp16 = add(x = x_29_cast_fp16, y = linear_6_cast_fp16)[name = tensor("x_41_cast_fp16")]; tensor x_41_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_41_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_3_begin_0 = const()[name = tensor("k_cache_3_begin_0"), val = tensor([1, 0, 0, 0, 0])]; tensor k_cache_3_end_0 = const()[name = tensor("k_cache_3_end_0"), val = tensor([2, 1, 4, 2048, 128])]; tensor k_cache_3_end_mask_0 = const()[name = tensor("k_cache_3_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_3_squeeze_mask_0 = const()[name = tensor("k_cache_3_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_3_cast_fp16 = slice_by_index(begin = k_cache_3_begin_0, end = k_cache_3_end_0, end_mask = k_cache_3_end_mask_0, squeeze_mask = k_cache_3_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_3_cast_fp16")]; tensor v_cache_3_begin_0 = const()[name = tensor("v_cache_3_begin_0"), val = tensor([1, 0, 0, 0, 0])]; tensor v_cache_3_end_0 = const()[name = tensor("v_cache_3_end_0"), val = tensor([2, 1, 4, 2048, 128])]; tensor v_cache_3_end_mask_0 = const()[name = tensor("v_cache_3_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_3_squeeze_mask_0 = const()[name = tensor("v_cache_3_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_3_cast_fp16 = slice_by_index(begin = v_cache_3_begin_0, end = v_cache_3_end_0, end_mask = v_cache_3_end_mask_0, squeeze_mask = v_cache_3_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_3_cast_fp16")]; tensor var_340 = const()[name = tensor("op_340"), val = tensor(-1)]; tensor var_339_promoted = const()[name = tensor("op_339_promoted"), val = tensor(0x1p+1)]; tensor x_41_cast_fp16_to_fp32 = cast(dtype = x_41_cast_fp16_to_fp32_dtype_0, x = x_41_cast_fp16)[name = tensor("cast_668")]; tensor var_349 = pow(x = x_41_cast_fp16_to_fp32, y = var_339_promoted)[name = tensor("op_349")]; tensor var_9_axes_0 = const()[name = tensor("var_9_axes_0"), val = tensor([-1])]; tensor var_9_keep_dims_0 = const()[name = tensor("var_9_keep_dims_0"), val = tensor(true)]; tensor var_9 = reduce_mean(axes = var_9_axes_0, keep_dims = var_9_keep_dims_0, x = var_349)[name = tensor("var_9")]; tensor var_9_to_fp16_dtype_0 = const()[name = tensor("var_9_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_353_to_fp16 = const()[name = tensor("op_353_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_9_to_fp16 = cast(dtype = var_9_to_fp16_dtype_0, x = var_9)[name = tensor("cast_667")]; tensor var_354_cast_fp16 = add(x = var_9_to_fp16, y = var_353_to_fp16)[name = tensor("op_354_cast_fp16")]; tensor var_354_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_354_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_355_epsilon_0 = const()[name = tensor("op_355_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_354_cast_fp16_to_fp32 = cast(dtype = var_354_cast_fp16_to_fp32_dtype_0, x = var_354_cast_fp16)[name = tensor("cast_666")]; tensor var_355 = rsqrt(epsilon = var_355_epsilon_0, x = var_354_cast_fp16_to_fp32)[name = tensor("op_355")]; tensor var_355_to_fp16_dtype_0 = const()[name = tensor("op_355_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_355_to_fp16 = cast(dtype = var_355_to_fp16_dtype_0, x = var_355)[name = tensor("cast_665")]; tensor x_47_cast_fp16 = mul(x = x_41_cast_fp16, y = var_355_to_fp16)[name = tensor("x_47_cast_fp16")]; tensor layers_1_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_1_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(231374592)))]; tensor x_49_cast_fp16 = mul(x = layers_1_input_layernorm_weight_to_fp16, y = x_47_cast_fp16)[name = tensor("x_49_cast_fp16")]; tensor layers_1_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_1_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(231375680)))]; tensor linear_7_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_1_self_attn_q_proj_weight_to_fp16, x = x_49_cast_fp16)[name = tensor("linear_7_cast_fp16")]; tensor var_369 = const()[name = tensor("op_369"), val = tensor([1, 1, 12, 128])]; tensor x_51_cast_fp16 = reshape(shape = var_369, x = linear_7_cast_fp16)[name = tensor("x_51_cast_fp16")]; tensor x_51_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_51_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_339_promoted_1 = const()[name = tensor("op_339_promoted_1"), val = tensor(0x1p+1)]; tensor x_51_cast_fp16_to_fp32 = cast(dtype = x_51_cast_fp16_to_fp32_dtype_0, x = x_51_cast_fp16)[name = tensor("cast_664")]; tensor var_373 = pow(x = x_51_cast_fp16_to_fp32, y = var_339_promoted_1)[name = tensor("op_373")]; tensor var_11_axes_0 = const()[name = tensor("var_11_axes_0"), val = tensor([-1])]; tensor var_11_keep_dims_0 = const()[name = tensor("var_11_keep_dims_0"), val = tensor(true)]; tensor var_11 = reduce_mean(axes = var_11_axes_0, keep_dims = var_11_keep_dims_0, x = var_373)[name = tensor("var_11")]; tensor var_11_to_fp16_dtype_0 = const()[name = tensor("var_11_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_377_to_fp16 = const()[name = tensor("op_377_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_11_to_fp16 = cast(dtype = var_11_to_fp16_dtype_0, x = var_11)[name = tensor("cast_663")]; tensor var_378_cast_fp16 = add(x = var_11_to_fp16, y = var_377_to_fp16)[name = tensor("op_378_cast_fp16")]; tensor var_378_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_378_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_379_epsilon_0 = const()[name = tensor("op_379_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_378_cast_fp16_to_fp32 = cast(dtype = var_378_cast_fp16_to_fp32_dtype_0, x = var_378_cast_fp16)[name = tensor("cast_662")]; tensor var_379 = rsqrt(epsilon = var_379_epsilon_0, x = var_378_cast_fp16_to_fp32)[name = tensor("op_379")]; tensor var_379_to_fp16_dtype_0 = const()[name = tensor("op_379_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_379_to_fp16 = cast(dtype = var_379_to_fp16_dtype_0, x = var_379)[name = tensor("cast_661")]; tensor x_57_cast_fp16 = mul(x = x_51_cast_fp16, y = var_379_to_fp16)[name = tensor("x_57_cast_fp16")]; tensor layers_1_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_1_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(232948608)))]; tensor var_381_cast_fp16 = mul(x = layers_1_self_attn_q_norm_weight_to_fp16, y = x_57_cast_fp16)[name = tensor("op_381_cast_fp16")]; tensor q_5_perm_0 = const()[name = tensor("q_5_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_1_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_1_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(232948928)))]; tensor linear_8_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_1_self_attn_k_proj_weight_to_fp16, x = x_49_cast_fp16)[name = tensor("linear_8_cast_fp16")]; tensor var_385 = const()[name = tensor("op_385"), val = tensor([1, 1, 4, 128])]; tensor x_59_cast_fp16 = reshape(shape = var_385, x = linear_8_cast_fp16)[name = tensor("x_59_cast_fp16")]; tensor x_59_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_59_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_339_promoted_2 = const()[name = tensor("op_339_promoted_2"), val = tensor(0x1p+1)]; tensor x_59_cast_fp16_to_fp32 = cast(dtype = x_59_cast_fp16_to_fp32_dtype_0, x = x_59_cast_fp16)[name = tensor("cast_660")]; tensor var_389 = pow(x = x_59_cast_fp16_to_fp32, y = var_339_promoted_2)[name = tensor("op_389")]; tensor var_13_axes_0 = const()[name = tensor("var_13_axes_0"), val = tensor([-1])]; tensor var_13_keep_dims_0 = const()[name = tensor("var_13_keep_dims_0"), val = tensor(true)]; tensor var_13 = reduce_mean(axes = var_13_axes_0, keep_dims = var_13_keep_dims_0, x = var_389)[name = tensor("var_13")]; tensor var_13_to_fp16_dtype_0 = const()[name = tensor("var_13_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_393_to_fp16 = const()[name = tensor("op_393_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_13_to_fp16 = cast(dtype = var_13_to_fp16_dtype_0, x = var_13)[name = tensor("cast_659")]; tensor var_394_cast_fp16 = add(x = var_13_to_fp16, y = var_393_to_fp16)[name = tensor("op_394_cast_fp16")]; tensor var_394_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_394_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_395_epsilon_0 = const()[name = tensor("op_395_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_394_cast_fp16_to_fp32 = cast(dtype = var_394_cast_fp16_to_fp32_dtype_0, x = var_394_cast_fp16)[name = tensor("cast_658")]; tensor var_395 = rsqrt(epsilon = var_395_epsilon_0, x = var_394_cast_fp16_to_fp32)[name = tensor("op_395")]; tensor var_395_to_fp16_dtype_0 = const()[name = tensor("op_395_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_395_to_fp16 = cast(dtype = var_395_to_fp16_dtype_0, x = var_395)[name = tensor("cast_657")]; tensor x_65_cast_fp16 = mul(x = x_59_cast_fp16, y = var_395_to_fp16)[name = tensor("x_65_cast_fp16")]; tensor layers_1_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_1_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(233473280)))]; tensor var_397_cast_fp16 = mul(x = layers_1_self_attn_k_norm_weight_to_fp16, y = x_65_cast_fp16)[name = tensor("op_397_cast_fp16")]; tensor k_5_perm_0 = const()[name = tensor("k_5_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_1_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_1_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(233473600)))]; tensor linear_9_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_1_self_attn_v_proj_weight_to_fp16, x = x_49_cast_fp16)[name = tensor("linear_9_cast_fp16")]; tensor var_401 = const()[name = tensor("op_401"), val = tensor([1, 1, 4, 128])]; tensor var_402_cast_fp16 = reshape(shape = var_401, x = linear_9_cast_fp16)[name = tensor("op_402_cast_fp16")]; tensor v_3_perm_0 = const()[name = tensor("v_3_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_5_cast_fp16 = transpose(perm = q_5_perm_0, x = var_381_cast_fp16)[name = tensor("transpose_107")]; tensor var_406_cast_fp16 = mul(x = q_5_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_406_cast_fp16")]; tensor x1_5_begin_0 = const()[name = tensor("x1_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_5_end_0 = const()[name = tensor("x1_5_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_5_end_mask_0 = const()[name = tensor("x1_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_5_cast_fp16 = slice_by_index(begin = x1_5_begin_0, end = x1_5_end_0, end_mask = x1_5_end_mask_0, x = q_5_cast_fp16)[name = tensor("x1_5_cast_fp16")]; tensor x2_5_begin_0 = const()[name = tensor("x2_5_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_5_end_0 = const()[name = tensor("x2_5_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_5_end_mask_0 = const()[name = tensor("x2_5_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_5_cast_fp16 = slice_by_index(begin = x2_5_begin_0, end = x2_5_end_0, end_mask = x2_5_end_mask_0, x = q_5_cast_fp16)[name = tensor("x2_5_cast_fp16")]; tensor const_3_promoted_to_fp16 = const()[name = tensor("const_3_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_409_cast_fp16 = mul(x = x2_5_cast_fp16, y = const_3_promoted_to_fp16)[name = tensor("op_409_cast_fp16")]; tensor var_411_interleave_0 = const()[name = tensor("op_411_interleave_0"), val = tensor(false)]; tensor var_411_cast_fp16 = concat(axis = var_340, interleave = var_411_interleave_0, values = (var_409_cast_fp16, x1_5_cast_fp16))[name = tensor("op_411_cast_fp16")]; tensor var_412_cast_fp16 = mul(x = var_411_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_412_cast_fp16")]; tensor q_7_cast_fp16 = add(x = var_406_cast_fp16, y = var_412_cast_fp16)[name = tensor("q_7_cast_fp16")]; tensor k_5_cast_fp16 = transpose(perm = k_5_perm_0, x = var_397_cast_fp16)[name = tensor("transpose_106")]; tensor var_414_cast_fp16 = mul(x = k_5_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_414_cast_fp16")]; tensor x1_7_begin_0 = const()[name = tensor("x1_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_7_end_0 = const()[name = tensor("x1_7_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_7_end_mask_0 = const()[name = tensor("x1_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_7_cast_fp16 = slice_by_index(begin = x1_7_begin_0, end = x1_7_end_0, end_mask = x1_7_end_mask_0, x = k_5_cast_fp16)[name = tensor("x1_7_cast_fp16")]; tensor x2_7_begin_0 = const()[name = tensor("x2_7_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_7_end_0 = const()[name = tensor("x2_7_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_7_end_mask_0 = const()[name = tensor("x2_7_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_7_cast_fp16 = slice_by_index(begin = x2_7_begin_0, end = x2_7_end_0, end_mask = x2_7_end_mask_0, x = k_5_cast_fp16)[name = tensor("x2_7_cast_fp16")]; tensor const_4_promoted_to_fp16 = const()[name = tensor("const_4_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_417_cast_fp16 = mul(x = x2_7_cast_fp16, y = const_4_promoted_to_fp16)[name = tensor("op_417_cast_fp16")]; tensor var_419_interleave_0 = const()[name = tensor("op_419_interleave_0"), val = tensor(false)]; tensor var_419_cast_fp16 = concat(axis = var_340, interleave = var_419_interleave_0, values = (var_417_cast_fp16, x1_7_cast_fp16))[name = tensor("op_419_cast_fp16")]; tensor var_420_cast_fp16 = mul(x = var_419_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_420_cast_fp16")]; tensor k_7_cast_fp16 = add(x = var_414_cast_fp16, y = var_420_cast_fp16)[name = tensor("k_7_cast_fp16")]; tensor var_423_cast_fp16 = mul(x = k_cache_3_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_423_cast_fp16")]; tensor var_424_cast_fp16 = mul(x = k_7_cast_fp16, y = var_107_to_fp16)[name = tensor("op_424_cast_fp16")]; tensor k_full_3_cast_fp16 = add(x = var_423_cast_fp16, y = var_424_cast_fp16)[name = tensor("k_full_3_cast_fp16")]; tensor var_427_cast_fp16 = mul(x = v_cache_3_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_427_cast_fp16")]; tensor v_3_cast_fp16 = transpose(perm = v_3_perm_0, x = var_402_cast_fp16)[name = tensor("transpose_105")]; tensor var_428_cast_fp16 = mul(x = v_3_cast_fp16, y = var_107_to_fp16)[name = tensor("op_428_cast_fp16")]; tensor v_full_3_cast_fp16 = add(x = var_427_cast_fp16, y = var_428_cast_fp16)[name = tensor("v_full_3_cast_fp16")]; tensor var_430_axes_0 = const()[name = tensor("op_430_axes_0"), val = tensor([2])]; tensor var_430_cast_fp16 = expand_dims(axes = var_430_axes_0, x = k_full_3_cast_fp16)[name = tensor("op_430_cast_fp16")]; tensor var_432_reps_0 = const()[name = tensor("op_432_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_432_cast_fp16 = tile(reps = var_432_reps_0, x = var_430_cast_fp16)[name = tensor("op_432_cast_fp16")]; tensor var_433 = const()[name = tensor("op_433"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_3_cast_fp16 = reshape(shape = var_433, x = var_432_cast_fp16)[name = tensor("k_rep_3_cast_fp16")]; tensor var_435_axes_0 = const()[name = tensor("op_435_axes_0"), val = tensor([2])]; tensor var_435_cast_fp16 = expand_dims(axes = var_435_axes_0, x = v_full_3_cast_fp16)[name = tensor("op_435_cast_fp16")]; tensor var_437_reps_0 = const()[name = tensor("op_437_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_437_cast_fp16 = tile(reps = var_437_reps_0, x = var_435_cast_fp16)[name = tensor("op_437_cast_fp16")]; tensor var_438 = const()[name = tensor("op_438"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_3_cast_fp16 = reshape(shape = var_438, x = var_437_cast_fp16)[name = tensor("v_rep_3_cast_fp16")]; tensor var_441_transpose_x_1 = const()[name = tensor("op_441_transpose_x_1"), val = tensor(false)]; tensor var_441_transpose_y_1 = const()[name = tensor("op_441_transpose_y_1"), val = tensor(true)]; tensor var_441_cast_fp16 = matmul(transpose_x = var_441_transpose_x_1, transpose_y = var_441_transpose_y_1, x = q_7_cast_fp16, y = k_rep_3_cast_fp16)[name = tensor("op_441_cast_fp16")]; tensor var_442_to_fp16 = const()[name = tensor("op_442_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_5_cast_fp16 = mul(x = var_441_cast_fp16, y = var_442_to_fp16)[name = tensor("attn_5_cast_fp16")]; tensor input_7_cast_fp16 = add(x = attn_5_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_7_cast_fp16")]; tensor input_7_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_7_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_7_cast_fp16_to_fp32 = cast(dtype = input_7_cast_fp16_to_fp32_dtype_0, x = input_7_cast_fp16)[name = tensor("cast_656")]; tensor attn_7 = softmax(axis = var_340, x = input_7_cast_fp16_to_fp32)[name = tensor("attn_7")]; tensor out_3_transpose_x_0 = const()[name = tensor("out_3_transpose_x_0"), val = tensor(false)]; tensor out_3_transpose_y_0 = const()[name = tensor("out_3_transpose_y_0"), val = tensor(false)]; tensor attn_7_to_fp16_dtype_0 = const()[name = tensor("attn_7_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_7_to_fp16 = cast(dtype = attn_7_to_fp16_dtype_0, x = attn_7)[name = tensor("cast_655")]; tensor out_3_cast_fp16 = matmul(transpose_x = out_3_transpose_x_0, transpose_y = out_3_transpose_y_0, x = attn_7_to_fp16, y = v_rep_3_cast_fp16)[name = tensor("out_3_cast_fp16")]; tensor var_447_perm_0 = const()[name = tensor("op_447_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_449 = const()[name = tensor("op_449"), val = tensor([1, 1, 1536])]; tensor var_447_cast_fp16 = transpose(perm = var_447_perm_0, x = out_3_cast_fp16)[name = tensor("transpose_104")]; tensor x_67_cast_fp16 = reshape(shape = var_449, x = var_447_cast_fp16)[name = tensor("x_67_cast_fp16")]; tensor layers_1_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_1_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(233997952)))]; tensor linear_10_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_1_self_attn_o_proj_weight_to_fp16, x = x_67_cast_fp16)[name = tensor("linear_10_cast_fp16")]; tensor x_69_cast_fp16 = add(x = x_41_cast_fp16, y = linear_10_cast_fp16)[name = tensor("x_69_cast_fp16")]; tensor x_69_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_69_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_339_promoted_3 = const()[name = tensor("op_339_promoted_3"), val = tensor(0x1p+1)]; tensor x_69_cast_fp16_to_fp32 = cast(dtype = x_69_cast_fp16_to_fp32_dtype_0, x = x_69_cast_fp16)[name = tensor("cast_654")]; tensor var_460 = pow(x = x_69_cast_fp16_to_fp32, y = var_339_promoted_3)[name = tensor("op_460")]; tensor var_15_axes_0 = const()[name = tensor("var_15_axes_0"), val = tensor([-1])]; tensor var_15_keep_dims_0 = const()[name = tensor("var_15_keep_dims_0"), val = tensor(true)]; tensor var_15 = reduce_mean(axes = var_15_axes_0, keep_dims = var_15_keep_dims_0, x = var_460)[name = tensor("var_15")]; tensor var_15_to_fp16_dtype_0 = const()[name = tensor("var_15_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_464_to_fp16 = const()[name = tensor("op_464_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_15_to_fp16 = cast(dtype = var_15_to_fp16_dtype_0, x = var_15)[name = tensor("cast_653")]; tensor var_465_cast_fp16 = add(x = var_15_to_fp16, y = var_464_to_fp16)[name = tensor("op_465_cast_fp16")]; tensor var_465_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_465_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_466_epsilon_0 = const()[name = tensor("op_466_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_465_cast_fp16_to_fp32 = cast(dtype = var_465_cast_fp16_to_fp32_dtype_0, x = var_465_cast_fp16)[name = tensor("cast_652")]; tensor var_466 = rsqrt(epsilon = var_466_epsilon_0, x = var_465_cast_fp16_to_fp32)[name = tensor("op_466")]; tensor var_466_to_fp16_dtype_0 = const()[name = tensor("op_466_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_466_to_fp16 = cast(dtype = var_466_to_fp16_dtype_0, x = var_466)[name = tensor("cast_651")]; tensor x_75_cast_fp16 = mul(x = x_69_cast_fp16, y = var_466_to_fp16)[name = tensor("x_75_cast_fp16")]; tensor layers_1_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_1_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(235570880)))]; tensor x_77_cast_fp16 = mul(x = layers_1_post_attention_layernorm_weight_to_fp16, y = x_75_cast_fp16)[name = tensor("x_77_cast_fp16")]; tensor layers_1_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_1_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(235571968)))]; tensor linear_11_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_1_mlp_gate_proj_weight_to_fp16, x = x_77_cast_fp16)[name = tensor("linear_11_cast_fp16")]; tensor var_477_cast_fp16 = silu(x = linear_11_cast_fp16)[name = tensor("op_477_cast_fp16")]; tensor layers_1_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_1_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(237144896)))]; tensor linear_12_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_1_mlp_up_proj_weight_to_fp16, x = x_77_cast_fp16)[name = tensor("linear_12_cast_fp16")]; tensor x_79_cast_fp16 = mul(x = var_477_cast_fp16, y = linear_12_cast_fp16)[name = tensor("x_79_cast_fp16")]; tensor layers_1_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_1_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(238717824)))]; tensor linear_13_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_1_mlp_down_proj_weight_to_fp16, x = x_79_cast_fp16)[name = tensor("linear_13_cast_fp16")]; tensor x_81_cast_fp16 = add(x = x_69_cast_fp16, y = linear_13_cast_fp16)[name = tensor("x_81_cast_fp16")]; tensor x_81_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_81_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_5_begin_0 = const()[name = tensor("k_cache_5_begin_0"), val = tensor([2, 0, 0, 0, 0])]; tensor k_cache_5_end_0 = const()[name = tensor("k_cache_5_end_0"), val = tensor([3, 1, 4, 2048, 128])]; tensor k_cache_5_end_mask_0 = const()[name = tensor("k_cache_5_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_5_squeeze_mask_0 = const()[name = tensor("k_cache_5_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_5_cast_fp16 = slice_by_index(begin = k_cache_5_begin_0, end = k_cache_5_end_0, end_mask = k_cache_5_end_mask_0, squeeze_mask = k_cache_5_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_5_cast_fp16")]; tensor v_cache_5_begin_0 = const()[name = tensor("v_cache_5_begin_0"), val = tensor([2, 0, 0, 0, 0])]; tensor v_cache_5_end_0 = const()[name = tensor("v_cache_5_end_0"), val = tensor([3, 1, 4, 2048, 128])]; tensor v_cache_5_end_mask_0 = const()[name = tensor("v_cache_5_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_5_squeeze_mask_0 = const()[name = tensor("v_cache_5_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_5_cast_fp16 = slice_by_index(begin = v_cache_5_begin_0, end = v_cache_5_end_0, end_mask = v_cache_5_end_mask_0, squeeze_mask = v_cache_5_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_5_cast_fp16")]; tensor var_509 = const()[name = tensor("op_509"), val = tensor(-1)]; tensor var_508_promoted = const()[name = tensor("op_508_promoted"), val = tensor(0x1p+1)]; tensor x_81_cast_fp16_to_fp32 = cast(dtype = x_81_cast_fp16_to_fp32_dtype_0, x = x_81_cast_fp16)[name = tensor("cast_650")]; tensor var_518 = pow(x = x_81_cast_fp16_to_fp32, y = var_508_promoted)[name = tensor("op_518")]; tensor var_17_axes_0 = const()[name = tensor("var_17_axes_0"), val = tensor([-1])]; tensor var_17_keep_dims_0 = const()[name = tensor("var_17_keep_dims_0"), val = tensor(true)]; tensor var_17 = reduce_mean(axes = var_17_axes_0, keep_dims = var_17_keep_dims_0, x = var_518)[name = tensor("var_17")]; tensor var_17_to_fp16_dtype_0 = const()[name = tensor("var_17_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_522_to_fp16 = const()[name = tensor("op_522_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_17_to_fp16 = cast(dtype = var_17_to_fp16_dtype_0, x = var_17)[name = tensor("cast_649")]; tensor var_523_cast_fp16 = add(x = var_17_to_fp16, y = var_522_to_fp16)[name = tensor("op_523_cast_fp16")]; tensor var_523_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_523_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_524_epsilon_0 = const()[name = tensor("op_524_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_523_cast_fp16_to_fp32 = cast(dtype = var_523_cast_fp16_to_fp32_dtype_0, x = var_523_cast_fp16)[name = tensor("cast_648")]; tensor var_524 = rsqrt(epsilon = var_524_epsilon_0, x = var_523_cast_fp16_to_fp32)[name = tensor("op_524")]; tensor var_524_to_fp16_dtype_0 = const()[name = tensor("op_524_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_524_to_fp16 = cast(dtype = var_524_to_fp16_dtype_0, x = var_524)[name = tensor("cast_647")]; tensor x_87_cast_fp16 = mul(x = x_81_cast_fp16, y = var_524_to_fp16)[name = tensor("x_87_cast_fp16")]; tensor layers_2_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_2_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(240290752)))]; tensor x_89_cast_fp16 = mul(x = layers_2_input_layernorm_weight_to_fp16, y = x_87_cast_fp16)[name = tensor("x_89_cast_fp16")]; tensor layers_2_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_2_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(240291840)))]; tensor linear_14_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_2_self_attn_q_proj_weight_to_fp16, x = x_89_cast_fp16)[name = tensor("linear_14_cast_fp16")]; tensor var_538 = const()[name = tensor("op_538"), val = tensor([1, 1, 12, 128])]; tensor x_91_cast_fp16 = reshape(shape = var_538, x = linear_14_cast_fp16)[name = tensor("x_91_cast_fp16")]; tensor x_91_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_91_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_508_promoted_1 = const()[name = tensor("op_508_promoted_1"), val = tensor(0x1p+1)]; tensor x_91_cast_fp16_to_fp32 = cast(dtype = x_91_cast_fp16_to_fp32_dtype_0, x = x_91_cast_fp16)[name = tensor("cast_646")]; tensor var_542 = pow(x = x_91_cast_fp16_to_fp32, y = var_508_promoted_1)[name = tensor("op_542")]; tensor var_19_axes_0 = const()[name = tensor("var_19_axes_0"), val = tensor([-1])]; tensor var_19_keep_dims_0 = const()[name = tensor("var_19_keep_dims_0"), val = tensor(true)]; tensor var_19 = reduce_mean(axes = var_19_axes_0, keep_dims = var_19_keep_dims_0, x = var_542)[name = tensor("var_19")]; tensor var_19_to_fp16_dtype_0 = const()[name = tensor("var_19_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_546_to_fp16 = const()[name = tensor("op_546_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_19_to_fp16 = cast(dtype = var_19_to_fp16_dtype_0, x = var_19)[name = tensor("cast_645")]; tensor var_547_cast_fp16 = add(x = var_19_to_fp16, y = var_546_to_fp16)[name = tensor("op_547_cast_fp16")]; tensor var_547_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_547_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_548_epsilon_0 = const()[name = tensor("op_548_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_547_cast_fp16_to_fp32 = cast(dtype = var_547_cast_fp16_to_fp32_dtype_0, x = var_547_cast_fp16)[name = tensor("cast_644")]; tensor var_548 = rsqrt(epsilon = var_548_epsilon_0, x = var_547_cast_fp16_to_fp32)[name = tensor("op_548")]; tensor var_548_to_fp16_dtype_0 = const()[name = tensor("op_548_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_548_to_fp16 = cast(dtype = var_548_to_fp16_dtype_0, x = var_548)[name = tensor("cast_643")]; tensor x_97_cast_fp16 = mul(x = x_91_cast_fp16, y = var_548_to_fp16)[name = tensor("x_97_cast_fp16")]; tensor layers_2_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_2_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(241864768)))]; tensor var_550_cast_fp16 = mul(x = layers_2_self_attn_q_norm_weight_to_fp16, y = x_97_cast_fp16)[name = tensor("op_550_cast_fp16")]; tensor q_9_perm_0 = const()[name = tensor("q_9_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_2_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_2_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(241865088)))]; tensor linear_15_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_2_self_attn_k_proj_weight_to_fp16, x = x_89_cast_fp16)[name = tensor("linear_15_cast_fp16")]; tensor var_554 = const()[name = tensor("op_554"), val = tensor([1, 1, 4, 128])]; tensor x_99_cast_fp16 = reshape(shape = var_554, x = linear_15_cast_fp16)[name = tensor("x_99_cast_fp16")]; tensor x_99_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_99_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_508_promoted_2 = const()[name = tensor("op_508_promoted_2"), val = tensor(0x1p+1)]; tensor x_99_cast_fp16_to_fp32 = cast(dtype = x_99_cast_fp16_to_fp32_dtype_0, x = x_99_cast_fp16)[name = tensor("cast_642")]; tensor var_558 = pow(x = x_99_cast_fp16_to_fp32, y = var_508_promoted_2)[name = tensor("op_558")]; tensor var_21_axes_0 = const()[name = tensor("var_21_axes_0"), val = tensor([-1])]; tensor var_21_keep_dims_0 = const()[name = tensor("var_21_keep_dims_0"), val = tensor(true)]; tensor var_21 = reduce_mean(axes = var_21_axes_0, keep_dims = var_21_keep_dims_0, x = var_558)[name = tensor("var_21")]; tensor var_21_to_fp16_dtype_0 = const()[name = tensor("var_21_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_562_to_fp16 = const()[name = tensor("op_562_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_21_to_fp16 = cast(dtype = var_21_to_fp16_dtype_0, x = var_21)[name = tensor("cast_641")]; tensor var_563_cast_fp16 = add(x = var_21_to_fp16, y = var_562_to_fp16)[name = tensor("op_563_cast_fp16")]; tensor var_563_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_563_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_564_epsilon_0 = const()[name = tensor("op_564_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_563_cast_fp16_to_fp32 = cast(dtype = var_563_cast_fp16_to_fp32_dtype_0, x = var_563_cast_fp16)[name = tensor("cast_640")]; tensor var_564 = rsqrt(epsilon = var_564_epsilon_0, x = var_563_cast_fp16_to_fp32)[name = tensor("op_564")]; tensor var_564_to_fp16_dtype_0 = const()[name = tensor("op_564_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_564_to_fp16 = cast(dtype = var_564_to_fp16_dtype_0, x = var_564)[name = tensor("cast_639")]; tensor x_105_cast_fp16 = mul(x = x_99_cast_fp16, y = var_564_to_fp16)[name = tensor("x_105_cast_fp16")]; tensor layers_2_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_2_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(242389440)))]; tensor var_566_cast_fp16 = mul(x = layers_2_self_attn_k_norm_weight_to_fp16, y = x_105_cast_fp16)[name = tensor("op_566_cast_fp16")]; tensor k_9_perm_0 = const()[name = tensor("k_9_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_2_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_2_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(242389760)))]; tensor linear_16_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_2_self_attn_v_proj_weight_to_fp16, x = x_89_cast_fp16)[name = tensor("linear_16_cast_fp16")]; tensor var_570 = const()[name = tensor("op_570"), val = tensor([1, 1, 4, 128])]; tensor var_571_cast_fp16 = reshape(shape = var_570, x = linear_16_cast_fp16)[name = tensor("op_571_cast_fp16")]; tensor v_5_perm_0 = const()[name = tensor("v_5_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_9_cast_fp16 = transpose(perm = q_9_perm_0, x = var_550_cast_fp16)[name = tensor("transpose_103")]; tensor var_575_cast_fp16 = mul(x = q_9_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_575_cast_fp16")]; tensor x1_9_begin_0 = const()[name = tensor("x1_9_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_9_end_0 = const()[name = tensor("x1_9_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_9_end_mask_0 = const()[name = tensor("x1_9_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_9_cast_fp16 = slice_by_index(begin = x1_9_begin_0, end = x1_9_end_0, end_mask = x1_9_end_mask_0, x = q_9_cast_fp16)[name = tensor("x1_9_cast_fp16")]; tensor x2_9_begin_0 = const()[name = tensor("x2_9_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_9_end_0 = const()[name = tensor("x2_9_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_9_end_mask_0 = const()[name = tensor("x2_9_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_9_cast_fp16 = slice_by_index(begin = x2_9_begin_0, end = x2_9_end_0, end_mask = x2_9_end_mask_0, x = q_9_cast_fp16)[name = tensor("x2_9_cast_fp16")]; tensor const_5_promoted_to_fp16 = const()[name = tensor("const_5_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_578_cast_fp16 = mul(x = x2_9_cast_fp16, y = const_5_promoted_to_fp16)[name = tensor("op_578_cast_fp16")]; tensor var_580_interleave_0 = const()[name = tensor("op_580_interleave_0"), val = tensor(false)]; tensor var_580_cast_fp16 = concat(axis = var_509, interleave = var_580_interleave_0, values = (var_578_cast_fp16, x1_9_cast_fp16))[name = tensor("op_580_cast_fp16")]; tensor var_581_cast_fp16 = mul(x = var_580_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_581_cast_fp16")]; tensor q_11_cast_fp16 = add(x = var_575_cast_fp16, y = var_581_cast_fp16)[name = tensor("q_11_cast_fp16")]; tensor k_9_cast_fp16 = transpose(perm = k_9_perm_0, x = var_566_cast_fp16)[name = tensor("transpose_102")]; tensor var_583_cast_fp16 = mul(x = k_9_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_583_cast_fp16")]; tensor x1_11_begin_0 = const()[name = tensor("x1_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_11_end_0 = const()[name = tensor("x1_11_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_11_end_mask_0 = const()[name = tensor("x1_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_11_cast_fp16 = slice_by_index(begin = x1_11_begin_0, end = x1_11_end_0, end_mask = x1_11_end_mask_0, x = k_9_cast_fp16)[name = tensor("x1_11_cast_fp16")]; tensor x2_11_begin_0 = const()[name = tensor("x2_11_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_11_end_0 = const()[name = tensor("x2_11_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_11_end_mask_0 = const()[name = tensor("x2_11_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_11_cast_fp16 = slice_by_index(begin = x2_11_begin_0, end = x2_11_end_0, end_mask = x2_11_end_mask_0, x = k_9_cast_fp16)[name = tensor("x2_11_cast_fp16")]; tensor const_6_promoted_to_fp16 = const()[name = tensor("const_6_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_586_cast_fp16 = mul(x = x2_11_cast_fp16, y = const_6_promoted_to_fp16)[name = tensor("op_586_cast_fp16")]; tensor var_588_interleave_0 = const()[name = tensor("op_588_interleave_0"), val = tensor(false)]; tensor var_588_cast_fp16 = concat(axis = var_509, interleave = var_588_interleave_0, values = (var_586_cast_fp16, x1_11_cast_fp16))[name = tensor("op_588_cast_fp16")]; tensor var_589_cast_fp16 = mul(x = var_588_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_589_cast_fp16")]; tensor k_11_cast_fp16 = add(x = var_583_cast_fp16, y = var_589_cast_fp16)[name = tensor("k_11_cast_fp16")]; tensor var_592_cast_fp16 = mul(x = k_cache_5_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_592_cast_fp16")]; tensor var_593_cast_fp16 = mul(x = k_11_cast_fp16, y = var_107_to_fp16)[name = tensor("op_593_cast_fp16")]; tensor k_full_5_cast_fp16 = add(x = var_592_cast_fp16, y = var_593_cast_fp16)[name = tensor("k_full_5_cast_fp16")]; tensor var_596_cast_fp16 = mul(x = v_cache_5_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_596_cast_fp16")]; tensor v_5_cast_fp16 = transpose(perm = v_5_perm_0, x = var_571_cast_fp16)[name = tensor("transpose_101")]; tensor var_597_cast_fp16 = mul(x = v_5_cast_fp16, y = var_107_to_fp16)[name = tensor("op_597_cast_fp16")]; tensor v_full_5_cast_fp16 = add(x = var_596_cast_fp16, y = var_597_cast_fp16)[name = tensor("v_full_5_cast_fp16")]; tensor var_599_axes_0 = const()[name = tensor("op_599_axes_0"), val = tensor([2])]; tensor var_599_cast_fp16 = expand_dims(axes = var_599_axes_0, x = k_full_5_cast_fp16)[name = tensor("op_599_cast_fp16")]; tensor var_601_reps_0 = const()[name = tensor("op_601_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_601_cast_fp16 = tile(reps = var_601_reps_0, x = var_599_cast_fp16)[name = tensor("op_601_cast_fp16")]; tensor var_602 = const()[name = tensor("op_602"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_5_cast_fp16 = reshape(shape = var_602, x = var_601_cast_fp16)[name = tensor("k_rep_5_cast_fp16")]; tensor var_604_axes_0 = const()[name = tensor("op_604_axes_0"), val = tensor([2])]; tensor var_604_cast_fp16 = expand_dims(axes = var_604_axes_0, x = v_full_5_cast_fp16)[name = tensor("op_604_cast_fp16")]; tensor var_606_reps_0 = const()[name = tensor("op_606_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_606_cast_fp16 = tile(reps = var_606_reps_0, x = var_604_cast_fp16)[name = tensor("op_606_cast_fp16")]; tensor var_607 = const()[name = tensor("op_607"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_5_cast_fp16 = reshape(shape = var_607, x = var_606_cast_fp16)[name = tensor("v_rep_5_cast_fp16")]; tensor var_610_transpose_x_1 = const()[name = tensor("op_610_transpose_x_1"), val = tensor(false)]; tensor var_610_transpose_y_1 = const()[name = tensor("op_610_transpose_y_1"), val = tensor(true)]; tensor var_610_cast_fp16 = matmul(transpose_x = var_610_transpose_x_1, transpose_y = var_610_transpose_y_1, x = q_11_cast_fp16, y = k_rep_5_cast_fp16)[name = tensor("op_610_cast_fp16")]; tensor var_611_to_fp16 = const()[name = tensor("op_611_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_9_cast_fp16 = mul(x = var_610_cast_fp16, y = var_611_to_fp16)[name = tensor("attn_9_cast_fp16")]; tensor input_11_cast_fp16 = add(x = attn_9_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_11_cast_fp16")]; tensor input_11_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_11_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_11_cast_fp16_to_fp32 = cast(dtype = input_11_cast_fp16_to_fp32_dtype_0, x = input_11_cast_fp16)[name = tensor("cast_638")]; tensor attn_11 = softmax(axis = var_509, x = input_11_cast_fp16_to_fp32)[name = tensor("attn_11")]; tensor out_5_transpose_x_0 = const()[name = tensor("out_5_transpose_x_0"), val = tensor(false)]; tensor out_5_transpose_y_0 = const()[name = tensor("out_5_transpose_y_0"), val = tensor(false)]; tensor attn_11_to_fp16_dtype_0 = const()[name = tensor("attn_11_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_11_to_fp16 = cast(dtype = attn_11_to_fp16_dtype_0, x = attn_11)[name = tensor("cast_637")]; tensor out_5_cast_fp16 = matmul(transpose_x = out_5_transpose_x_0, transpose_y = out_5_transpose_y_0, x = attn_11_to_fp16, y = v_rep_5_cast_fp16)[name = tensor("out_5_cast_fp16")]; tensor var_616_perm_0 = const()[name = tensor("op_616_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_618 = const()[name = tensor("op_618"), val = tensor([1, 1, 1536])]; tensor var_616_cast_fp16 = transpose(perm = var_616_perm_0, x = out_5_cast_fp16)[name = tensor("transpose_100")]; tensor x_107_cast_fp16 = reshape(shape = var_618, x = var_616_cast_fp16)[name = tensor("x_107_cast_fp16")]; tensor layers_2_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_2_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(242914112)))]; tensor linear_17_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_2_self_attn_o_proj_weight_to_fp16, x = x_107_cast_fp16)[name = tensor("linear_17_cast_fp16")]; tensor x_109_cast_fp16 = add(x = x_81_cast_fp16, y = linear_17_cast_fp16)[name = tensor("x_109_cast_fp16")]; tensor x_109_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_109_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_508_promoted_3 = const()[name = tensor("op_508_promoted_3"), val = tensor(0x1p+1)]; tensor x_109_cast_fp16_to_fp32 = cast(dtype = x_109_cast_fp16_to_fp32_dtype_0, x = x_109_cast_fp16)[name = tensor("cast_636")]; tensor var_629 = pow(x = x_109_cast_fp16_to_fp32, y = var_508_promoted_3)[name = tensor("op_629")]; tensor var_23_axes_0 = const()[name = tensor("var_23_axes_0"), val = tensor([-1])]; tensor var_23_keep_dims_0 = const()[name = tensor("var_23_keep_dims_0"), val = tensor(true)]; tensor var_23 = reduce_mean(axes = var_23_axes_0, keep_dims = var_23_keep_dims_0, x = var_629)[name = tensor("var_23")]; tensor var_23_to_fp16_dtype_0 = const()[name = tensor("var_23_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_633_to_fp16 = const()[name = tensor("op_633_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_23_to_fp16 = cast(dtype = var_23_to_fp16_dtype_0, x = var_23)[name = tensor("cast_635")]; tensor var_634_cast_fp16 = add(x = var_23_to_fp16, y = var_633_to_fp16)[name = tensor("op_634_cast_fp16")]; tensor var_634_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_634_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_635_epsilon_0 = const()[name = tensor("op_635_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_634_cast_fp16_to_fp32 = cast(dtype = var_634_cast_fp16_to_fp32_dtype_0, x = var_634_cast_fp16)[name = tensor("cast_634")]; tensor var_635 = rsqrt(epsilon = var_635_epsilon_0, x = var_634_cast_fp16_to_fp32)[name = tensor("op_635")]; tensor var_635_to_fp16_dtype_0 = const()[name = tensor("op_635_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_635_to_fp16 = cast(dtype = var_635_to_fp16_dtype_0, x = var_635)[name = tensor("cast_633")]; tensor x_115_cast_fp16 = mul(x = x_109_cast_fp16, y = var_635_to_fp16)[name = tensor("x_115_cast_fp16")]; tensor layers_2_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_2_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(244487040)))]; tensor x_117_cast_fp16 = mul(x = layers_2_post_attention_layernorm_weight_to_fp16, y = x_115_cast_fp16)[name = tensor("x_117_cast_fp16")]; tensor layers_2_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_2_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(244488128)))]; tensor linear_18_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_2_mlp_gate_proj_weight_to_fp16, x = x_117_cast_fp16)[name = tensor("linear_18_cast_fp16")]; tensor var_646_cast_fp16 = silu(x = linear_18_cast_fp16)[name = tensor("op_646_cast_fp16")]; tensor layers_2_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_2_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(246061056)))]; tensor linear_19_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_2_mlp_up_proj_weight_to_fp16, x = x_117_cast_fp16)[name = tensor("linear_19_cast_fp16")]; tensor x_119_cast_fp16 = mul(x = var_646_cast_fp16, y = linear_19_cast_fp16)[name = tensor("x_119_cast_fp16")]; tensor layers_2_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_2_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(247633984)))]; tensor linear_20_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_2_mlp_down_proj_weight_to_fp16, x = x_119_cast_fp16)[name = tensor("linear_20_cast_fp16")]; tensor x_121_cast_fp16 = add(x = x_109_cast_fp16, y = linear_20_cast_fp16)[name = tensor("x_121_cast_fp16")]; tensor x_121_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_121_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_7_begin_0 = const()[name = tensor("k_cache_7_begin_0"), val = tensor([3, 0, 0, 0, 0])]; tensor k_cache_7_end_0 = const()[name = tensor("k_cache_7_end_0"), val = tensor([4, 1, 4, 2048, 128])]; tensor k_cache_7_end_mask_0 = const()[name = tensor("k_cache_7_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_7_squeeze_mask_0 = const()[name = tensor("k_cache_7_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_7_cast_fp16 = slice_by_index(begin = k_cache_7_begin_0, end = k_cache_7_end_0, end_mask = k_cache_7_end_mask_0, squeeze_mask = k_cache_7_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_7_cast_fp16")]; tensor v_cache_7_begin_0 = const()[name = tensor("v_cache_7_begin_0"), val = tensor([3, 0, 0, 0, 0])]; tensor v_cache_7_end_0 = const()[name = tensor("v_cache_7_end_0"), val = tensor([4, 1, 4, 2048, 128])]; tensor v_cache_7_end_mask_0 = const()[name = tensor("v_cache_7_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_7_squeeze_mask_0 = const()[name = tensor("v_cache_7_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_7_cast_fp16 = slice_by_index(begin = v_cache_7_begin_0, end = v_cache_7_end_0, end_mask = v_cache_7_end_mask_0, squeeze_mask = v_cache_7_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_7_cast_fp16")]; tensor var_678 = const()[name = tensor("op_678"), val = tensor(-1)]; tensor var_677_promoted = const()[name = tensor("op_677_promoted"), val = tensor(0x1p+1)]; tensor x_121_cast_fp16_to_fp32 = cast(dtype = x_121_cast_fp16_to_fp32_dtype_0, x = x_121_cast_fp16)[name = tensor("cast_632")]; tensor var_687 = pow(x = x_121_cast_fp16_to_fp32, y = var_677_promoted)[name = tensor("op_687")]; tensor var_25_axes_0 = const()[name = tensor("var_25_axes_0"), val = tensor([-1])]; tensor var_25_keep_dims_0 = const()[name = tensor("var_25_keep_dims_0"), val = tensor(true)]; tensor var_25 = reduce_mean(axes = var_25_axes_0, keep_dims = var_25_keep_dims_0, x = var_687)[name = tensor("var_25")]; tensor var_25_to_fp16_dtype_0 = const()[name = tensor("var_25_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_691_to_fp16 = const()[name = tensor("op_691_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_25_to_fp16 = cast(dtype = var_25_to_fp16_dtype_0, x = var_25)[name = tensor("cast_631")]; tensor var_692_cast_fp16 = add(x = var_25_to_fp16, y = var_691_to_fp16)[name = tensor("op_692_cast_fp16")]; tensor var_692_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_692_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_693_epsilon_0 = const()[name = tensor("op_693_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_692_cast_fp16_to_fp32 = cast(dtype = var_692_cast_fp16_to_fp32_dtype_0, x = var_692_cast_fp16)[name = tensor("cast_630")]; tensor var_693 = rsqrt(epsilon = var_693_epsilon_0, x = var_692_cast_fp16_to_fp32)[name = tensor("op_693")]; tensor var_693_to_fp16_dtype_0 = const()[name = tensor("op_693_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_693_to_fp16 = cast(dtype = var_693_to_fp16_dtype_0, x = var_693)[name = tensor("cast_629")]; tensor x_127_cast_fp16 = mul(x = x_121_cast_fp16, y = var_693_to_fp16)[name = tensor("x_127_cast_fp16")]; tensor layers_3_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_3_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(249206912)))]; tensor x_129_cast_fp16 = mul(x = layers_3_input_layernorm_weight_to_fp16, y = x_127_cast_fp16)[name = tensor("x_129_cast_fp16")]; tensor layers_3_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_3_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(249208000)))]; tensor linear_21_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_3_self_attn_q_proj_weight_to_fp16, x = x_129_cast_fp16)[name = tensor("linear_21_cast_fp16")]; tensor var_707 = const()[name = tensor("op_707"), val = tensor([1, 1, 12, 128])]; tensor x_131_cast_fp16 = reshape(shape = var_707, x = linear_21_cast_fp16)[name = tensor("x_131_cast_fp16")]; tensor x_131_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_131_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_677_promoted_1 = const()[name = tensor("op_677_promoted_1"), val = tensor(0x1p+1)]; tensor x_131_cast_fp16_to_fp32 = cast(dtype = x_131_cast_fp16_to_fp32_dtype_0, x = x_131_cast_fp16)[name = tensor("cast_628")]; tensor var_711 = pow(x = x_131_cast_fp16_to_fp32, y = var_677_promoted_1)[name = tensor("op_711")]; tensor var_27_axes_0 = const()[name = tensor("var_27_axes_0"), val = tensor([-1])]; tensor var_27_keep_dims_0 = const()[name = tensor("var_27_keep_dims_0"), val = tensor(true)]; tensor var_27 = reduce_mean(axes = var_27_axes_0, keep_dims = var_27_keep_dims_0, x = var_711)[name = tensor("var_27")]; tensor var_27_to_fp16_dtype_0 = const()[name = tensor("var_27_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_715_to_fp16 = const()[name = tensor("op_715_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_27_to_fp16 = cast(dtype = var_27_to_fp16_dtype_0, x = var_27)[name = tensor("cast_627")]; tensor var_716_cast_fp16 = add(x = var_27_to_fp16, y = var_715_to_fp16)[name = tensor("op_716_cast_fp16")]; tensor var_716_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_716_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_717_epsilon_0 = const()[name = tensor("op_717_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_716_cast_fp16_to_fp32 = cast(dtype = var_716_cast_fp16_to_fp32_dtype_0, x = var_716_cast_fp16)[name = tensor("cast_626")]; tensor var_717 = rsqrt(epsilon = var_717_epsilon_0, x = var_716_cast_fp16_to_fp32)[name = tensor("op_717")]; tensor var_717_to_fp16_dtype_0 = const()[name = tensor("op_717_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_717_to_fp16 = cast(dtype = var_717_to_fp16_dtype_0, x = var_717)[name = tensor("cast_625")]; tensor x_137_cast_fp16 = mul(x = x_131_cast_fp16, y = var_717_to_fp16)[name = tensor("x_137_cast_fp16")]; tensor layers_3_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_3_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(250780928)))]; tensor var_719_cast_fp16 = mul(x = layers_3_self_attn_q_norm_weight_to_fp16, y = x_137_cast_fp16)[name = tensor("op_719_cast_fp16")]; tensor q_13_perm_0 = const()[name = tensor("q_13_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_3_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_3_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(250781248)))]; tensor linear_22_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_3_self_attn_k_proj_weight_to_fp16, x = x_129_cast_fp16)[name = tensor("linear_22_cast_fp16")]; tensor var_723 = const()[name = tensor("op_723"), val = tensor([1, 1, 4, 128])]; tensor x_139_cast_fp16 = reshape(shape = var_723, x = linear_22_cast_fp16)[name = tensor("x_139_cast_fp16")]; tensor x_139_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_139_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_677_promoted_2 = const()[name = tensor("op_677_promoted_2"), val = tensor(0x1p+1)]; tensor x_139_cast_fp16_to_fp32 = cast(dtype = x_139_cast_fp16_to_fp32_dtype_0, x = x_139_cast_fp16)[name = tensor("cast_624")]; tensor var_727 = pow(x = x_139_cast_fp16_to_fp32, y = var_677_promoted_2)[name = tensor("op_727")]; tensor var_29_axes_0 = const()[name = tensor("var_29_axes_0"), val = tensor([-1])]; tensor var_29_keep_dims_0 = const()[name = tensor("var_29_keep_dims_0"), val = tensor(true)]; tensor var_29 = reduce_mean(axes = var_29_axes_0, keep_dims = var_29_keep_dims_0, x = var_727)[name = tensor("var_29")]; tensor var_29_to_fp16_dtype_0 = const()[name = tensor("var_29_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_731_to_fp16 = const()[name = tensor("op_731_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_29_to_fp16 = cast(dtype = var_29_to_fp16_dtype_0, x = var_29)[name = tensor("cast_623")]; tensor var_732_cast_fp16 = add(x = var_29_to_fp16, y = var_731_to_fp16)[name = tensor("op_732_cast_fp16")]; tensor var_732_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_732_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_733_epsilon_0 = const()[name = tensor("op_733_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_732_cast_fp16_to_fp32 = cast(dtype = var_732_cast_fp16_to_fp32_dtype_0, x = var_732_cast_fp16)[name = tensor("cast_622")]; tensor var_733 = rsqrt(epsilon = var_733_epsilon_0, x = var_732_cast_fp16_to_fp32)[name = tensor("op_733")]; tensor var_733_to_fp16_dtype_0 = const()[name = tensor("op_733_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_733_to_fp16 = cast(dtype = var_733_to_fp16_dtype_0, x = var_733)[name = tensor("cast_621")]; tensor x_145_cast_fp16 = mul(x = x_139_cast_fp16, y = var_733_to_fp16)[name = tensor("x_145_cast_fp16")]; tensor layers_3_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_3_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(251305600)))]; tensor var_735_cast_fp16 = mul(x = layers_3_self_attn_k_norm_weight_to_fp16, y = x_145_cast_fp16)[name = tensor("op_735_cast_fp16")]; tensor k_13_perm_0 = const()[name = tensor("k_13_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_3_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_3_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(251305920)))]; tensor linear_23_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_3_self_attn_v_proj_weight_to_fp16, x = x_129_cast_fp16)[name = tensor("linear_23_cast_fp16")]; tensor var_739 = const()[name = tensor("op_739"), val = tensor([1, 1, 4, 128])]; tensor var_740_cast_fp16 = reshape(shape = var_739, x = linear_23_cast_fp16)[name = tensor("op_740_cast_fp16")]; tensor v_7_perm_0 = const()[name = tensor("v_7_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_13_cast_fp16 = transpose(perm = q_13_perm_0, x = var_719_cast_fp16)[name = tensor("transpose_99")]; tensor var_744_cast_fp16 = mul(x = q_13_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_744_cast_fp16")]; tensor x1_13_begin_0 = const()[name = tensor("x1_13_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_13_end_0 = const()[name = tensor("x1_13_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_13_end_mask_0 = const()[name = tensor("x1_13_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_13_cast_fp16 = slice_by_index(begin = x1_13_begin_0, end = x1_13_end_0, end_mask = x1_13_end_mask_0, x = q_13_cast_fp16)[name = tensor("x1_13_cast_fp16")]; tensor x2_13_begin_0 = const()[name = tensor("x2_13_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_13_end_0 = const()[name = tensor("x2_13_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_13_end_mask_0 = const()[name = tensor("x2_13_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_13_cast_fp16 = slice_by_index(begin = x2_13_begin_0, end = x2_13_end_0, end_mask = x2_13_end_mask_0, x = q_13_cast_fp16)[name = tensor("x2_13_cast_fp16")]; tensor const_7_promoted_to_fp16 = const()[name = tensor("const_7_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_747_cast_fp16 = mul(x = x2_13_cast_fp16, y = const_7_promoted_to_fp16)[name = tensor("op_747_cast_fp16")]; tensor var_749_interleave_0 = const()[name = tensor("op_749_interleave_0"), val = tensor(false)]; tensor var_749_cast_fp16 = concat(axis = var_678, interleave = var_749_interleave_0, values = (var_747_cast_fp16, x1_13_cast_fp16))[name = tensor("op_749_cast_fp16")]; tensor var_750_cast_fp16 = mul(x = var_749_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_750_cast_fp16")]; tensor q_15_cast_fp16 = add(x = var_744_cast_fp16, y = var_750_cast_fp16)[name = tensor("q_15_cast_fp16")]; tensor k_13_cast_fp16 = transpose(perm = k_13_perm_0, x = var_735_cast_fp16)[name = tensor("transpose_98")]; tensor var_752_cast_fp16 = mul(x = k_13_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_752_cast_fp16")]; tensor x1_15_begin_0 = const()[name = tensor("x1_15_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_15_end_0 = const()[name = tensor("x1_15_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_15_end_mask_0 = const()[name = tensor("x1_15_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_15_cast_fp16 = slice_by_index(begin = x1_15_begin_0, end = x1_15_end_0, end_mask = x1_15_end_mask_0, x = k_13_cast_fp16)[name = tensor("x1_15_cast_fp16")]; tensor x2_15_begin_0 = const()[name = tensor("x2_15_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_15_end_0 = const()[name = tensor("x2_15_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_15_end_mask_0 = const()[name = tensor("x2_15_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_15_cast_fp16 = slice_by_index(begin = x2_15_begin_0, end = x2_15_end_0, end_mask = x2_15_end_mask_0, x = k_13_cast_fp16)[name = tensor("x2_15_cast_fp16")]; tensor const_8_promoted_to_fp16 = const()[name = tensor("const_8_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_755_cast_fp16 = mul(x = x2_15_cast_fp16, y = const_8_promoted_to_fp16)[name = tensor("op_755_cast_fp16")]; tensor var_757_interleave_0 = const()[name = tensor("op_757_interleave_0"), val = tensor(false)]; tensor var_757_cast_fp16 = concat(axis = var_678, interleave = var_757_interleave_0, values = (var_755_cast_fp16, x1_15_cast_fp16))[name = tensor("op_757_cast_fp16")]; tensor var_758_cast_fp16 = mul(x = var_757_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_758_cast_fp16")]; tensor k_15_cast_fp16 = add(x = var_752_cast_fp16, y = var_758_cast_fp16)[name = tensor("k_15_cast_fp16")]; tensor var_761_cast_fp16 = mul(x = k_cache_7_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_761_cast_fp16")]; tensor var_762_cast_fp16 = mul(x = k_15_cast_fp16, y = var_107_to_fp16)[name = tensor("op_762_cast_fp16")]; tensor k_full_7_cast_fp16 = add(x = var_761_cast_fp16, y = var_762_cast_fp16)[name = tensor("k_full_7_cast_fp16")]; tensor var_765_cast_fp16 = mul(x = v_cache_7_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_765_cast_fp16")]; tensor v_7_cast_fp16 = transpose(perm = v_7_perm_0, x = var_740_cast_fp16)[name = tensor("transpose_97")]; tensor var_766_cast_fp16 = mul(x = v_7_cast_fp16, y = var_107_to_fp16)[name = tensor("op_766_cast_fp16")]; tensor v_full_7_cast_fp16 = add(x = var_765_cast_fp16, y = var_766_cast_fp16)[name = tensor("v_full_7_cast_fp16")]; tensor var_768_axes_0 = const()[name = tensor("op_768_axes_0"), val = tensor([2])]; tensor var_768_cast_fp16 = expand_dims(axes = var_768_axes_0, x = k_full_7_cast_fp16)[name = tensor("op_768_cast_fp16")]; tensor var_770_reps_0 = const()[name = tensor("op_770_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_770_cast_fp16 = tile(reps = var_770_reps_0, x = var_768_cast_fp16)[name = tensor("op_770_cast_fp16")]; tensor var_771 = const()[name = tensor("op_771"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_7_cast_fp16 = reshape(shape = var_771, x = var_770_cast_fp16)[name = tensor("k_rep_7_cast_fp16")]; tensor var_773_axes_0 = const()[name = tensor("op_773_axes_0"), val = tensor([2])]; tensor var_773_cast_fp16 = expand_dims(axes = var_773_axes_0, x = v_full_7_cast_fp16)[name = tensor("op_773_cast_fp16")]; tensor var_775_reps_0 = const()[name = tensor("op_775_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_775_cast_fp16 = tile(reps = var_775_reps_0, x = var_773_cast_fp16)[name = tensor("op_775_cast_fp16")]; tensor var_776 = const()[name = tensor("op_776"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_7_cast_fp16 = reshape(shape = var_776, x = var_775_cast_fp16)[name = tensor("v_rep_7_cast_fp16")]; tensor var_779_transpose_x_1 = const()[name = tensor("op_779_transpose_x_1"), val = tensor(false)]; tensor var_779_transpose_y_1 = const()[name = tensor("op_779_transpose_y_1"), val = tensor(true)]; tensor var_779_cast_fp16 = matmul(transpose_x = var_779_transpose_x_1, transpose_y = var_779_transpose_y_1, x = q_15_cast_fp16, y = k_rep_7_cast_fp16)[name = tensor("op_779_cast_fp16")]; tensor var_780_to_fp16 = const()[name = tensor("op_780_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_13_cast_fp16 = mul(x = var_779_cast_fp16, y = var_780_to_fp16)[name = tensor("attn_13_cast_fp16")]; tensor input_15_cast_fp16 = add(x = attn_13_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_15_cast_fp16")]; tensor input_15_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_15_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_15_cast_fp16_to_fp32 = cast(dtype = input_15_cast_fp16_to_fp32_dtype_0, x = input_15_cast_fp16)[name = tensor("cast_620")]; tensor attn_15 = softmax(axis = var_678, x = input_15_cast_fp16_to_fp32)[name = tensor("attn_15")]; tensor out_7_transpose_x_0 = const()[name = tensor("out_7_transpose_x_0"), val = tensor(false)]; tensor out_7_transpose_y_0 = const()[name = tensor("out_7_transpose_y_0"), val = tensor(false)]; tensor attn_15_to_fp16_dtype_0 = const()[name = tensor("attn_15_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_15_to_fp16 = cast(dtype = attn_15_to_fp16_dtype_0, x = attn_15)[name = tensor("cast_619")]; tensor out_7_cast_fp16 = matmul(transpose_x = out_7_transpose_x_0, transpose_y = out_7_transpose_y_0, x = attn_15_to_fp16, y = v_rep_7_cast_fp16)[name = tensor("out_7_cast_fp16")]; tensor var_785_perm_0 = const()[name = tensor("op_785_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_787 = const()[name = tensor("op_787"), val = tensor([1, 1, 1536])]; tensor var_785_cast_fp16 = transpose(perm = var_785_perm_0, x = out_7_cast_fp16)[name = tensor("transpose_96")]; tensor x_147_cast_fp16 = reshape(shape = var_787, x = var_785_cast_fp16)[name = tensor("x_147_cast_fp16")]; tensor layers_3_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_3_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(251830272)))]; tensor linear_24_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_3_self_attn_o_proj_weight_to_fp16, x = x_147_cast_fp16)[name = tensor("linear_24_cast_fp16")]; tensor x_149_cast_fp16 = add(x = x_121_cast_fp16, y = linear_24_cast_fp16)[name = tensor("x_149_cast_fp16")]; tensor x_149_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_149_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_677_promoted_3 = const()[name = tensor("op_677_promoted_3"), val = tensor(0x1p+1)]; tensor x_149_cast_fp16_to_fp32 = cast(dtype = x_149_cast_fp16_to_fp32_dtype_0, x = x_149_cast_fp16)[name = tensor("cast_618")]; tensor var_798 = pow(x = x_149_cast_fp16_to_fp32, y = var_677_promoted_3)[name = tensor("op_798")]; tensor var_31_axes_0 = const()[name = tensor("var_31_axes_0"), val = tensor([-1])]; tensor var_31_keep_dims_0 = const()[name = tensor("var_31_keep_dims_0"), val = tensor(true)]; tensor var_31 = reduce_mean(axes = var_31_axes_0, keep_dims = var_31_keep_dims_0, x = var_798)[name = tensor("var_31")]; tensor var_31_to_fp16_dtype_0 = const()[name = tensor("var_31_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_802_to_fp16 = const()[name = tensor("op_802_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_31_to_fp16 = cast(dtype = var_31_to_fp16_dtype_0, x = var_31)[name = tensor("cast_617")]; tensor var_803_cast_fp16 = add(x = var_31_to_fp16, y = var_802_to_fp16)[name = tensor("op_803_cast_fp16")]; tensor var_803_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_803_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_804_epsilon_0 = const()[name = tensor("op_804_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_803_cast_fp16_to_fp32 = cast(dtype = var_803_cast_fp16_to_fp32_dtype_0, x = var_803_cast_fp16)[name = tensor("cast_616")]; tensor var_804 = rsqrt(epsilon = var_804_epsilon_0, x = var_803_cast_fp16_to_fp32)[name = tensor("op_804")]; tensor var_804_to_fp16_dtype_0 = const()[name = tensor("op_804_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_804_to_fp16 = cast(dtype = var_804_to_fp16_dtype_0, x = var_804)[name = tensor("cast_615")]; tensor x_155_cast_fp16 = mul(x = x_149_cast_fp16, y = var_804_to_fp16)[name = tensor("x_155_cast_fp16")]; tensor layers_3_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_3_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(253403200)))]; tensor x_157_cast_fp16 = mul(x = layers_3_post_attention_layernorm_weight_to_fp16, y = x_155_cast_fp16)[name = tensor("x_157_cast_fp16")]; tensor layers_3_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_3_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(253404288)))]; tensor linear_25_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_3_mlp_gate_proj_weight_to_fp16, x = x_157_cast_fp16)[name = tensor("linear_25_cast_fp16")]; tensor var_815_cast_fp16 = silu(x = linear_25_cast_fp16)[name = tensor("op_815_cast_fp16")]; tensor layers_3_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_3_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(254977216)))]; tensor linear_26_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_3_mlp_up_proj_weight_to_fp16, x = x_157_cast_fp16)[name = tensor("linear_26_cast_fp16")]; tensor x_159_cast_fp16 = mul(x = var_815_cast_fp16, y = linear_26_cast_fp16)[name = tensor("x_159_cast_fp16")]; tensor layers_3_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_3_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(256550144)))]; tensor linear_27_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_3_mlp_down_proj_weight_to_fp16, x = x_159_cast_fp16)[name = tensor("linear_27_cast_fp16")]; tensor x_161_cast_fp16 = add(x = x_149_cast_fp16, y = linear_27_cast_fp16)[name = tensor("x_161_cast_fp16")]; tensor x_161_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_161_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_9_begin_0 = const()[name = tensor("k_cache_9_begin_0"), val = tensor([4, 0, 0, 0, 0])]; tensor k_cache_9_end_0 = const()[name = tensor("k_cache_9_end_0"), val = tensor([5, 1, 4, 2048, 128])]; tensor k_cache_9_end_mask_0 = const()[name = tensor("k_cache_9_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_9_squeeze_mask_0 = const()[name = tensor("k_cache_9_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_9_cast_fp16 = slice_by_index(begin = k_cache_9_begin_0, end = k_cache_9_end_0, end_mask = k_cache_9_end_mask_0, squeeze_mask = k_cache_9_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_9_cast_fp16")]; tensor v_cache_9_begin_0 = const()[name = tensor("v_cache_9_begin_0"), val = tensor([4, 0, 0, 0, 0])]; tensor v_cache_9_end_0 = const()[name = tensor("v_cache_9_end_0"), val = tensor([5, 1, 4, 2048, 128])]; tensor v_cache_9_end_mask_0 = const()[name = tensor("v_cache_9_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_9_squeeze_mask_0 = const()[name = tensor("v_cache_9_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_9_cast_fp16 = slice_by_index(begin = v_cache_9_begin_0, end = v_cache_9_end_0, end_mask = v_cache_9_end_mask_0, squeeze_mask = v_cache_9_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_9_cast_fp16")]; tensor var_847 = const()[name = tensor("op_847"), val = tensor(-1)]; tensor var_846_promoted = const()[name = tensor("op_846_promoted"), val = tensor(0x1p+1)]; tensor x_161_cast_fp16_to_fp32 = cast(dtype = x_161_cast_fp16_to_fp32_dtype_0, x = x_161_cast_fp16)[name = tensor("cast_614")]; tensor var_856 = pow(x = x_161_cast_fp16_to_fp32, y = var_846_promoted)[name = tensor("op_856")]; tensor var_33_axes_0 = const()[name = tensor("var_33_axes_0"), val = tensor([-1])]; tensor var_33_keep_dims_0 = const()[name = tensor("var_33_keep_dims_0"), val = tensor(true)]; tensor var_33 = reduce_mean(axes = var_33_axes_0, keep_dims = var_33_keep_dims_0, x = var_856)[name = tensor("var_33")]; tensor var_33_to_fp16_dtype_0 = const()[name = tensor("var_33_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_860_to_fp16 = const()[name = tensor("op_860_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_33_to_fp16 = cast(dtype = var_33_to_fp16_dtype_0, x = var_33)[name = tensor("cast_613")]; tensor var_861_cast_fp16 = add(x = var_33_to_fp16, y = var_860_to_fp16)[name = tensor("op_861_cast_fp16")]; tensor var_861_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_861_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_862_epsilon_0 = const()[name = tensor("op_862_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_861_cast_fp16_to_fp32 = cast(dtype = var_861_cast_fp16_to_fp32_dtype_0, x = var_861_cast_fp16)[name = tensor("cast_612")]; tensor var_862 = rsqrt(epsilon = var_862_epsilon_0, x = var_861_cast_fp16_to_fp32)[name = tensor("op_862")]; tensor var_862_to_fp16_dtype_0 = const()[name = tensor("op_862_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_862_to_fp16 = cast(dtype = var_862_to_fp16_dtype_0, x = var_862)[name = tensor("cast_611")]; tensor x_167_cast_fp16 = mul(x = x_161_cast_fp16, y = var_862_to_fp16)[name = tensor("x_167_cast_fp16")]; tensor layers_4_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_4_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(258123072)))]; tensor x_169_cast_fp16 = mul(x = layers_4_input_layernorm_weight_to_fp16, y = x_167_cast_fp16)[name = tensor("x_169_cast_fp16")]; tensor layers_4_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_4_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(258124160)))]; tensor linear_28_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_4_self_attn_q_proj_weight_to_fp16, x = x_169_cast_fp16)[name = tensor("linear_28_cast_fp16")]; tensor var_876 = const()[name = tensor("op_876"), val = tensor([1, 1, 12, 128])]; tensor x_171_cast_fp16 = reshape(shape = var_876, x = linear_28_cast_fp16)[name = tensor("x_171_cast_fp16")]; tensor x_171_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_171_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_846_promoted_1 = const()[name = tensor("op_846_promoted_1"), val = tensor(0x1p+1)]; tensor x_171_cast_fp16_to_fp32 = cast(dtype = x_171_cast_fp16_to_fp32_dtype_0, x = x_171_cast_fp16)[name = tensor("cast_610")]; tensor var_880 = pow(x = x_171_cast_fp16_to_fp32, y = var_846_promoted_1)[name = tensor("op_880")]; tensor var_35_axes_0 = const()[name = tensor("var_35_axes_0"), val = tensor([-1])]; tensor var_35_keep_dims_0 = const()[name = tensor("var_35_keep_dims_0"), val = tensor(true)]; tensor var_35 = reduce_mean(axes = var_35_axes_0, keep_dims = var_35_keep_dims_0, x = var_880)[name = tensor("var_35")]; tensor var_35_to_fp16_dtype_0 = const()[name = tensor("var_35_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_884_to_fp16 = const()[name = tensor("op_884_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_35_to_fp16 = cast(dtype = var_35_to_fp16_dtype_0, x = var_35)[name = tensor("cast_609")]; tensor var_885_cast_fp16 = add(x = var_35_to_fp16, y = var_884_to_fp16)[name = tensor("op_885_cast_fp16")]; tensor var_885_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_885_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_886_epsilon_0 = const()[name = tensor("op_886_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_885_cast_fp16_to_fp32 = cast(dtype = var_885_cast_fp16_to_fp32_dtype_0, x = var_885_cast_fp16)[name = tensor("cast_608")]; tensor var_886 = rsqrt(epsilon = var_886_epsilon_0, x = var_885_cast_fp16_to_fp32)[name = tensor("op_886")]; tensor var_886_to_fp16_dtype_0 = const()[name = tensor("op_886_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_886_to_fp16 = cast(dtype = var_886_to_fp16_dtype_0, x = var_886)[name = tensor("cast_607")]; tensor x_177_cast_fp16 = mul(x = x_171_cast_fp16, y = var_886_to_fp16)[name = tensor("x_177_cast_fp16")]; tensor layers_4_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_4_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(259697088)))]; tensor var_888_cast_fp16 = mul(x = layers_4_self_attn_q_norm_weight_to_fp16, y = x_177_cast_fp16)[name = tensor("op_888_cast_fp16")]; tensor q_17_perm_0 = const()[name = tensor("q_17_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_4_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_4_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(259697408)))]; tensor linear_29_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_4_self_attn_k_proj_weight_to_fp16, x = x_169_cast_fp16)[name = tensor("linear_29_cast_fp16")]; tensor var_892 = const()[name = tensor("op_892"), val = tensor([1, 1, 4, 128])]; tensor x_179_cast_fp16 = reshape(shape = var_892, x = linear_29_cast_fp16)[name = tensor("x_179_cast_fp16")]; tensor x_179_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_179_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_846_promoted_2 = const()[name = tensor("op_846_promoted_2"), val = tensor(0x1p+1)]; tensor x_179_cast_fp16_to_fp32 = cast(dtype = x_179_cast_fp16_to_fp32_dtype_0, x = x_179_cast_fp16)[name = tensor("cast_606")]; tensor var_896 = pow(x = x_179_cast_fp16_to_fp32, y = var_846_promoted_2)[name = tensor("op_896")]; tensor var_37_axes_0 = const()[name = tensor("var_37_axes_0"), val = tensor([-1])]; tensor var_37_keep_dims_0 = const()[name = tensor("var_37_keep_dims_0"), val = tensor(true)]; tensor var_37 = reduce_mean(axes = var_37_axes_0, keep_dims = var_37_keep_dims_0, x = var_896)[name = tensor("var_37")]; tensor var_37_to_fp16_dtype_0 = const()[name = tensor("var_37_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_900_to_fp16 = const()[name = tensor("op_900_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_37_to_fp16 = cast(dtype = var_37_to_fp16_dtype_0, x = var_37)[name = tensor("cast_605")]; tensor var_901_cast_fp16 = add(x = var_37_to_fp16, y = var_900_to_fp16)[name = tensor("op_901_cast_fp16")]; tensor var_901_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_901_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_902_epsilon_0 = const()[name = tensor("op_902_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_901_cast_fp16_to_fp32 = cast(dtype = var_901_cast_fp16_to_fp32_dtype_0, x = var_901_cast_fp16)[name = tensor("cast_604")]; tensor var_902 = rsqrt(epsilon = var_902_epsilon_0, x = var_901_cast_fp16_to_fp32)[name = tensor("op_902")]; tensor var_902_to_fp16_dtype_0 = const()[name = tensor("op_902_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_902_to_fp16 = cast(dtype = var_902_to_fp16_dtype_0, x = var_902)[name = tensor("cast_603")]; tensor x_185_cast_fp16 = mul(x = x_179_cast_fp16, y = var_902_to_fp16)[name = tensor("x_185_cast_fp16")]; tensor layers_4_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_4_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(260221760)))]; tensor var_904_cast_fp16 = mul(x = layers_4_self_attn_k_norm_weight_to_fp16, y = x_185_cast_fp16)[name = tensor("op_904_cast_fp16")]; tensor k_17_perm_0 = const()[name = tensor("k_17_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_4_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_4_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(260222080)))]; tensor linear_30_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_4_self_attn_v_proj_weight_to_fp16, x = x_169_cast_fp16)[name = tensor("linear_30_cast_fp16")]; tensor var_908 = const()[name = tensor("op_908"), val = tensor([1, 1, 4, 128])]; tensor var_909_cast_fp16 = reshape(shape = var_908, x = linear_30_cast_fp16)[name = tensor("op_909_cast_fp16")]; tensor v_9_perm_0 = const()[name = tensor("v_9_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_17_cast_fp16 = transpose(perm = q_17_perm_0, x = var_888_cast_fp16)[name = tensor("transpose_95")]; tensor var_913_cast_fp16 = mul(x = q_17_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_913_cast_fp16")]; tensor x1_17_begin_0 = const()[name = tensor("x1_17_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_17_end_0 = const()[name = tensor("x1_17_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_17_end_mask_0 = const()[name = tensor("x1_17_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_17_cast_fp16 = slice_by_index(begin = x1_17_begin_0, end = x1_17_end_0, end_mask = x1_17_end_mask_0, x = q_17_cast_fp16)[name = tensor("x1_17_cast_fp16")]; tensor x2_17_begin_0 = const()[name = tensor("x2_17_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_17_end_0 = const()[name = tensor("x2_17_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_17_end_mask_0 = const()[name = tensor("x2_17_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_17_cast_fp16 = slice_by_index(begin = x2_17_begin_0, end = x2_17_end_0, end_mask = x2_17_end_mask_0, x = q_17_cast_fp16)[name = tensor("x2_17_cast_fp16")]; tensor const_9_promoted_to_fp16 = const()[name = tensor("const_9_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_916_cast_fp16 = mul(x = x2_17_cast_fp16, y = const_9_promoted_to_fp16)[name = tensor("op_916_cast_fp16")]; tensor var_918_interleave_0 = const()[name = tensor("op_918_interleave_0"), val = tensor(false)]; tensor var_918_cast_fp16 = concat(axis = var_847, interleave = var_918_interleave_0, values = (var_916_cast_fp16, x1_17_cast_fp16))[name = tensor("op_918_cast_fp16")]; tensor var_919_cast_fp16 = mul(x = var_918_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_919_cast_fp16")]; tensor q_19_cast_fp16 = add(x = var_913_cast_fp16, y = var_919_cast_fp16)[name = tensor("q_19_cast_fp16")]; tensor k_17_cast_fp16 = transpose(perm = k_17_perm_0, x = var_904_cast_fp16)[name = tensor("transpose_94")]; tensor var_921_cast_fp16 = mul(x = k_17_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_921_cast_fp16")]; tensor x1_19_begin_0 = const()[name = tensor("x1_19_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_19_end_0 = const()[name = tensor("x1_19_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_19_end_mask_0 = const()[name = tensor("x1_19_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_19_cast_fp16 = slice_by_index(begin = x1_19_begin_0, end = x1_19_end_0, end_mask = x1_19_end_mask_0, x = k_17_cast_fp16)[name = tensor("x1_19_cast_fp16")]; tensor x2_19_begin_0 = const()[name = tensor("x2_19_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_19_end_0 = const()[name = tensor("x2_19_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_19_end_mask_0 = const()[name = tensor("x2_19_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_19_cast_fp16 = slice_by_index(begin = x2_19_begin_0, end = x2_19_end_0, end_mask = x2_19_end_mask_0, x = k_17_cast_fp16)[name = tensor("x2_19_cast_fp16")]; tensor const_10_promoted_to_fp16 = const()[name = tensor("const_10_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_924_cast_fp16 = mul(x = x2_19_cast_fp16, y = const_10_promoted_to_fp16)[name = tensor("op_924_cast_fp16")]; tensor var_926_interleave_0 = const()[name = tensor("op_926_interleave_0"), val = tensor(false)]; tensor var_926_cast_fp16 = concat(axis = var_847, interleave = var_926_interleave_0, values = (var_924_cast_fp16, x1_19_cast_fp16))[name = tensor("op_926_cast_fp16")]; tensor var_927_cast_fp16 = mul(x = var_926_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_927_cast_fp16")]; tensor k_19_cast_fp16 = add(x = var_921_cast_fp16, y = var_927_cast_fp16)[name = tensor("k_19_cast_fp16")]; tensor var_930_cast_fp16 = mul(x = k_cache_9_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_930_cast_fp16")]; tensor var_931_cast_fp16 = mul(x = k_19_cast_fp16, y = var_107_to_fp16)[name = tensor("op_931_cast_fp16")]; tensor k_full_9_cast_fp16 = add(x = var_930_cast_fp16, y = var_931_cast_fp16)[name = tensor("k_full_9_cast_fp16")]; tensor var_934_cast_fp16 = mul(x = v_cache_9_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_934_cast_fp16")]; tensor v_9_cast_fp16 = transpose(perm = v_9_perm_0, x = var_909_cast_fp16)[name = tensor("transpose_93")]; tensor var_935_cast_fp16 = mul(x = v_9_cast_fp16, y = var_107_to_fp16)[name = tensor("op_935_cast_fp16")]; tensor v_full_9_cast_fp16 = add(x = var_934_cast_fp16, y = var_935_cast_fp16)[name = tensor("v_full_9_cast_fp16")]; tensor var_937_axes_0 = const()[name = tensor("op_937_axes_0"), val = tensor([2])]; tensor var_937_cast_fp16 = expand_dims(axes = var_937_axes_0, x = k_full_9_cast_fp16)[name = tensor("op_937_cast_fp16")]; tensor var_939_reps_0 = const()[name = tensor("op_939_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_939_cast_fp16 = tile(reps = var_939_reps_0, x = var_937_cast_fp16)[name = tensor("op_939_cast_fp16")]; tensor var_940 = const()[name = tensor("op_940"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_9_cast_fp16 = reshape(shape = var_940, x = var_939_cast_fp16)[name = tensor("k_rep_9_cast_fp16")]; tensor var_942_axes_0 = const()[name = tensor("op_942_axes_0"), val = tensor([2])]; tensor var_942_cast_fp16 = expand_dims(axes = var_942_axes_0, x = v_full_9_cast_fp16)[name = tensor("op_942_cast_fp16")]; tensor var_944_reps_0 = const()[name = tensor("op_944_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_944_cast_fp16 = tile(reps = var_944_reps_0, x = var_942_cast_fp16)[name = tensor("op_944_cast_fp16")]; tensor var_945 = const()[name = tensor("op_945"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_9_cast_fp16 = reshape(shape = var_945, x = var_944_cast_fp16)[name = tensor("v_rep_9_cast_fp16")]; tensor var_948_transpose_x_1 = const()[name = tensor("op_948_transpose_x_1"), val = tensor(false)]; tensor var_948_transpose_y_1 = const()[name = tensor("op_948_transpose_y_1"), val = tensor(true)]; tensor var_948_cast_fp16 = matmul(transpose_x = var_948_transpose_x_1, transpose_y = var_948_transpose_y_1, x = q_19_cast_fp16, y = k_rep_9_cast_fp16)[name = tensor("op_948_cast_fp16")]; tensor var_949_to_fp16 = const()[name = tensor("op_949_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_17_cast_fp16 = mul(x = var_948_cast_fp16, y = var_949_to_fp16)[name = tensor("attn_17_cast_fp16")]; tensor input_19_cast_fp16 = add(x = attn_17_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_19_cast_fp16")]; tensor input_19_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_19_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_19_cast_fp16_to_fp32 = cast(dtype = input_19_cast_fp16_to_fp32_dtype_0, x = input_19_cast_fp16)[name = tensor("cast_602")]; tensor attn_19 = softmax(axis = var_847, x = input_19_cast_fp16_to_fp32)[name = tensor("attn_19")]; tensor out_9_transpose_x_0 = const()[name = tensor("out_9_transpose_x_0"), val = tensor(false)]; tensor out_9_transpose_y_0 = const()[name = tensor("out_9_transpose_y_0"), val = tensor(false)]; tensor attn_19_to_fp16_dtype_0 = const()[name = tensor("attn_19_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_19_to_fp16 = cast(dtype = attn_19_to_fp16_dtype_0, x = attn_19)[name = tensor("cast_601")]; tensor out_9_cast_fp16 = matmul(transpose_x = out_9_transpose_x_0, transpose_y = out_9_transpose_y_0, x = attn_19_to_fp16, y = v_rep_9_cast_fp16)[name = tensor("out_9_cast_fp16")]; tensor var_954_perm_0 = const()[name = tensor("op_954_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_956 = const()[name = tensor("op_956"), val = tensor([1, 1, 1536])]; tensor var_954_cast_fp16 = transpose(perm = var_954_perm_0, x = out_9_cast_fp16)[name = tensor("transpose_92")]; tensor x_187_cast_fp16 = reshape(shape = var_956, x = var_954_cast_fp16)[name = tensor("x_187_cast_fp16")]; tensor layers_4_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_4_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(260746432)))]; tensor linear_31_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_4_self_attn_o_proj_weight_to_fp16, x = x_187_cast_fp16)[name = tensor("linear_31_cast_fp16")]; tensor x_189_cast_fp16 = add(x = x_161_cast_fp16, y = linear_31_cast_fp16)[name = tensor("x_189_cast_fp16")]; tensor x_189_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_189_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_846_promoted_3 = const()[name = tensor("op_846_promoted_3"), val = tensor(0x1p+1)]; tensor x_189_cast_fp16_to_fp32 = cast(dtype = x_189_cast_fp16_to_fp32_dtype_0, x = x_189_cast_fp16)[name = tensor("cast_600")]; tensor var_967 = pow(x = x_189_cast_fp16_to_fp32, y = var_846_promoted_3)[name = tensor("op_967")]; tensor var_39_axes_0 = const()[name = tensor("var_39_axes_0"), val = tensor([-1])]; tensor var_39_keep_dims_0 = const()[name = tensor("var_39_keep_dims_0"), val = tensor(true)]; tensor var_39 = reduce_mean(axes = var_39_axes_0, keep_dims = var_39_keep_dims_0, x = var_967)[name = tensor("var_39")]; tensor var_39_to_fp16_dtype_0 = const()[name = tensor("var_39_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_971_to_fp16 = const()[name = tensor("op_971_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_39_to_fp16 = cast(dtype = var_39_to_fp16_dtype_0, x = var_39)[name = tensor("cast_599")]; tensor var_972_cast_fp16 = add(x = var_39_to_fp16, y = var_971_to_fp16)[name = tensor("op_972_cast_fp16")]; tensor var_972_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_972_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_973_epsilon_0 = const()[name = tensor("op_973_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_972_cast_fp16_to_fp32 = cast(dtype = var_972_cast_fp16_to_fp32_dtype_0, x = var_972_cast_fp16)[name = tensor("cast_598")]; tensor var_973 = rsqrt(epsilon = var_973_epsilon_0, x = var_972_cast_fp16_to_fp32)[name = tensor("op_973")]; tensor var_973_to_fp16_dtype_0 = const()[name = tensor("op_973_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_973_to_fp16 = cast(dtype = var_973_to_fp16_dtype_0, x = var_973)[name = tensor("cast_597")]; tensor x_195_cast_fp16 = mul(x = x_189_cast_fp16, y = var_973_to_fp16)[name = tensor("x_195_cast_fp16")]; tensor layers_4_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_4_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(262319360)))]; tensor x_197_cast_fp16 = mul(x = layers_4_post_attention_layernorm_weight_to_fp16, y = x_195_cast_fp16)[name = tensor("x_197_cast_fp16")]; tensor layers_4_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_4_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(262320448)))]; tensor linear_32_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_4_mlp_gate_proj_weight_to_fp16, x = x_197_cast_fp16)[name = tensor("linear_32_cast_fp16")]; tensor var_984_cast_fp16 = silu(x = linear_32_cast_fp16)[name = tensor("op_984_cast_fp16")]; tensor layers_4_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_4_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263893376)))]; tensor linear_33_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_4_mlp_up_proj_weight_to_fp16, x = x_197_cast_fp16)[name = tensor("linear_33_cast_fp16")]; tensor x_199_cast_fp16 = mul(x = var_984_cast_fp16, y = linear_33_cast_fp16)[name = tensor("x_199_cast_fp16")]; tensor layers_4_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_4_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(265466304)))]; tensor linear_34_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_4_mlp_down_proj_weight_to_fp16, x = x_199_cast_fp16)[name = tensor("linear_34_cast_fp16")]; tensor x_201_cast_fp16 = add(x = x_189_cast_fp16, y = linear_34_cast_fp16)[name = tensor("x_201_cast_fp16")]; tensor x_201_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_201_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_11_begin_0 = const()[name = tensor("k_cache_11_begin_0"), val = tensor([5, 0, 0, 0, 0])]; tensor k_cache_11_end_0 = const()[name = tensor("k_cache_11_end_0"), val = tensor([6, 1, 4, 2048, 128])]; tensor k_cache_11_end_mask_0 = const()[name = tensor("k_cache_11_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_11_squeeze_mask_0 = const()[name = tensor("k_cache_11_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_11_cast_fp16 = slice_by_index(begin = k_cache_11_begin_0, end = k_cache_11_end_0, end_mask = k_cache_11_end_mask_0, squeeze_mask = k_cache_11_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_11_cast_fp16")]; tensor v_cache_11_begin_0 = const()[name = tensor("v_cache_11_begin_0"), val = tensor([5, 0, 0, 0, 0])]; tensor v_cache_11_end_0 = const()[name = tensor("v_cache_11_end_0"), val = tensor([6, 1, 4, 2048, 128])]; tensor v_cache_11_end_mask_0 = const()[name = tensor("v_cache_11_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_11_squeeze_mask_0 = const()[name = tensor("v_cache_11_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_11_cast_fp16 = slice_by_index(begin = v_cache_11_begin_0, end = v_cache_11_end_0, end_mask = v_cache_11_end_mask_0, squeeze_mask = v_cache_11_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_11_cast_fp16")]; tensor var_1016 = const()[name = tensor("op_1016"), val = tensor(-1)]; tensor var_1015_promoted = const()[name = tensor("op_1015_promoted"), val = tensor(0x1p+1)]; tensor x_201_cast_fp16_to_fp32 = cast(dtype = x_201_cast_fp16_to_fp32_dtype_0, x = x_201_cast_fp16)[name = tensor("cast_596")]; tensor var_1025 = pow(x = x_201_cast_fp16_to_fp32, y = var_1015_promoted)[name = tensor("op_1025")]; tensor var_41_axes_0 = const()[name = tensor("var_41_axes_0"), val = tensor([-1])]; tensor var_41_keep_dims_0 = const()[name = tensor("var_41_keep_dims_0"), val = tensor(true)]; tensor var_41 = reduce_mean(axes = var_41_axes_0, keep_dims = var_41_keep_dims_0, x = var_1025)[name = tensor("var_41")]; tensor var_41_to_fp16_dtype_0 = const()[name = tensor("var_41_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1029_to_fp16 = const()[name = tensor("op_1029_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_41_to_fp16 = cast(dtype = var_41_to_fp16_dtype_0, x = var_41)[name = tensor("cast_595")]; tensor var_1030_cast_fp16 = add(x = var_41_to_fp16, y = var_1029_to_fp16)[name = tensor("op_1030_cast_fp16")]; tensor var_1030_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1030_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1031_epsilon_0 = const()[name = tensor("op_1031_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1030_cast_fp16_to_fp32 = cast(dtype = var_1030_cast_fp16_to_fp32_dtype_0, x = var_1030_cast_fp16)[name = tensor("cast_594")]; tensor var_1031 = rsqrt(epsilon = var_1031_epsilon_0, x = var_1030_cast_fp16_to_fp32)[name = tensor("op_1031")]; tensor var_1031_to_fp16_dtype_0 = const()[name = tensor("op_1031_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1031_to_fp16 = cast(dtype = var_1031_to_fp16_dtype_0, x = var_1031)[name = tensor("cast_593")]; tensor x_207_cast_fp16 = mul(x = x_201_cast_fp16, y = var_1031_to_fp16)[name = tensor("x_207_cast_fp16")]; tensor layers_5_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_5_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267039232)))]; tensor x_209_cast_fp16 = mul(x = layers_5_input_layernorm_weight_to_fp16, y = x_207_cast_fp16)[name = tensor("x_209_cast_fp16")]; tensor layers_5_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_5_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267040320)))]; tensor linear_35_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_5_self_attn_q_proj_weight_to_fp16, x = x_209_cast_fp16)[name = tensor("linear_35_cast_fp16")]; tensor var_1045 = const()[name = tensor("op_1045"), val = tensor([1, 1, 12, 128])]; tensor x_211_cast_fp16 = reshape(shape = var_1045, x = linear_35_cast_fp16)[name = tensor("x_211_cast_fp16")]; tensor x_211_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_211_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1015_promoted_1 = const()[name = tensor("op_1015_promoted_1"), val = tensor(0x1p+1)]; tensor x_211_cast_fp16_to_fp32 = cast(dtype = x_211_cast_fp16_to_fp32_dtype_0, x = x_211_cast_fp16)[name = tensor("cast_592")]; tensor var_1049 = pow(x = x_211_cast_fp16_to_fp32, y = var_1015_promoted_1)[name = tensor("op_1049")]; tensor var_43_axes_0 = const()[name = tensor("var_43_axes_0"), val = tensor([-1])]; tensor var_43_keep_dims_0 = const()[name = tensor("var_43_keep_dims_0"), val = tensor(true)]; tensor var_43 = reduce_mean(axes = var_43_axes_0, keep_dims = var_43_keep_dims_0, x = var_1049)[name = tensor("var_43")]; tensor var_43_to_fp16_dtype_0 = const()[name = tensor("var_43_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1053_to_fp16 = const()[name = tensor("op_1053_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_43_to_fp16 = cast(dtype = var_43_to_fp16_dtype_0, x = var_43)[name = tensor("cast_591")]; tensor var_1054_cast_fp16 = add(x = var_43_to_fp16, y = var_1053_to_fp16)[name = tensor("op_1054_cast_fp16")]; tensor var_1054_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1054_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1055_epsilon_0 = const()[name = tensor("op_1055_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1054_cast_fp16_to_fp32 = cast(dtype = var_1054_cast_fp16_to_fp32_dtype_0, x = var_1054_cast_fp16)[name = tensor("cast_590")]; tensor var_1055 = rsqrt(epsilon = var_1055_epsilon_0, x = var_1054_cast_fp16_to_fp32)[name = tensor("op_1055")]; tensor var_1055_to_fp16_dtype_0 = const()[name = tensor("op_1055_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1055_to_fp16 = cast(dtype = var_1055_to_fp16_dtype_0, x = var_1055)[name = tensor("cast_589")]; tensor x_217_cast_fp16 = mul(x = x_211_cast_fp16, y = var_1055_to_fp16)[name = tensor("x_217_cast_fp16")]; tensor layers_5_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_5_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(268613248)))]; tensor var_1057_cast_fp16 = mul(x = layers_5_self_attn_q_norm_weight_to_fp16, y = x_217_cast_fp16)[name = tensor("op_1057_cast_fp16")]; tensor q_21_perm_0 = const()[name = tensor("q_21_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_5_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_5_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(268613568)))]; tensor linear_36_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_5_self_attn_k_proj_weight_to_fp16, x = x_209_cast_fp16)[name = tensor("linear_36_cast_fp16")]; tensor var_1061 = const()[name = tensor("op_1061"), val = tensor([1, 1, 4, 128])]; tensor x_219_cast_fp16 = reshape(shape = var_1061, x = linear_36_cast_fp16)[name = tensor("x_219_cast_fp16")]; tensor x_219_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_219_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1015_promoted_2 = const()[name = tensor("op_1015_promoted_2"), val = tensor(0x1p+1)]; tensor x_219_cast_fp16_to_fp32 = cast(dtype = x_219_cast_fp16_to_fp32_dtype_0, x = x_219_cast_fp16)[name = tensor("cast_588")]; tensor var_1065 = pow(x = x_219_cast_fp16_to_fp32, y = var_1015_promoted_2)[name = tensor("op_1065")]; tensor var_45_axes_0 = const()[name = tensor("var_45_axes_0"), val = tensor([-1])]; tensor var_45_keep_dims_0 = const()[name = tensor("var_45_keep_dims_0"), val = tensor(true)]; tensor var_45 = reduce_mean(axes = var_45_axes_0, keep_dims = var_45_keep_dims_0, x = var_1065)[name = tensor("var_45")]; tensor var_45_to_fp16_dtype_0 = const()[name = tensor("var_45_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1069_to_fp16 = const()[name = tensor("op_1069_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_45_to_fp16 = cast(dtype = var_45_to_fp16_dtype_0, x = var_45)[name = tensor("cast_587")]; tensor var_1070_cast_fp16 = add(x = var_45_to_fp16, y = var_1069_to_fp16)[name = tensor("op_1070_cast_fp16")]; tensor var_1070_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1070_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1071_epsilon_0 = const()[name = tensor("op_1071_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1070_cast_fp16_to_fp32 = cast(dtype = var_1070_cast_fp16_to_fp32_dtype_0, x = var_1070_cast_fp16)[name = tensor("cast_586")]; tensor var_1071 = rsqrt(epsilon = var_1071_epsilon_0, x = var_1070_cast_fp16_to_fp32)[name = tensor("op_1071")]; tensor var_1071_to_fp16_dtype_0 = const()[name = tensor("op_1071_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1071_to_fp16 = cast(dtype = var_1071_to_fp16_dtype_0, x = var_1071)[name = tensor("cast_585")]; tensor x_225_cast_fp16 = mul(x = x_219_cast_fp16, y = var_1071_to_fp16)[name = tensor("x_225_cast_fp16")]; tensor layers_5_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_5_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(269137920)))]; tensor var_1073_cast_fp16 = mul(x = layers_5_self_attn_k_norm_weight_to_fp16, y = x_225_cast_fp16)[name = tensor("op_1073_cast_fp16")]; tensor k_21_perm_0 = const()[name = tensor("k_21_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_5_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_5_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(269138240)))]; tensor linear_37_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_5_self_attn_v_proj_weight_to_fp16, x = x_209_cast_fp16)[name = tensor("linear_37_cast_fp16")]; tensor var_1077 = const()[name = tensor("op_1077"), val = tensor([1, 1, 4, 128])]; tensor var_1078_cast_fp16 = reshape(shape = var_1077, x = linear_37_cast_fp16)[name = tensor("op_1078_cast_fp16")]; tensor v_11_perm_0 = const()[name = tensor("v_11_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_21_cast_fp16 = transpose(perm = q_21_perm_0, x = var_1057_cast_fp16)[name = tensor("transpose_91")]; tensor var_1082_cast_fp16 = mul(x = q_21_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1082_cast_fp16")]; tensor x1_21_begin_0 = const()[name = tensor("x1_21_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_21_end_0 = const()[name = tensor("x1_21_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_21_end_mask_0 = const()[name = tensor("x1_21_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_21_cast_fp16 = slice_by_index(begin = x1_21_begin_0, end = x1_21_end_0, end_mask = x1_21_end_mask_0, x = q_21_cast_fp16)[name = tensor("x1_21_cast_fp16")]; tensor x2_21_begin_0 = const()[name = tensor("x2_21_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_21_end_0 = const()[name = tensor("x2_21_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_21_end_mask_0 = const()[name = tensor("x2_21_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_21_cast_fp16 = slice_by_index(begin = x2_21_begin_0, end = x2_21_end_0, end_mask = x2_21_end_mask_0, x = q_21_cast_fp16)[name = tensor("x2_21_cast_fp16")]; tensor const_11_promoted_to_fp16 = const()[name = tensor("const_11_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1085_cast_fp16 = mul(x = x2_21_cast_fp16, y = const_11_promoted_to_fp16)[name = tensor("op_1085_cast_fp16")]; tensor var_1087_interleave_0 = const()[name = tensor("op_1087_interleave_0"), val = tensor(false)]; tensor var_1087_cast_fp16 = concat(axis = var_1016, interleave = var_1087_interleave_0, values = (var_1085_cast_fp16, x1_21_cast_fp16))[name = tensor("op_1087_cast_fp16")]; tensor var_1088_cast_fp16 = mul(x = var_1087_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1088_cast_fp16")]; tensor q_23_cast_fp16 = add(x = var_1082_cast_fp16, y = var_1088_cast_fp16)[name = tensor("q_23_cast_fp16")]; tensor k_21_cast_fp16 = transpose(perm = k_21_perm_0, x = var_1073_cast_fp16)[name = tensor("transpose_90")]; tensor var_1090_cast_fp16 = mul(x = k_21_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1090_cast_fp16")]; tensor x1_23_begin_0 = const()[name = tensor("x1_23_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_23_end_0 = const()[name = tensor("x1_23_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_23_end_mask_0 = const()[name = tensor("x1_23_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_23_cast_fp16 = slice_by_index(begin = x1_23_begin_0, end = x1_23_end_0, end_mask = x1_23_end_mask_0, x = k_21_cast_fp16)[name = tensor("x1_23_cast_fp16")]; tensor x2_23_begin_0 = const()[name = tensor("x2_23_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_23_end_0 = const()[name = tensor("x2_23_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_23_end_mask_0 = const()[name = tensor("x2_23_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_23_cast_fp16 = slice_by_index(begin = x2_23_begin_0, end = x2_23_end_0, end_mask = x2_23_end_mask_0, x = k_21_cast_fp16)[name = tensor("x2_23_cast_fp16")]; tensor const_12_promoted_to_fp16 = const()[name = tensor("const_12_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1093_cast_fp16 = mul(x = x2_23_cast_fp16, y = const_12_promoted_to_fp16)[name = tensor("op_1093_cast_fp16")]; tensor var_1095_interleave_0 = const()[name = tensor("op_1095_interleave_0"), val = tensor(false)]; tensor var_1095_cast_fp16 = concat(axis = var_1016, interleave = var_1095_interleave_0, values = (var_1093_cast_fp16, x1_23_cast_fp16))[name = tensor("op_1095_cast_fp16")]; tensor var_1096_cast_fp16 = mul(x = var_1095_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1096_cast_fp16")]; tensor k_23_cast_fp16 = add(x = var_1090_cast_fp16, y = var_1096_cast_fp16)[name = tensor("k_23_cast_fp16")]; tensor var_1099_cast_fp16 = mul(x = k_cache_11_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1099_cast_fp16")]; tensor var_1100_cast_fp16 = mul(x = k_23_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1100_cast_fp16")]; tensor k_full_11_cast_fp16 = add(x = var_1099_cast_fp16, y = var_1100_cast_fp16)[name = tensor("k_full_11_cast_fp16")]; tensor var_1103_cast_fp16 = mul(x = v_cache_11_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1103_cast_fp16")]; tensor v_11_cast_fp16 = transpose(perm = v_11_perm_0, x = var_1078_cast_fp16)[name = tensor("transpose_89")]; tensor var_1104_cast_fp16 = mul(x = v_11_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1104_cast_fp16")]; tensor v_full_11_cast_fp16 = add(x = var_1103_cast_fp16, y = var_1104_cast_fp16)[name = tensor("v_full_11_cast_fp16")]; tensor var_1106_axes_0 = const()[name = tensor("op_1106_axes_0"), val = tensor([2])]; tensor var_1106_cast_fp16 = expand_dims(axes = var_1106_axes_0, x = k_full_11_cast_fp16)[name = tensor("op_1106_cast_fp16")]; tensor var_1108_reps_0 = const()[name = tensor("op_1108_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1108_cast_fp16 = tile(reps = var_1108_reps_0, x = var_1106_cast_fp16)[name = tensor("op_1108_cast_fp16")]; tensor var_1109 = const()[name = tensor("op_1109"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_11_cast_fp16 = reshape(shape = var_1109, x = var_1108_cast_fp16)[name = tensor("k_rep_11_cast_fp16")]; tensor var_1111_axes_0 = const()[name = tensor("op_1111_axes_0"), val = tensor([2])]; tensor var_1111_cast_fp16 = expand_dims(axes = var_1111_axes_0, x = v_full_11_cast_fp16)[name = tensor("op_1111_cast_fp16")]; tensor var_1113_reps_0 = const()[name = tensor("op_1113_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1113_cast_fp16 = tile(reps = var_1113_reps_0, x = var_1111_cast_fp16)[name = tensor("op_1113_cast_fp16")]; tensor var_1114 = const()[name = tensor("op_1114"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_11_cast_fp16 = reshape(shape = var_1114, x = var_1113_cast_fp16)[name = tensor("v_rep_11_cast_fp16")]; tensor var_1117_transpose_x_1 = const()[name = tensor("op_1117_transpose_x_1"), val = tensor(false)]; tensor var_1117_transpose_y_1 = const()[name = tensor("op_1117_transpose_y_1"), val = tensor(true)]; tensor var_1117_cast_fp16 = matmul(transpose_x = var_1117_transpose_x_1, transpose_y = var_1117_transpose_y_1, x = q_23_cast_fp16, y = k_rep_11_cast_fp16)[name = tensor("op_1117_cast_fp16")]; tensor var_1118_to_fp16 = const()[name = tensor("op_1118_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_21_cast_fp16 = mul(x = var_1117_cast_fp16, y = var_1118_to_fp16)[name = tensor("attn_21_cast_fp16")]; tensor input_23_cast_fp16 = add(x = attn_21_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_23_cast_fp16")]; tensor input_23_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_23_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_23_cast_fp16_to_fp32 = cast(dtype = input_23_cast_fp16_to_fp32_dtype_0, x = input_23_cast_fp16)[name = tensor("cast_584")]; tensor attn_23 = softmax(axis = var_1016, x = input_23_cast_fp16_to_fp32)[name = tensor("attn_23")]; tensor out_11_transpose_x_0 = const()[name = tensor("out_11_transpose_x_0"), val = tensor(false)]; tensor out_11_transpose_y_0 = const()[name = tensor("out_11_transpose_y_0"), val = tensor(false)]; tensor attn_23_to_fp16_dtype_0 = const()[name = tensor("attn_23_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_23_to_fp16 = cast(dtype = attn_23_to_fp16_dtype_0, x = attn_23)[name = tensor("cast_583")]; tensor out_11_cast_fp16 = matmul(transpose_x = out_11_transpose_x_0, transpose_y = out_11_transpose_y_0, x = attn_23_to_fp16, y = v_rep_11_cast_fp16)[name = tensor("out_11_cast_fp16")]; tensor var_1123_perm_0 = const()[name = tensor("op_1123_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1125 = const()[name = tensor("op_1125"), val = tensor([1, 1, 1536])]; tensor var_1123_cast_fp16 = transpose(perm = var_1123_perm_0, x = out_11_cast_fp16)[name = tensor("transpose_88")]; tensor x_227_cast_fp16 = reshape(shape = var_1125, x = var_1123_cast_fp16)[name = tensor("x_227_cast_fp16")]; tensor layers_5_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_5_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(269662592)))]; tensor linear_38_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_5_self_attn_o_proj_weight_to_fp16, x = x_227_cast_fp16)[name = tensor("linear_38_cast_fp16")]; tensor x_229_cast_fp16 = add(x = x_201_cast_fp16, y = linear_38_cast_fp16)[name = tensor("x_229_cast_fp16")]; tensor x_229_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_229_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1015_promoted_3 = const()[name = tensor("op_1015_promoted_3"), val = tensor(0x1p+1)]; tensor x_229_cast_fp16_to_fp32 = cast(dtype = x_229_cast_fp16_to_fp32_dtype_0, x = x_229_cast_fp16)[name = tensor("cast_582")]; tensor var_1136 = pow(x = x_229_cast_fp16_to_fp32, y = var_1015_promoted_3)[name = tensor("op_1136")]; tensor var_47_axes_0 = const()[name = tensor("var_47_axes_0"), val = tensor([-1])]; tensor var_47_keep_dims_0 = const()[name = tensor("var_47_keep_dims_0"), val = tensor(true)]; tensor var_47 = reduce_mean(axes = var_47_axes_0, keep_dims = var_47_keep_dims_0, x = var_1136)[name = tensor("var_47")]; tensor var_47_to_fp16_dtype_0 = const()[name = tensor("var_47_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1140_to_fp16 = const()[name = tensor("op_1140_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_47_to_fp16 = cast(dtype = var_47_to_fp16_dtype_0, x = var_47)[name = tensor("cast_581")]; tensor var_1141_cast_fp16 = add(x = var_47_to_fp16, y = var_1140_to_fp16)[name = tensor("op_1141_cast_fp16")]; tensor var_1141_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1141_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1142_epsilon_0 = const()[name = tensor("op_1142_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1141_cast_fp16_to_fp32 = cast(dtype = var_1141_cast_fp16_to_fp32_dtype_0, x = var_1141_cast_fp16)[name = tensor("cast_580")]; tensor var_1142 = rsqrt(epsilon = var_1142_epsilon_0, x = var_1141_cast_fp16_to_fp32)[name = tensor("op_1142")]; tensor var_1142_to_fp16_dtype_0 = const()[name = tensor("op_1142_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1142_to_fp16 = cast(dtype = var_1142_to_fp16_dtype_0, x = var_1142)[name = tensor("cast_579")]; tensor x_235_cast_fp16 = mul(x = x_229_cast_fp16, y = var_1142_to_fp16)[name = tensor("x_235_cast_fp16")]; tensor layers_5_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_5_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(271235520)))]; tensor x_237_cast_fp16 = mul(x = layers_5_post_attention_layernorm_weight_to_fp16, y = x_235_cast_fp16)[name = tensor("x_237_cast_fp16")]; tensor layers_5_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_5_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(271236608)))]; tensor linear_39_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_5_mlp_gate_proj_weight_to_fp16, x = x_237_cast_fp16)[name = tensor("linear_39_cast_fp16")]; tensor var_1153_cast_fp16 = silu(x = linear_39_cast_fp16)[name = tensor("op_1153_cast_fp16")]; tensor layers_5_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_5_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(272809536)))]; tensor linear_40_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_5_mlp_up_proj_weight_to_fp16, x = x_237_cast_fp16)[name = tensor("linear_40_cast_fp16")]; tensor x_239_cast_fp16 = mul(x = var_1153_cast_fp16, y = linear_40_cast_fp16)[name = tensor("x_239_cast_fp16")]; tensor layers_5_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_5_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(274382464)))]; tensor linear_41_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_5_mlp_down_proj_weight_to_fp16, x = x_239_cast_fp16)[name = tensor("linear_41_cast_fp16")]; tensor x_241_cast_fp16 = add(x = x_229_cast_fp16, y = linear_41_cast_fp16)[name = tensor("x_241_cast_fp16")]; tensor x_241_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_241_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_13_begin_0 = const()[name = tensor("k_cache_13_begin_0"), val = tensor([6, 0, 0, 0, 0])]; tensor k_cache_13_end_0 = const()[name = tensor("k_cache_13_end_0"), val = tensor([7, 1, 4, 2048, 128])]; tensor k_cache_13_end_mask_0 = const()[name = tensor("k_cache_13_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_13_squeeze_mask_0 = const()[name = tensor("k_cache_13_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_13_cast_fp16 = slice_by_index(begin = k_cache_13_begin_0, end = k_cache_13_end_0, end_mask = k_cache_13_end_mask_0, squeeze_mask = k_cache_13_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_13_cast_fp16")]; tensor v_cache_13_begin_0 = const()[name = tensor("v_cache_13_begin_0"), val = tensor([6, 0, 0, 0, 0])]; tensor v_cache_13_end_0 = const()[name = tensor("v_cache_13_end_0"), val = tensor([7, 1, 4, 2048, 128])]; tensor v_cache_13_end_mask_0 = const()[name = tensor("v_cache_13_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_13_squeeze_mask_0 = const()[name = tensor("v_cache_13_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_13_cast_fp16 = slice_by_index(begin = v_cache_13_begin_0, end = v_cache_13_end_0, end_mask = v_cache_13_end_mask_0, squeeze_mask = v_cache_13_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_13_cast_fp16")]; tensor var_1185 = const()[name = tensor("op_1185"), val = tensor(-1)]; tensor var_1184_promoted = const()[name = tensor("op_1184_promoted"), val = tensor(0x1p+1)]; tensor x_241_cast_fp16_to_fp32 = cast(dtype = x_241_cast_fp16_to_fp32_dtype_0, x = x_241_cast_fp16)[name = tensor("cast_578")]; tensor var_1194 = pow(x = x_241_cast_fp16_to_fp32, y = var_1184_promoted)[name = tensor("op_1194")]; tensor var_49_axes_0 = const()[name = tensor("var_49_axes_0"), val = tensor([-1])]; tensor var_49_keep_dims_0 = const()[name = tensor("var_49_keep_dims_0"), val = tensor(true)]; tensor var_49 = reduce_mean(axes = var_49_axes_0, keep_dims = var_49_keep_dims_0, x = var_1194)[name = tensor("var_49")]; tensor var_49_to_fp16_dtype_0 = const()[name = tensor("var_49_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1198_to_fp16 = const()[name = tensor("op_1198_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_49_to_fp16 = cast(dtype = var_49_to_fp16_dtype_0, x = var_49)[name = tensor("cast_577")]; tensor var_1199_cast_fp16 = add(x = var_49_to_fp16, y = var_1198_to_fp16)[name = tensor("op_1199_cast_fp16")]; tensor var_1199_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1199_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1200_epsilon_0 = const()[name = tensor("op_1200_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1199_cast_fp16_to_fp32 = cast(dtype = var_1199_cast_fp16_to_fp32_dtype_0, x = var_1199_cast_fp16)[name = tensor("cast_576")]; tensor var_1200 = rsqrt(epsilon = var_1200_epsilon_0, x = var_1199_cast_fp16_to_fp32)[name = tensor("op_1200")]; tensor var_1200_to_fp16_dtype_0 = const()[name = tensor("op_1200_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1200_to_fp16 = cast(dtype = var_1200_to_fp16_dtype_0, x = var_1200)[name = tensor("cast_575")]; tensor x_247_cast_fp16 = mul(x = x_241_cast_fp16, y = var_1200_to_fp16)[name = tensor("x_247_cast_fp16")]; tensor layers_6_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_6_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(275955392)))]; tensor x_249_cast_fp16 = mul(x = layers_6_input_layernorm_weight_to_fp16, y = x_247_cast_fp16)[name = tensor("x_249_cast_fp16")]; tensor layers_6_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_6_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(275956480)))]; tensor linear_42_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_6_self_attn_q_proj_weight_to_fp16, x = x_249_cast_fp16)[name = tensor("linear_42_cast_fp16")]; tensor var_1214 = const()[name = tensor("op_1214"), val = tensor([1, 1, 12, 128])]; tensor x_251_cast_fp16 = reshape(shape = var_1214, x = linear_42_cast_fp16)[name = tensor("x_251_cast_fp16")]; tensor x_251_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_251_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1184_promoted_1 = const()[name = tensor("op_1184_promoted_1"), val = tensor(0x1p+1)]; tensor x_251_cast_fp16_to_fp32 = cast(dtype = x_251_cast_fp16_to_fp32_dtype_0, x = x_251_cast_fp16)[name = tensor("cast_574")]; tensor var_1218 = pow(x = x_251_cast_fp16_to_fp32, y = var_1184_promoted_1)[name = tensor("op_1218")]; tensor var_51_axes_0 = const()[name = tensor("var_51_axes_0"), val = tensor([-1])]; tensor var_51_keep_dims_0 = const()[name = tensor("var_51_keep_dims_0"), val = tensor(true)]; tensor var_51 = reduce_mean(axes = var_51_axes_0, keep_dims = var_51_keep_dims_0, x = var_1218)[name = tensor("var_51")]; tensor var_51_to_fp16_dtype_0 = const()[name = tensor("var_51_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1222_to_fp16 = const()[name = tensor("op_1222_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_51_to_fp16 = cast(dtype = var_51_to_fp16_dtype_0, x = var_51)[name = tensor("cast_573")]; tensor var_1223_cast_fp16 = add(x = var_51_to_fp16, y = var_1222_to_fp16)[name = tensor("op_1223_cast_fp16")]; tensor var_1223_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1223_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1224_epsilon_0 = const()[name = tensor("op_1224_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1223_cast_fp16_to_fp32 = cast(dtype = var_1223_cast_fp16_to_fp32_dtype_0, x = var_1223_cast_fp16)[name = tensor("cast_572")]; tensor var_1224 = rsqrt(epsilon = var_1224_epsilon_0, x = var_1223_cast_fp16_to_fp32)[name = tensor("op_1224")]; tensor var_1224_to_fp16_dtype_0 = const()[name = tensor("op_1224_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1224_to_fp16 = cast(dtype = var_1224_to_fp16_dtype_0, x = var_1224)[name = tensor("cast_571")]; tensor x_257_cast_fp16 = mul(x = x_251_cast_fp16, y = var_1224_to_fp16)[name = tensor("x_257_cast_fp16")]; tensor layers_6_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_6_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(277529408)))]; tensor var_1226_cast_fp16 = mul(x = layers_6_self_attn_q_norm_weight_to_fp16, y = x_257_cast_fp16)[name = tensor("op_1226_cast_fp16")]; tensor q_25_perm_0 = const()[name = tensor("q_25_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_6_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_6_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(277529728)))]; tensor linear_43_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_6_self_attn_k_proj_weight_to_fp16, x = x_249_cast_fp16)[name = tensor("linear_43_cast_fp16")]; tensor var_1230 = const()[name = tensor("op_1230"), val = tensor([1, 1, 4, 128])]; tensor x_259_cast_fp16 = reshape(shape = var_1230, x = linear_43_cast_fp16)[name = tensor("x_259_cast_fp16")]; tensor x_259_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_259_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1184_promoted_2 = const()[name = tensor("op_1184_promoted_2"), val = tensor(0x1p+1)]; tensor x_259_cast_fp16_to_fp32 = cast(dtype = x_259_cast_fp16_to_fp32_dtype_0, x = x_259_cast_fp16)[name = tensor("cast_570")]; tensor var_1234 = pow(x = x_259_cast_fp16_to_fp32, y = var_1184_promoted_2)[name = tensor("op_1234")]; tensor var_53_axes_0 = const()[name = tensor("var_53_axes_0"), val = tensor([-1])]; tensor var_53_keep_dims_0 = const()[name = tensor("var_53_keep_dims_0"), val = tensor(true)]; tensor var_53 = reduce_mean(axes = var_53_axes_0, keep_dims = var_53_keep_dims_0, x = var_1234)[name = tensor("var_53")]; tensor var_53_to_fp16_dtype_0 = const()[name = tensor("var_53_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1238_to_fp16 = const()[name = tensor("op_1238_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_53_to_fp16 = cast(dtype = var_53_to_fp16_dtype_0, x = var_53)[name = tensor("cast_569")]; tensor var_1239_cast_fp16 = add(x = var_53_to_fp16, y = var_1238_to_fp16)[name = tensor("op_1239_cast_fp16")]; tensor var_1239_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1239_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1240_epsilon_0 = const()[name = tensor("op_1240_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1239_cast_fp16_to_fp32 = cast(dtype = var_1239_cast_fp16_to_fp32_dtype_0, x = var_1239_cast_fp16)[name = tensor("cast_568")]; tensor var_1240 = rsqrt(epsilon = var_1240_epsilon_0, x = var_1239_cast_fp16_to_fp32)[name = tensor("op_1240")]; tensor var_1240_to_fp16_dtype_0 = const()[name = tensor("op_1240_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1240_to_fp16 = cast(dtype = var_1240_to_fp16_dtype_0, x = var_1240)[name = tensor("cast_567")]; tensor x_265_cast_fp16 = mul(x = x_259_cast_fp16, y = var_1240_to_fp16)[name = tensor("x_265_cast_fp16")]; tensor layers_6_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_6_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(278054080)))]; tensor var_1242_cast_fp16 = mul(x = layers_6_self_attn_k_norm_weight_to_fp16, y = x_265_cast_fp16)[name = tensor("op_1242_cast_fp16")]; tensor k_25_perm_0 = const()[name = tensor("k_25_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_6_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_6_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(278054400)))]; tensor linear_44_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_6_self_attn_v_proj_weight_to_fp16, x = x_249_cast_fp16)[name = tensor("linear_44_cast_fp16")]; tensor var_1246 = const()[name = tensor("op_1246"), val = tensor([1, 1, 4, 128])]; tensor var_1247_cast_fp16 = reshape(shape = var_1246, x = linear_44_cast_fp16)[name = tensor("op_1247_cast_fp16")]; tensor v_13_perm_0 = const()[name = tensor("v_13_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_25_cast_fp16 = transpose(perm = q_25_perm_0, x = var_1226_cast_fp16)[name = tensor("transpose_87")]; tensor var_1251_cast_fp16 = mul(x = q_25_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1251_cast_fp16")]; tensor x1_25_begin_0 = const()[name = tensor("x1_25_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_25_end_0 = const()[name = tensor("x1_25_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_25_end_mask_0 = const()[name = tensor("x1_25_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_25_cast_fp16 = slice_by_index(begin = x1_25_begin_0, end = x1_25_end_0, end_mask = x1_25_end_mask_0, x = q_25_cast_fp16)[name = tensor("x1_25_cast_fp16")]; tensor x2_25_begin_0 = const()[name = tensor("x2_25_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_25_end_0 = const()[name = tensor("x2_25_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_25_end_mask_0 = const()[name = tensor("x2_25_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_25_cast_fp16 = slice_by_index(begin = x2_25_begin_0, end = x2_25_end_0, end_mask = x2_25_end_mask_0, x = q_25_cast_fp16)[name = tensor("x2_25_cast_fp16")]; tensor const_13_promoted_to_fp16 = const()[name = tensor("const_13_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1254_cast_fp16 = mul(x = x2_25_cast_fp16, y = const_13_promoted_to_fp16)[name = tensor("op_1254_cast_fp16")]; tensor var_1256_interleave_0 = const()[name = tensor("op_1256_interleave_0"), val = tensor(false)]; tensor var_1256_cast_fp16 = concat(axis = var_1185, interleave = var_1256_interleave_0, values = (var_1254_cast_fp16, x1_25_cast_fp16))[name = tensor("op_1256_cast_fp16")]; tensor var_1257_cast_fp16 = mul(x = var_1256_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1257_cast_fp16")]; tensor q_27_cast_fp16 = add(x = var_1251_cast_fp16, y = var_1257_cast_fp16)[name = tensor("q_27_cast_fp16")]; tensor k_25_cast_fp16 = transpose(perm = k_25_perm_0, x = var_1242_cast_fp16)[name = tensor("transpose_86")]; tensor var_1259_cast_fp16 = mul(x = k_25_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1259_cast_fp16")]; tensor x1_27_begin_0 = const()[name = tensor("x1_27_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_27_end_0 = const()[name = tensor("x1_27_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_27_end_mask_0 = const()[name = tensor("x1_27_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_27_cast_fp16 = slice_by_index(begin = x1_27_begin_0, end = x1_27_end_0, end_mask = x1_27_end_mask_0, x = k_25_cast_fp16)[name = tensor("x1_27_cast_fp16")]; tensor x2_27_begin_0 = const()[name = tensor("x2_27_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_27_end_0 = const()[name = tensor("x2_27_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_27_end_mask_0 = const()[name = tensor("x2_27_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_27_cast_fp16 = slice_by_index(begin = x2_27_begin_0, end = x2_27_end_0, end_mask = x2_27_end_mask_0, x = k_25_cast_fp16)[name = tensor("x2_27_cast_fp16")]; tensor const_14_promoted_to_fp16 = const()[name = tensor("const_14_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1262_cast_fp16 = mul(x = x2_27_cast_fp16, y = const_14_promoted_to_fp16)[name = tensor("op_1262_cast_fp16")]; tensor var_1264_interleave_0 = const()[name = tensor("op_1264_interleave_0"), val = tensor(false)]; tensor var_1264_cast_fp16 = concat(axis = var_1185, interleave = var_1264_interleave_0, values = (var_1262_cast_fp16, x1_27_cast_fp16))[name = tensor("op_1264_cast_fp16")]; tensor var_1265_cast_fp16 = mul(x = var_1264_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1265_cast_fp16")]; tensor k_27_cast_fp16 = add(x = var_1259_cast_fp16, y = var_1265_cast_fp16)[name = tensor("k_27_cast_fp16")]; tensor var_1268_cast_fp16 = mul(x = k_cache_13_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1268_cast_fp16")]; tensor var_1269_cast_fp16 = mul(x = k_27_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1269_cast_fp16")]; tensor k_full_13_cast_fp16 = add(x = var_1268_cast_fp16, y = var_1269_cast_fp16)[name = tensor("k_full_13_cast_fp16")]; tensor var_1272_cast_fp16 = mul(x = v_cache_13_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1272_cast_fp16")]; tensor v_13_cast_fp16 = transpose(perm = v_13_perm_0, x = var_1247_cast_fp16)[name = tensor("transpose_85")]; tensor var_1273_cast_fp16 = mul(x = v_13_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1273_cast_fp16")]; tensor v_full_13_cast_fp16 = add(x = var_1272_cast_fp16, y = var_1273_cast_fp16)[name = tensor("v_full_13_cast_fp16")]; tensor var_1275_axes_0 = const()[name = tensor("op_1275_axes_0"), val = tensor([2])]; tensor var_1275_cast_fp16 = expand_dims(axes = var_1275_axes_0, x = k_full_13_cast_fp16)[name = tensor("op_1275_cast_fp16")]; tensor var_1277_reps_0 = const()[name = tensor("op_1277_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1277_cast_fp16 = tile(reps = var_1277_reps_0, x = var_1275_cast_fp16)[name = tensor("op_1277_cast_fp16")]; tensor var_1278 = const()[name = tensor("op_1278"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_13_cast_fp16 = reshape(shape = var_1278, x = var_1277_cast_fp16)[name = tensor("k_rep_13_cast_fp16")]; tensor var_1280_axes_0 = const()[name = tensor("op_1280_axes_0"), val = tensor([2])]; tensor var_1280_cast_fp16 = expand_dims(axes = var_1280_axes_0, x = v_full_13_cast_fp16)[name = tensor("op_1280_cast_fp16")]; tensor var_1282_reps_0 = const()[name = tensor("op_1282_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1282_cast_fp16 = tile(reps = var_1282_reps_0, x = var_1280_cast_fp16)[name = tensor("op_1282_cast_fp16")]; tensor var_1283 = const()[name = tensor("op_1283"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_13_cast_fp16 = reshape(shape = var_1283, x = var_1282_cast_fp16)[name = tensor("v_rep_13_cast_fp16")]; tensor var_1286_transpose_x_1 = const()[name = tensor("op_1286_transpose_x_1"), val = tensor(false)]; tensor var_1286_transpose_y_1 = const()[name = tensor("op_1286_transpose_y_1"), val = tensor(true)]; tensor var_1286_cast_fp16 = matmul(transpose_x = var_1286_transpose_x_1, transpose_y = var_1286_transpose_y_1, x = q_27_cast_fp16, y = k_rep_13_cast_fp16)[name = tensor("op_1286_cast_fp16")]; tensor var_1287_to_fp16 = const()[name = tensor("op_1287_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_25_cast_fp16 = mul(x = var_1286_cast_fp16, y = var_1287_to_fp16)[name = tensor("attn_25_cast_fp16")]; tensor input_27_cast_fp16 = add(x = attn_25_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_27_cast_fp16")]; tensor input_27_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_27_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_27_cast_fp16_to_fp32 = cast(dtype = input_27_cast_fp16_to_fp32_dtype_0, x = input_27_cast_fp16)[name = tensor("cast_566")]; tensor attn_27 = softmax(axis = var_1185, x = input_27_cast_fp16_to_fp32)[name = tensor("attn_27")]; tensor out_13_transpose_x_0 = const()[name = tensor("out_13_transpose_x_0"), val = tensor(false)]; tensor out_13_transpose_y_0 = const()[name = tensor("out_13_transpose_y_0"), val = tensor(false)]; tensor attn_27_to_fp16_dtype_0 = const()[name = tensor("attn_27_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_27_to_fp16 = cast(dtype = attn_27_to_fp16_dtype_0, x = attn_27)[name = tensor("cast_565")]; tensor out_13_cast_fp16 = matmul(transpose_x = out_13_transpose_x_0, transpose_y = out_13_transpose_y_0, x = attn_27_to_fp16, y = v_rep_13_cast_fp16)[name = tensor("out_13_cast_fp16")]; tensor var_1292_perm_0 = const()[name = tensor("op_1292_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1294 = const()[name = tensor("op_1294"), val = tensor([1, 1, 1536])]; tensor var_1292_cast_fp16 = transpose(perm = var_1292_perm_0, x = out_13_cast_fp16)[name = tensor("transpose_84")]; tensor x_267_cast_fp16 = reshape(shape = var_1294, x = var_1292_cast_fp16)[name = tensor("x_267_cast_fp16")]; tensor layers_6_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_6_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(278578752)))]; tensor linear_45_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_6_self_attn_o_proj_weight_to_fp16, x = x_267_cast_fp16)[name = tensor("linear_45_cast_fp16")]; tensor x_269_cast_fp16 = add(x = x_241_cast_fp16, y = linear_45_cast_fp16)[name = tensor("x_269_cast_fp16")]; tensor x_269_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_269_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1184_promoted_3 = const()[name = tensor("op_1184_promoted_3"), val = tensor(0x1p+1)]; tensor x_269_cast_fp16_to_fp32 = cast(dtype = x_269_cast_fp16_to_fp32_dtype_0, x = x_269_cast_fp16)[name = tensor("cast_564")]; tensor var_1305 = pow(x = x_269_cast_fp16_to_fp32, y = var_1184_promoted_3)[name = tensor("op_1305")]; tensor var_55_axes_0 = const()[name = tensor("var_55_axes_0"), val = tensor([-1])]; tensor var_55_keep_dims_0 = const()[name = tensor("var_55_keep_dims_0"), val = tensor(true)]; tensor var_55 = reduce_mean(axes = var_55_axes_0, keep_dims = var_55_keep_dims_0, x = var_1305)[name = tensor("var_55")]; tensor var_55_to_fp16_dtype_0 = const()[name = tensor("var_55_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1309_to_fp16 = const()[name = tensor("op_1309_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_55_to_fp16 = cast(dtype = var_55_to_fp16_dtype_0, x = var_55)[name = tensor("cast_563")]; tensor var_1310_cast_fp16 = add(x = var_55_to_fp16, y = var_1309_to_fp16)[name = tensor("op_1310_cast_fp16")]; tensor var_1310_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1310_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1311_epsilon_0 = const()[name = tensor("op_1311_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1310_cast_fp16_to_fp32 = cast(dtype = var_1310_cast_fp16_to_fp32_dtype_0, x = var_1310_cast_fp16)[name = tensor("cast_562")]; tensor var_1311 = rsqrt(epsilon = var_1311_epsilon_0, x = var_1310_cast_fp16_to_fp32)[name = tensor("op_1311")]; tensor var_1311_to_fp16_dtype_0 = const()[name = tensor("op_1311_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1311_to_fp16 = cast(dtype = var_1311_to_fp16_dtype_0, x = var_1311)[name = tensor("cast_561")]; tensor x_275_cast_fp16 = mul(x = x_269_cast_fp16, y = var_1311_to_fp16)[name = tensor("x_275_cast_fp16")]; tensor layers_6_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_6_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(280151680)))]; tensor x_277_cast_fp16 = mul(x = layers_6_post_attention_layernorm_weight_to_fp16, y = x_275_cast_fp16)[name = tensor("x_277_cast_fp16")]; tensor layers_6_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_6_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(280152768)))]; tensor linear_46_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_6_mlp_gate_proj_weight_to_fp16, x = x_277_cast_fp16)[name = tensor("linear_46_cast_fp16")]; tensor var_1322_cast_fp16 = silu(x = linear_46_cast_fp16)[name = tensor("op_1322_cast_fp16")]; tensor layers_6_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_6_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(281725696)))]; tensor linear_47_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_6_mlp_up_proj_weight_to_fp16, x = x_277_cast_fp16)[name = tensor("linear_47_cast_fp16")]; tensor x_279_cast_fp16 = mul(x = var_1322_cast_fp16, y = linear_47_cast_fp16)[name = tensor("x_279_cast_fp16")]; tensor layers_6_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_6_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(283298624)))]; tensor linear_48_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_6_mlp_down_proj_weight_to_fp16, x = x_279_cast_fp16)[name = tensor("linear_48_cast_fp16")]; tensor x_281_cast_fp16 = add(x = x_269_cast_fp16, y = linear_48_cast_fp16)[name = tensor("x_281_cast_fp16")]; tensor x_281_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_281_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_15_begin_0 = const()[name = tensor("k_cache_15_begin_0"), val = tensor([7, 0, 0, 0, 0])]; tensor k_cache_15_end_0 = const()[name = tensor("k_cache_15_end_0"), val = tensor([8, 1, 4, 2048, 128])]; tensor k_cache_15_end_mask_0 = const()[name = tensor("k_cache_15_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_15_squeeze_mask_0 = const()[name = tensor("k_cache_15_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_15_cast_fp16 = slice_by_index(begin = k_cache_15_begin_0, end = k_cache_15_end_0, end_mask = k_cache_15_end_mask_0, squeeze_mask = k_cache_15_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_15_cast_fp16")]; tensor v_cache_15_begin_0 = const()[name = tensor("v_cache_15_begin_0"), val = tensor([7, 0, 0, 0, 0])]; tensor v_cache_15_end_0 = const()[name = tensor("v_cache_15_end_0"), val = tensor([8, 1, 4, 2048, 128])]; tensor v_cache_15_end_mask_0 = const()[name = tensor("v_cache_15_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_15_squeeze_mask_0 = const()[name = tensor("v_cache_15_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_15_cast_fp16 = slice_by_index(begin = v_cache_15_begin_0, end = v_cache_15_end_0, end_mask = v_cache_15_end_mask_0, squeeze_mask = v_cache_15_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_15_cast_fp16")]; tensor var_1354 = const()[name = tensor("op_1354"), val = tensor(-1)]; tensor var_1353_promoted = const()[name = tensor("op_1353_promoted"), val = tensor(0x1p+1)]; tensor x_281_cast_fp16_to_fp32 = cast(dtype = x_281_cast_fp16_to_fp32_dtype_0, x = x_281_cast_fp16)[name = tensor("cast_560")]; tensor var_1363 = pow(x = x_281_cast_fp16_to_fp32, y = var_1353_promoted)[name = tensor("op_1363")]; tensor var_57_axes_0 = const()[name = tensor("var_57_axes_0"), val = tensor([-1])]; tensor var_57_keep_dims_0 = const()[name = tensor("var_57_keep_dims_0"), val = tensor(true)]; tensor var_57 = reduce_mean(axes = var_57_axes_0, keep_dims = var_57_keep_dims_0, x = var_1363)[name = tensor("var_57")]; tensor var_57_to_fp16_dtype_0 = const()[name = tensor("var_57_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1367_to_fp16 = const()[name = tensor("op_1367_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_57_to_fp16 = cast(dtype = var_57_to_fp16_dtype_0, x = var_57)[name = tensor("cast_559")]; tensor var_1368_cast_fp16 = add(x = var_57_to_fp16, y = var_1367_to_fp16)[name = tensor("op_1368_cast_fp16")]; tensor var_1368_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1368_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1369_epsilon_0 = const()[name = tensor("op_1369_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1368_cast_fp16_to_fp32 = cast(dtype = var_1368_cast_fp16_to_fp32_dtype_0, x = var_1368_cast_fp16)[name = tensor("cast_558")]; tensor var_1369 = rsqrt(epsilon = var_1369_epsilon_0, x = var_1368_cast_fp16_to_fp32)[name = tensor("op_1369")]; tensor var_1369_to_fp16_dtype_0 = const()[name = tensor("op_1369_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1369_to_fp16 = cast(dtype = var_1369_to_fp16_dtype_0, x = var_1369)[name = tensor("cast_557")]; tensor x_287_cast_fp16 = mul(x = x_281_cast_fp16, y = var_1369_to_fp16)[name = tensor("x_287_cast_fp16")]; tensor layers_7_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_7_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284871552)))]; tensor x_289_cast_fp16 = mul(x = layers_7_input_layernorm_weight_to_fp16, y = x_287_cast_fp16)[name = tensor("x_289_cast_fp16")]; tensor layers_7_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_7_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284872640)))]; tensor linear_49_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_7_self_attn_q_proj_weight_to_fp16, x = x_289_cast_fp16)[name = tensor("linear_49_cast_fp16")]; tensor var_1383 = const()[name = tensor("op_1383"), val = tensor([1, 1, 12, 128])]; tensor x_291_cast_fp16 = reshape(shape = var_1383, x = linear_49_cast_fp16)[name = tensor("x_291_cast_fp16")]; tensor x_291_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_291_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1353_promoted_1 = const()[name = tensor("op_1353_promoted_1"), val = tensor(0x1p+1)]; tensor x_291_cast_fp16_to_fp32 = cast(dtype = x_291_cast_fp16_to_fp32_dtype_0, x = x_291_cast_fp16)[name = tensor("cast_556")]; tensor var_1387 = pow(x = x_291_cast_fp16_to_fp32, y = var_1353_promoted_1)[name = tensor("op_1387")]; tensor var_59_axes_0 = const()[name = tensor("var_59_axes_0"), val = tensor([-1])]; tensor var_59_keep_dims_0 = const()[name = tensor("var_59_keep_dims_0"), val = tensor(true)]; tensor var_59 = reduce_mean(axes = var_59_axes_0, keep_dims = var_59_keep_dims_0, x = var_1387)[name = tensor("var_59")]; tensor var_59_to_fp16_dtype_0 = const()[name = tensor("var_59_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1391_to_fp16 = const()[name = tensor("op_1391_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_59_to_fp16 = cast(dtype = var_59_to_fp16_dtype_0, x = var_59)[name = tensor("cast_555")]; tensor var_1392_cast_fp16 = add(x = var_59_to_fp16, y = var_1391_to_fp16)[name = tensor("op_1392_cast_fp16")]; tensor var_1392_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1392_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1393_epsilon_0 = const()[name = tensor("op_1393_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1392_cast_fp16_to_fp32 = cast(dtype = var_1392_cast_fp16_to_fp32_dtype_0, x = var_1392_cast_fp16)[name = tensor("cast_554")]; tensor var_1393 = rsqrt(epsilon = var_1393_epsilon_0, x = var_1392_cast_fp16_to_fp32)[name = tensor("op_1393")]; tensor var_1393_to_fp16_dtype_0 = const()[name = tensor("op_1393_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1393_to_fp16 = cast(dtype = var_1393_to_fp16_dtype_0, x = var_1393)[name = tensor("cast_553")]; tensor x_297_cast_fp16 = mul(x = x_291_cast_fp16, y = var_1393_to_fp16)[name = tensor("x_297_cast_fp16")]; tensor layers_7_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_7_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(286445568)))]; tensor var_1395_cast_fp16 = mul(x = layers_7_self_attn_q_norm_weight_to_fp16, y = x_297_cast_fp16)[name = tensor("op_1395_cast_fp16")]; tensor q_29_perm_0 = const()[name = tensor("q_29_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_7_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_7_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(286445888)))]; tensor linear_50_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_7_self_attn_k_proj_weight_to_fp16, x = x_289_cast_fp16)[name = tensor("linear_50_cast_fp16")]; tensor var_1399 = const()[name = tensor("op_1399"), val = tensor([1, 1, 4, 128])]; tensor x_299_cast_fp16 = reshape(shape = var_1399, x = linear_50_cast_fp16)[name = tensor("x_299_cast_fp16")]; tensor x_299_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_299_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1353_promoted_2 = const()[name = tensor("op_1353_promoted_2"), val = tensor(0x1p+1)]; tensor x_299_cast_fp16_to_fp32 = cast(dtype = x_299_cast_fp16_to_fp32_dtype_0, x = x_299_cast_fp16)[name = tensor("cast_552")]; tensor var_1403 = pow(x = x_299_cast_fp16_to_fp32, y = var_1353_promoted_2)[name = tensor("op_1403")]; tensor var_61_axes_0 = const()[name = tensor("var_61_axes_0"), val = tensor([-1])]; tensor var_61_keep_dims_0 = const()[name = tensor("var_61_keep_dims_0"), val = tensor(true)]; tensor var_61 = reduce_mean(axes = var_61_axes_0, keep_dims = var_61_keep_dims_0, x = var_1403)[name = tensor("var_61")]; tensor var_61_to_fp16_dtype_0 = const()[name = tensor("var_61_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1407_to_fp16 = const()[name = tensor("op_1407_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_61_to_fp16 = cast(dtype = var_61_to_fp16_dtype_0, x = var_61)[name = tensor("cast_551")]; tensor var_1408_cast_fp16 = add(x = var_61_to_fp16, y = var_1407_to_fp16)[name = tensor("op_1408_cast_fp16")]; tensor var_1408_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1408_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1409_epsilon_0 = const()[name = tensor("op_1409_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1408_cast_fp16_to_fp32 = cast(dtype = var_1408_cast_fp16_to_fp32_dtype_0, x = var_1408_cast_fp16)[name = tensor("cast_550")]; tensor var_1409 = rsqrt(epsilon = var_1409_epsilon_0, x = var_1408_cast_fp16_to_fp32)[name = tensor("op_1409")]; tensor var_1409_to_fp16_dtype_0 = const()[name = tensor("op_1409_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1409_to_fp16 = cast(dtype = var_1409_to_fp16_dtype_0, x = var_1409)[name = tensor("cast_549")]; tensor x_305_cast_fp16 = mul(x = x_299_cast_fp16, y = var_1409_to_fp16)[name = tensor("x_305_cast_fp16")]; tensor layers_7_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_7_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(286970240)))]; tensor var_1411_cast_fp16 = mul(x = layers_7_self_attn_k_norm_weight_to_fp16, y = x_305_cast_fp16)[name = tensor("op_1411_cast_fp16")]; tensor k_29_perm_0 = const()[name = tensor("k_29_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_7_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_7_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(286970560)))]; tensor linear_51_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_7_self_attn_v_proj_weight_to_fp16, x = x_289_cast_fp16)[name = tensor("linear_51_cast_fp16")]; tensor var_1415 = const()[name = tensor("op_1415"), val = tensor([1, 1, 4, 128])]; tensor var_1416_cast_fp16 = reshape(shape = var_1415, x = linear_51_cast_fp16)[name = tensor("op_1416_cast_fp16")]; tensor v_15_perm_0 = const()[name = tensor("v_15_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_29_cast_fp16 = transpose(perm = q_29_perm_0, x = var_1395_cast_fp16)[name = tensor("transpose_83")]; tensor var_1420_cast_fp16 = mul(x = q_29_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1420_cast_fp16")]; tensor x1_29_begin_0 = const()[name = tensor("x1_29_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_29_end_0 = const()[name = tensor("x1_29_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_29_end_mask_0 = const()[name = tensor("x1_29_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_29_cast_fp16 = slice_by_index(begin = x1_29_begin_0, end = x1_29_end_0, end_mask = x1_29_end_mask_0, x = q_29_cast_fp16)[name = tensor("x1_29_cast_fp16")]; tensor x2_29_begin_0 = const()[name = tensor("x2_29_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_29_end_0 = const()[name = tensor("x2_29_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_29_end_mask_0 = const()[name = tensor("x2_29_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_29_cast_fp16 = slice_by_index(begin = x2_29_begin_0, end = x2_29_end_0, end_mask = x2_29_end_mask_0, x = q_29_cast_fp16)[name = tensor("x2_29_cast_fp16")]; tensor const_15_promoted_to_fp16 = const()[name = tensor("const_15_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1423_cast_fp16 = mul(x = x2_29_cast_fp16, y = const_15_promoted_to_fp16)[name = tensor("op_1423_cast_fp16")]; tensor var_1425_interleave_0 = const()[name = tensor("op_1425_interleave_0"), val = tensor(false)]; tensor var_1425_cast_fp16 = concat(axis = var_1354, interleave = var_1425_interleave_0, values = (var_1423_cast_fp16, x1_29_cast_fp16))[name = tensor("op_1425_cast_fp16")]; tensor var_1426_cast_fp16 = mul(x = var_1425_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1426_cast_fp16")]; tensor q_31_cast_fp16 = add(x = var_1420_cast_fp16, y = var_1426_cast_fp16)[name = tensor("q_31_cast_fp16")]; tensor k_29_cast_fp16 = transpose(perm = k_29_perm_0, x = var_1411_cast_fp16)[name = tensor("transpose_82")]; tensor var_1428_cast_fp16 = mul(x = k_29_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1428_cast_fp16")]; tensor x1_31_begin_0 = const()[name = tensor("x1_31_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_31_end_0 = const()[name = tensor("x1_31_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_31_end_mask_0 = const()[name = tensor("x1_31_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_31_cast_fp16 = slice_by_index(begin = x1_31_begin_0, end = x1_31_end_0, end_mask = x1_31_end_mask_0, x = k_29_cast_fp16)[name = tensor("x1_31_cast_fp16")]; tensor x2_31_begin_0 = const()[name = tensor("x2_31_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_31_end_0 = const()[name = tensor("x2_31_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_31_end_mask_0 = const()[name = tensor("x2_31_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_31_cast_fp16 = slice_by_index(begin = x2_31_begin_0, end = x2_31_end_0, end_mask = x2_31_end_mask_0, x = k_29_cast_fp16)[name = tensor("x2_31_cast_fp16")]; tensor const_16_promoted_to_fp16 = const()[name = tensor("const_16_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1431_cast_fp16 = mul(x = x2_31_cast_fp16, y = const_16_promoted_to_fp16)[name = tensor("op_1431_cast_fp16")]; tensor var_1433_interleave_0 = const()[name = tensor("op_1433_interleave_0"), val = tensor(false)]; tensor var_1433_cast_fp16 = concat(axis = var_1354, interleave = var_1433_interleave_0, values = (var_1431_cast_fp16, x1_31_cast_fp16))[name = tensor("op_1433_cast_fp16")]; tensor var_1434_cast_fp16 = mul(x = var_1433_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1434_cast_fp16")]; tensor k_31_cast_fp16 = add(x = var_1428_cast_fp16, y = var_1434_cast_fp16)[name = tensor("k_31_cast_fp16")]; tensor var_1437_cast_fp16 = mul(x = k_cache_15_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1437_cast_fp16")]; tensor var_1438_cast_fp16 = mul(x = k_31_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1438_cast_fp16")]; tensor k_full_15_cast_fp16 = add(x = var_1437_cast_fp16, y = var_1438_cast_fp16)[name = tensor("k_full_15_cast_fp16")]; tensor var_1441_cast_fp16 = mul(x = v_cache_15_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1441_cast_fp16")]; tensor v_15_cast_fp16 = transpose(perm = v_15_perm_0, x = var_1416_cast_fp16)[name = tensor("transpose_81")]; tensor var_1442_cast_fp16 = mul(x = v_15_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1442_cast_fp16")]; tensor v_full_15_cast_fp16 = add(x = var_1441_cast_fp16, y = var_1442_cast_fp16)[name = tensor("v_full_15_cast_fp16")]; tensor var_1444_axes_0 = const()[name = tensor("op_1444_axes_0"), val = tensor([2])]; tensor var_1444_cast_fp16 = expand_dims(axes = var_1444_axes_0, x = k_full_15_cast_fp16)[name = tensor("op_1444_cast_fp16")]; tensor var_1446_reps_0 = const()[name = tensor("op_1446_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1446_cast_fp16 = tile(reps = var_1446_reps_0, x = var_1444_cast_fp16)[name = tensor("op_1446_cast_fp16")]; tensor var_1447 = const()[name = tensor("op_1447"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_15_cast_fp16 = reshape(shape = var_1447, x = var_1446_cast_fp16)[name = tensor("k_rep_15_cast_fp16")]; tensor var_1449_axes_0 = const()[name = tensor("op_1449_axes_0"), val = tensor([2])]; tensor var_1449_cast_fp16 = expand_dims(axes = var_1449_axes_0, x = v_full_15_cast_fp16)[name = tensor("op_1449_cast_fp16")]; tensor var_1451_reps_0 = const()[name = tensor("op_1451_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1451_cast_fp16 = tile(reps = var_1451_reps_0, x = var_1449_cast_fp16)[name = tensor("op_1451_cast_fp16")]; tensor var_1452 = const()[name = tensor("op_1452"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_15_cast_fp16 = reshape(shape = var_1452, x = var_1451_cast_fp16)[name = tensor("v_rep_15_cast_fp16")]; tensor var_1455_transpose_x_1 = const()[name = tensor("op_1455_transpose_x_1"), val = tensor(false)]; tensor var_1455_transpose_y_1 = const()[name = tensor("op_1455_transpose_y_1"), val = tensor(true)]; tensor var_1455_cast_fp16 = matmul(transpose_x = var_1455_transpose_x_1, transpose_y = var_1455_transpose_y_1, x = q_31_cast_fp16, y = k_rep_15_cast_fp16)[name = tensor("op_1455_cast_fp16")]; tensor var_1456_to_fp16 = const()[name = tensor("op_1456_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_29_cast_fp16 = mul(x = var_1455_cast_fp16, y = var_1456_to_fp16)[name = tensor("attn_29_cast_fp16")]; tensor input_31_cast_fp16 = add(x = attn_29_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_31_cast_fp16")]; tensor input_31_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_31_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_31_cast_fp16_to_fp32 = cast(dtype = input_31_cast_fp16_to_fp32_dtype_0, x = input_31_cast_fp16)[name = tensor("cast_548")]; tensor attn_31 = softmax(axis = var_1354, x = input_31_cast_fp16_to_fp32)[name = tensor("attn_31")]; tensor out_15_transpose_x_0 = const()[name = tensor("out_15_transpose_x_0"), val = tensor(false)]; tensor out_15_transpose_y_0 = const()[name = tensor("out_15_transpose_y_0"), val = tensor(false)]; tensor attn_31_to_fp16_dtype_0 = const()[name = tensor("attn_31_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_31_to_fp16 = cast(dtype = attn_31_to_fp16_dtype_0, x = attn_31)[name = tensor("cast_547")]; tensor out_15_cast_fp16 = matmul(transpose_x = out_15_transpose_x_0, transpose_y = out_15_transpose_y_0, x = attn_31_to_fp16, y = v_rep_15_cast_fp16)[name = tensor("out_15_cast_fp16")]; tensor var_1461_perm_0 = const()[name = tensor("op_1461_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1463 = const()[name = tensor("op_1463"), val = tensor([1, 1, 1536])]; tensor var_1461_cast_fp16 = transpose(perm = var_1461_perm_0, x = out_15_cast_fp16)[name = tensor("transpose_80")]; tensor x_307_cast_fp16 = reshape(shape = var_1463, x = var_1461_cast_fp16)[name = tensor("x_307_cast_fp16")]; tensor layers_7_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_7_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(287494912)))]; tensor linear_52_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_7_self_attn_o_proj_weight_to_fp16, x = x_307_cast_fp16)[name = tensor("linear_52_cast_fp16")]; tensor x_309_cast_fp16 = add(x = x_281_cast_fp16, y = linear_52_cast_fp16)[name = tensor("x_309_cast_fp16")]; tensor x_309_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_309_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1353_promoted_3 = const()[name = tensor("op_1353_promoted_3"), val = tensor(0x1p+1)]; tensor x_309_cast_fp16_to_fp32 = cast(dtype = x_309_cast_fp16_to_fp32_dtype_0, x = x_309_cast_fp16)[name = tensor("cast_546")]; tensor var_1474 = pow(x = x_309_cast_fp16_to_fp32, y = var_1353_promoted_3)[name = tensor("op_1474")]; tensor var_63_axes_0 = const()[name = tensor("var_63_axes_0"), val = tensor([-1])]; tensor var_63_keep_dims_0 = const()[name = tensor("var_63_keep_dims_0"), val = tensor(true)]; tensor var_63 = reduce_mean(axes = var_63_axes_0, keep_dims = var_63_keep_dims_0, x = var_1474)[name = tensor("var_63")]; tensor var_63_to_fp16_dtype_0 = const()[name = tensor("var_63_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1478_to_fp16 = const()[name = tensor("op_1478_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_63_to_fp16 = cast(dtype = var_63_to_fp16_dtype_0, x = var_63)[name = tensor("cast_545")]; tensor var_1479_cast_fp16 = add(x = var_63_to_fp16, y = var_1478_to_fp16)[name = tensor("op_1479_cast_fp16")]; tensor var_1479_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1479_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1480_epsilon_0 = const()[name = tensor("op_1480_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1479_cast_fp16_to_fp32 = cast(dtype = var_1479_cast_fp16_to_fp32_dtype_0, x = var_1479_cast_fp16)[name = tensor("cast_544")]; tensor var_1480 = rsqrt(epsilon = var_1480_epsilon_0, x = var_1479_cast_fp16_to_fp32)[name = tensor("op_1480")]; tensor var_1480_to_fp16_dtype_0 = const()[name = tensor("op_1480_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1480_to_fp16 = cast(dtype = var_1480_to_fp16_dtype_0, x = var_1480)[name = tensor("cast_543")]; tensor x_315_cast_fp16 = mul(x = x_309_cast_fp16, y = var_1480_to_fp16)[name = tensor("x_315_cast_fp16")]; tensor layers_7_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_7_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(289067840)))]; tensor x_317_cast_fp16 = mul(x = layers_7_post_attention_layernorm_weight_to_fp16, y = x_315_cast_fp16)[name = tensor("x_317_cast_fp16")]; tensor layers_7_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_7_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(289068928)))]; tensor linear_53_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_7_mlp_gate_proj_weight_to_fp16, x = x_317_cast_fp16)[name = tensor("linear_53_cast_fp16")]; tensor var_1491_cast_fp16 = silu(x = linear_53_cast_fp16)[name = tensor("op_1491_cast_fp16")]; tensor layers_7_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_7_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(290641856)))]; tensor linear_54_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_7_mlp_up_proj_weight_to_fp16, x = x_317_cast_fp16)[name = tensor("linear_54_cast_fp16")]; tensor x_319_cast_fp16 = mul(x = var_1491_cast_fp16, y = linear_54_cast_fp16)[name = tensor("x_319_cast_fp16")]; tensor layers_7_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_7_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(292214784)))]; tensor linear_55_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_7_mlp_down_proj_weight_to_fp16, x = x_319_cast_fp16)[name = tensor("linear_55_cast_fp16")]; tensor x_321_cast_fp16 = add(x = x_309_cast_fp16, y = linear_55_cast_fp16)[name = tensor("x_321_cast_fp16")]; tensor x_321_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_321_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_17_begin_0 = const()[name = tensor("k_cache_17_begin_0"), val = tensor([8, 0, 0, 0, 0])]; tensor k_cache_17_end_0 = const()[name = tensor("k_cache_17_end_0"), val = tensor([9, 1, 4, 2048, 128])]; tensor k_cache_17_end_mask_0 = const()[name = tensor("k_cache_17_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_17_squeeze_mask_0 = const()[name = tensor("k_cache_17_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_17_cast_fp16 = slice_by_index(begin = k_cache_17_begin_0, end = k_cache_17_end_0, end_mask = k_cache_17_end_mask_0, squeeze_mask = k_cache_17_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_17_cast_fp16")]; tensor v_cache_17_begin_0 = const()[name = tensor("v_cache_17_begin_0"), val = tensor([8, 0, 0, 0, 0])]; tensor v_cache_17_end_0 = const()[name = tensor("v_cache_17_end_0"), val = tensor([9, 1, 4, 2048, 128])]; tensor v_cache_17_end_mask_0 = const()[name = tensor("v_cache_17_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_17_squeeze_mask_0 = const()[name = tensor("v_cache_17_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_17_cast_fp16 = slice_by_index(begin = v_cache_17_begin_0, end = v_cache_17_end_0, end_mask = v_cache_17_end_mask_0, squeeze_mask = v_cache_17_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_17_cast_fp16")]; tensor var_1523 = const()[name = tensor("op_1523"), val = tensor(-1)]; tensor var_1522_promoted = const()[name = tensor("op_1522_promoted"), val = tensor(0x1p+1)]; tensor x_321_cast_fp16_to_fp32 = cast(dtype = x_321_cast_fp16_to_fp32_dtype_0, x = x_321_cast_fp16)[name = tensor("cast_542")]; tensor var_1532 = pow(x = x_321_cast_fp16_to_fp32, y = var_1522_promoted)[name = tensor("op_1532")]; tensor var_65_axes_0 = const()[name = tensor("var_65_axes_0"), val = tensor([-1])]; tensor var_65_keep_dims_0 = const()[name = tensor("var_65_keep_dims_0"), val = tensor(true)]; tensor var_65 = reduce_mean(axes = var_65_axes_0, keep_dims = var_65_keep_dims_0, x = var_1532)[name = tensor("var_65")]; tensor var_65_to_fp16_dtype_0 = const()[name = tensor("var_65_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1536_to_fp16 = const()[name = tensor("op_1536_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_65_to_fp16 = cast(dtype = var_65_to_fp16_dtype_0, x = var_65)[name = tensor("cast_541")]; tensor var_1537_cast_fp16 = add(x = var_65_to_fp16, y = var_1536_to_fp16)[name = tensor("op_1537_cast_fp16")]; tensor var_1537_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1537_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1538_epsilon_0 = const()[name = tensor("op_1538_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1537_cast_fp16_to_fp32 = cast(dtype = var_1537_cast_fp16_to_fp32_dtype_0, x = var_1537_cast_fp16)[name = tensor("cast_540")]; tensor var_1538 = rsqrt(epsilon = var_1538_epsilon_0, x = var_1537_cast_fp16_to_fp32)[name = tensor("op_1538")]; tensor var_1538_to_fp16_dtype_0 = const()[name = tensor("op_1538_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1538_to_fp16 = cast(dtype = var_1538_to_fp16_dtype_0, x = var_1538)[name = tensor("cast_539")]; tensor x_327_cast_fp16 = mul(x = x_321_cast_fp16, y = var_1538_to_fp16)[name = tensor("x_327_cast_fp16")]; tensor layers_8_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_8_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(293787712)))]; tensor x_329_cast_fp16 = mul(x = layers_8_input_layernorm_weight_to_fp16, y = x_327_cast_fp16)[name = tensor("x_329_cast_fp16")]; tensor layers_8_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_8_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(293788800)))]; tensor linear_56_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_8_self_attn_q_proj_weight_to_fp16, x = x_329_cast_fp16)[name = tensor("linear_56_cast_fp16")]; tensor var_1552 = const()[name = tensor("op_1552"), val = tensor([1, 1, 12, 128])]; tensor x_331_cast_fp16 = reshape(shape = var_1552, x = linear_56_cast_fp16)[name = tensor("x_331_cast_fp16")]; tensor x_331_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_331_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1522_promoted_1 = const()[name = tensor("op_1522_promoted_1"), val = tensor(0x1p+1)]; tensor x_331_cast_fp16_to_fp32 = cast(dtype = x_331_cast_fp16_to_fp32_dtype_0, x = x_331_cast_fp16)[name = tensor("cast_538")]; tensor var_1556 = pow(x = x_331_cast_fp16_to_fp32, y = var_1522_promoted_1)[name = tensor("op_1556")]; tensor var_67_axes_0 = const()[name = tensor("var_67_axes_0"), val = tensor([-1])]; tensor var_67_keep_dims_0 = const()[name = tensor("var_67_keep_dims_0"), val = tensor(true)]; tensor var_67 = reduce_mean(axes = var_67_axes_0, keep_dims = var_67_keep_dims_0, x = var_1556)[name = tensor("var_67")]; tensor var_67_to_fp16_dtype_0 = const()[name = tensor("var_67_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1560_to_fp16 = const()[name = tensor("op_1560_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_67_to_fp16 = cast(dtype = var_67_to_fp16_dtype_0, x = var_67)[name = tensor("cast_537")]; tensor var_1561_cast_fp16 = add(x = var_67_to_fp16, y = var_1560_to_fp16)[name = tensor("op_1561_cast_fp16")]; tensor var_1561_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1561_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1562_epsilon_0 = const()[name = tensor("op_1562_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1561_cast_fp16_to_fp32 = cast(dtype = var_1561_cast_fp16_to_fp32_dtype_0, x = var_1561_cast_fp16)[name = tensor("cast_536")]; tensor var_1562 = rsqrt(epsilon = var_1562_epsilon_0, x = var_1561_cast_fp16_to_fp32)[name = tensor("op_1562")]; tensor var_1562_to_fp16_dtype_0 = const()[name = tensor("op_1562_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1562_to_fp16 = cast(dtype = var_1562_to_fp16_dtype_0, x = var_1562)[name = tensor("cast_535")]; tensor x_337_cast_fp16 = mul(x = x_331_cast_fp16, y = var_1562_to_fp16)[name = tensor("x_337_cast_fp16")]; tensor layers_8_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_8_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(295361728)))]; tensor var_1564_cast_fp16 = mul(x = layers_8_self_attn_q_norm_weight_to_fp16, y = x_337_cast_fp16)[name = tensor("op_1564_cast_fp16")]; tensor q_33_perm_0 = const()[name = tensor("q_33_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_8_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_8_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(295362048)))]; tensor linear_57_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_8_self_attn_k_proj_weight_to_fp16, x = x_329_cast_fp16)[name = tensor("linear_57_cast_fp16")]; tensor var_1568 = const()[name = tensor("op_1568"), val = tensor([1, 1, 4, 128])]; tensor x_339_cast_fp16 = reshape(shape = var_1568, x = linear_57_cast_fp16)[name = tensor("x_339_cast_fp16")]; tensor x_339_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_339_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1522_promoted_2 = const()[name = tensor("op_1522_promoted_2"), val = tensor(0x1p+1)]; tensor x_339_cast_fp16_to_fp32 = cast(dtype = x_339_cast_fp16_to_fp32_dtype_0, x = x_339_cast_fp16)[name = tensor("cast_534")]; tensor var_1572 = pow(x = x_339_cast_fp16_to_fp32, y = var_1522_promoted_2)[name = tensor("op_1572")]; tensor var_69_axes_0 = const()[name = tensor("var_69_axes_0"), val = tensor([-1])]; tensor var_69_keep_dims_0 = const()[name = tensor("var_69_keep_dims_0"), val = tensor(true)]; tensor var_69 = reduce_mean(axes = var_69_axes_0, keep_dims = var_69_keep_dims_0, x = var_1572)[name = tensor("var_69")]; tensor var_69_to_fp16_dtype_0 = const()[name = tensor("var_69_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1576_to_fp16 = const()[name = tensor("op_1576_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_69_to_fp16 = cast(dtype = var_69_to_fp16_dtype_0, x = var_69)[name = tensor("cast_533")]; tensor var_1577_cast_fp16 = add(x = var_69_to_fp16, y = var_1576_to_fp16)[name = tensor("op_1577_cast_fp16")]; tensor var_1577_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1577_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1578_epsilon_0 = const()[name = tensor("op_1578_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1577_cast_fp16_to_fp32 = cast(dtype = var_1577_cast_fp16_to_fp32_dtype_0, x = var_1577_cast_fp16)[name = tensor("cast_532")]; tensor var_1578 = rsqrt(epsilon = var_1578_epsilon_0, x = var_1577_cast_fp16_to_fp32)[name = tensor("op_1578")]; tensor var_1578_to_fp16_dtype_0 = const()[name = tensor("op_1578_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1578_to_fp16 = cast(dtype = var_1578_to_fp16_dtype_0, x = var_1578)[name = tensor("cast_531")]; tensor x_345_cast_fp16 = mul(x = x_339_cast_fp16, y = var_1578_to_fp16)[name = tensor("x_345_cast_fp16")]; tensor layers_8_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_8_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(295886400)))]; tensor var_1580_cast_fp16 = mul(x = layers_8_self_attn_k_norm_weight_to_fp16, y = x_345_cast_fp16)[name = tensor("op_1580_cast_fp16")]; tensor k_33_perm_0 = const()[name = tensor("k_33_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_8_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_8_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(295886720)))]; tensor linear_58_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_8_self_attn_v_proj_weight_to_fp16, x = x_329_cast_fp16)[name = tensor("linear_58_cast_fp16")]; tensor var_1584 = const()[name = tensor("op_1584"), val = tensor([1, 1, 4, 128])]; tensor var_1585_cast_fp16 = reshape(shape = var_1584, x = linear_58_cast_fp16)[name = tensor("op_1585_cast_fp16")]; tensor v_17_perm_0 = const()[name = tensor("v_17_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_33_cast_fp16 = transpose(perm = q_33_perm_0, x = var_1564_cast_fp16)[name = tensor("transpose_79")]; tensor var_1589_cast_fp16 = mul(x = q_33_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1589_cast_fp16")]; tensor x1_33_begin_0 = const()[name = tensor("x1_33_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_33_end_0 = const()[name = tensor("x1_33_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_33_end_mask_0 = const()[name = tensor("x1_33_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_33_cast_fp16 = slice_by_index(begin = x1_33_begin_0, end = x1_33_end_0, end_mask = x1_33_end_mask_0, x = q_33_cast_fp16)[name = tensor("x1_33_cast_fp16")]; tensor x2_33_begin_0 = const()[name = tensor("x2_33_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_33_end_0 = const()[name = tensor("x2_33_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_33_end_mask_0 = const()[name = tensor("x2_33_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_33_cast_fp16 = slice_by_index(begin = x2_33_begin_0, end = x2_33_end_0, end_mask = x2_33_end_mask_0, x = q_33_cast_fp16)[name = tensor("x2_33_cast_fp16")]; tensor const_17_promoted_to_fp16 = const()[name = tensor("const_17_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1592_cast_fp16 = mul(x = x2_33_cast_fp16, y = const_17_promoted_to_fp16)[name = tensor("op_1592_cast_fp16")]; tensor var_1594_interleave_0 = const()[name = tensor("op_1594_interleave_0"), val = tensor(false)]; tensor var_1594_cast_fp16 = concat(axis = var_1523, interleave = var_1594_interleave_0, values = (var_1592_cast_fp16, x1_33_cast_fp16))[name = tensor("op_1594_cast_fp16")]; tensor var_1595_cast_fp16 = mul(x = var_1594_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1595_cast_fp16")]; tensor q_35_cast_fp16 = add(x = var_1589_cast_fp16, y = var_1595_cast_fp16)[name = tensor("q_35_cast_fp16")]; tensor k_33_cast_fp16 = transpose(perm = k_33_perm_0, x = var_1580_cast_fp16)[name = tensor("transpose_78")]; tensor var_1597_cast_fp16 = mul(x = k_33_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1597_cast_fp16")]; tensor x1_35_begin_0 = const()[name = tensor("x1_35_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_35_end_0 = const()[name = tensor("x1_35_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_35_end_mask_0 = const()[name = tensor("x1_35_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_35_cast_fp16 = slice_by_index(begin = x1_35_begin_0, end = x1_35_end_0, end_mask = x1_35_end_mask_0, x = k_33_cast_fp16)[name = tensor("x1_35_cast_fp16")]; tensor x2_35_begin_0 = const()[name = tensor("x2_35_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_35_end_0 = const()[name = tensor("x2_35_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_35_end_mask_0 = const()[name = tensor("x2_35_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_35_cast_fp16 = slice_by_index(begin = x2_35_begin_0, end = x2_35_end_0, end_mask = x2_35_end_mask_0, x = k_33_cast_fp16)[name = tensor("x2_35_cast_fp16")]; tensor const_18_promoted_to_fp16 = const()[name = tensor("const_18_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1600_cast_fp16 = mul(x = x2_35_cast_fp16, y = const_18_promoted_to_fp16)[name = tensor("op_1600_cast_fp16")]; tensor var_1602_interleave_0 = const()[name = tensor("op_1602_interleave_0"), val = tensor(false)]; tensor var_1602_cast_fp16 = concat(axis = var_1523, interleave = var_1602_interleave_0, values = (var_1600_cast_fp16, x1_35_cast_fp16))[name = tensor("op_1602_cast_fp16")]; tensor var_1603_cast_fp16 = mul(x = var_1602_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1603_cast_fp16")]; tensor k_35_cast_fp16 = add(x = var_1597_cast_fp16, y = var_1603_cast_fp16)[name = tensor("k_35_cast_fp16")]; tensor var_1606_cast_fp16 = mul(x = k_cache_17_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1606_cast_fp16")]; tensor var_1607_cast_fp16 = mul(x = k_35_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1607_cast_fp16")]; tensor k_full_17_cast_fp16 = add(x = var_1606_cast_fp16, y = var_1607_cast_fp16)[name = tensor("k_full_17_cast_fp16")]; tensor var_1610_cast_fp16 = mul(x = v_cache_17_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1610_cast_fp16")]; tensor v_17_cast_fp16 = transpose(perm = v_17_perm_0, x = var_1585_cast_fp16)[name = tensor("transpose_77")]; tensor var_1611_cast_fp16 = mul(x = v_17_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1611_cast_fp16")]; tensor v_full_17_cast_fp16 = add(x = var_1610_cast_fp16, y = var_1611_cast_fp16)[name = tensor("v_full_17_cast_fp16")]; tensor var_1613_axes_0 = const()[name = tensor("op_1613_axes_0"), val = tensor([2])]; tensor var_1613_cast_fp16 = expand_dims(axes = var_1613_axes_0, x = k_full_17_cast_fp16)[name = tensor("op_1613_cast_fp16")]; tensor var_1615_reps_0 = const()[name = tensor("op_1615_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1615_cast_fp16 = tile(reps = var_1615_reps_0, x = var_1613_cast_fp16)[name = tensor("op_1615_cast_fp16")]; tensor var_1616 = const()[name = tensor("op_1616"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_17_cast_fp16 = reshape(shape = var_1616, x = var_1615_cast_fp16)[name = tensor("k_rep_17_cast_fp16")]; tensor var_1618_axes_0 = const()[name = tensor("op_1618_axes_0"), val = tensor([2])]; tensor var_1618_cast_fp16 = expand_dims(axes = var_1618_axes_0, x = v_full_17_cast_fp16)[name = tensor("op_1618_cast_fp16")]; tensor var_1620_reps_0 = const()[name = tensor("op_1620_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1620_cast_fp16 = tile(reps = var_1620_reps_0, x = var_1618_cast_fp16)[name = tensor("op_1620_cast_fp16")]; tensor var_1621 = const()[name = tensor("op_1621"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_17_cast_fp16 = reshape(shape = var_1621, x = var_1620_cast_fp16)[name = tensor("v_rep_17_cast_fp16")]; tensor var_1624_transpose_x_1 = const()[name = tensor("op_1624_transpose_x_1"), val = tensor(false)]; tensor var_1624_transpose_y_1 = const()[name = tensor("op_1624_transpose_y_1"), val = tensor(true)]; tensor var_1624_cast_fp16 = matmul(transpose_x = var_1624_transpose_x_1, transpose_y = var_1624_transpose_y_1, x = q_35_cast_fp16, y = k_rep_17_cast_fp16)[name = tensor("op_1624_cast_fp16")]; tensor var_1625_to_fp16 = const()[name = tensor("op_1625_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_33_cast_fp16 = mul(x = var_1624_cast_fp16, y = var_1625_to_fp16)[name = tensor("attn_33_cast_fp16")]; tensor input_35_cast_fp16 = add(x = attn_33_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_35_cast_fp16")]; tensor input_35_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_35_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_35_cast_fp16_to_fp32 = cast(dtype = input_35_cast_fp16_to_fp32_dtype_0, x = input_35_cast_fp16)[name = tensor("cast_530")]; tensor attn_35 = softmax(axis = var_1523, x = input_35_cast_fp16_to_fp32)[name = tensor("attn_35")]; tensor out_17_transpose_x_0 = const()[name = tensor("out_17_transpose_x_0"), val = tensor(false)]; tensor out_17_transpose_y_0 = const()[name = tensor("out_17_transpose_y_0"), val = tensor(false)]; tensor attn_35_to_fp16_dtype_0 = const()[name = tensor("attn_35_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_35_to_fp16 = cast(dtype = attn_35_to_fp16_dtype_0, x = attn_35)[name = tensor("cast_529")]; tensor out_17_cast_fp16 = matmul(transpose_x = out_17_transpose_x_0, transpose_y = out_17_transpose_y_0, x = attn_35_to_fp16, y = v_rep_17_cast_fp16)[name = tensor("out_17_cast_fp16")]; tensor var_1630_perm_0 = const()[name = tensor("op_1630_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1632 = const()[name = tensor("op_1632"), val = tensor([1, 1, 1536])]; tensor var_1630_cast_fp16 = transpose(perm = var_1630_perm_0, x = out_17_cast_fp16)[name = tensor("transpose_76")]; tensor x_347_cast_fp16 = reshape(shape = var_1632, x = var_1630_cast_fp16)[name = tensor("x_347_cast_fp16")]; tensor layers_8_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_8_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(296411072)))]; tensor linear_59_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_8_self_attn_o_proj_weight_to_fp16, x = x_347_cast_fp16)[name = tensor("linear_59_cast_fp16")]; tensor x_349_cast_fp16 = add(x = x_321_cast_fp16, y = linear_59_cast_fp16)[name = tensor("x_349_cast_fp16")]; tensor x_349_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_349_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1522_promoted_3 = const()[name = tensor("op_1522_promoted_3"), val = tensor(0x1p+1)]; tensor x_349_cast_fp16_to_fp32 = cast(dtype = x_349_cast_fp16_to_fp32_dtype_0, x = x_349_cast_fp16)[name = tensor("cast_528")]; tensor var_1643 = pow(x = x_349_cast_fp16_to_fp32, y = var_1522_promoted_3)[name = tensor("op_1643")]; tensor var_71_axes_0 = const()[name = tensor("var_71_axes_0"), val = tensor([-1])]; tensor var_71_keep_dims_0 = const()[name = tensor("var_71_keep_dims_0"), val = tensor(true)]; tensor var_71 = reduce_mean(axes = var_71_axes_0, keep_dims = var_71_keep_dims_0, x = var_1643)[name = tensor("var_71")]; tensor var_71_to_fp16_dtype_0 = const()[name = tensor("var_71_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1647_to_fp16 = const()[name = tensor("op_1647_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_71_to_fp16 = cast(dtype = var_71_to_fp16_dtype_0, x = var_71)[name = tensor("cast_527")]; tensor var_1648_cast_fp16 = add(x = var_71_to_fp16, y = var_1647_to_fp16)[name = tensor("op_1648_cast_fp16")]; tensor var_1648_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1648_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1649_epsilon_0 = const()[name = tensor("op_1649_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1648_cast_fp16_to_fp32 = cast(dtype = var_1648_cast_fp16_to_fp32_dtype_0, x = var_1648_cast_fp16)[name = tensor("cast_526")]; tensor var_1649 = rsqrt(epsilon = var_1649_epsilon_0, x = var_1648_cast_fp16_to_fp32)[name = tensor("op_1649")]; tensor var_1649_to_fp16_dtype_0 = const()[name = tensor("op_1649_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1649_to_fp16 = cast(dtype = var_1649_to_fp16_dtype_0, x = var_1649)[name = tensor("cast_525")]; tensor x_355_cast_fp16 = mul(x = x_349_cast_fp16, y = var_1649_to_fp16)[name = tensor("x_355_cast_fp16")]; tensor layers_8_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_8_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297984000)))]; tensor x_357_cast_fp16 = mul(x = layers_8_post_attention_layernorm_weight_to_fp16, y = x_355_cast_fp16)[name = tensor("x_357_cast_fp16")]; tensor layers_8_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_8_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297985088)))]; tensor linear_60_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_8_mlp_gate_proj_weight_to_fp16, x = x_357_cast_fp16)[name = tensor("linear_60_cast_fp16")]; tensor var_1660_cast_fp16 = silu(x = linear_60_cast_fp16)[name = tensor("op_1660_cast_fp16")]; tensor layers_8_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_8_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(299558016)))]; tensor linear_61_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_8_mlp_up_proj_weight_to_fp16, x = x_357_cast_fp16)[name = tensor("linear_61_cast_fp16")]; tensor x_359_cast_fp16 = mul(x = var_1660_cast_fp16, y = linear_61_cast_fp16)[name = tensor("x_359_cast_fp16")]; tensor layers_8_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_8_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(301130944)))]; tensor linear_62_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_8_mlp_down_proj_weight_to_fp16, x = x_359_cast_fp16)[name = tensor("linear_62_cast_fp16")]; tensor x_361_cast_fp16 = add(x = x_349_cast_fp16, y = linear_62_cast_fp16)[name = tensor("x_361_cast_fp16")]; tensor x_361_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_361_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_19_begin_0 = const()[name = tensor("k_cache_19_begin_0"), val = tensor([9, 0, 0, 0, 0])]; tensor k_cache_19_end_0 = const()[name = tensor("k_cache_19_end_0"), val = tensor([10, 1, 4, 2048, 128])]; tensor k_cache_19_end_mask_0 = const()[name = tensor("k_cache_19_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_19_squeeze_mask_0 = const()[name = tensor("k_cache_19_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_19_cast_fp16 = slice_by_index(begin = k_cache_19_begin_0, end = k_cache_19_end_0, end_mask = k_cache_19_end_mask_0, squeeze_mask = k_cache_19_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_19_cast_fp16")]; tensor v_cache_19_begin_0 = const()[name = tensor("v_cache_19_begin_0"), val = tensor([9, 0, 0, 0, 0])]; tensor v_cache_19_end_0 = const()[name = tensor("v_cache_19_end_0"), val = tensor([10, 1, 4, 2048, 128])]; tensor v_cache_19_end_mask_0 = const()[name = tensor("v_cache_19_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_19_squeeze_mask_0 = const()[name = tensor("v_cache_19_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_19_cast_fp16 = slice_by_index(begin = v_cache_19_begin_0, end = v_cache_19_end_0, end_mask = v_cache_19_end_mask_0, squeeze_mask = v_cache_19_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_19_cast_fp16")]; tensor var_1692 = const()[name = tensor("op_1692"), val = tensor(-1)]; tensor var_1691_promoted = const()[name = tensor("op_1691_promoted"), val = tensor(0x1p+1)]; tensor x_361_cast_fp16_to_fp32 = cast(dtype = x_361_cast_fp16_to_fp32_dtype_0, x = x_361_cast_fp16)[name = tensor("cast_524")]; tensor var_1701 = pow(x = x_361_cast_fp16_to_fp32, y = var_1691_promoted)[name = tensor("op_1701")]; tensor var_73_axes_0 = const()[name = tensor("var_73_axes_0"), val = tensor([-1])]; tensor var_73_keep_dims_0 = const()[name = tensor("var_73_keep_dims_0"), val = tensor(true)]; tensor var_73 = reduce_mean(axes = var_73_axes_0, keep_dims = var_73_keep_dims_0, x = var_1701)[name = tensor("var_73")]; tensor var_73_to_fp16_dtype_0 = const()[name = tensor("var_73_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1705_to_fp16 = const()[name = tensor("op_1705_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_73_to_fp16 = cast(dtype = var_73_to_fp16_dtype_0, x = var_73)[name = tensor("cast_523")]; tensor var_1706_cast_fp16 = add(x = var_73_to_fp16, y = var_1705_to_fp16)[name = tensor("op_1706_cast_fp16")]; tensor var_1706_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1706_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1707_epsilon_0 = const()[name = tensor("op_1707_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1706_cast_fp16_to_fp32 = cast(dtype = var_1706_cast_fp16_to_fp32_dtype_0, x = var_1706_cast_fp16)[name = tensor("cast_522")]; tensor var_1707 = rsqrt(epsilon = var_1707_epsilon_0, x = var_1706_cast_fp16_to_fp32)[name = tensor("op_1707")]; tensor var_1707_to_fp16_dtype_0 = const()[name = tensor("op_1707_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1707_to_fp16 = cast(dtype = var_1707_to_fp16_dtype_0, x = var_1707)[name = tensor("cast_521")]; tensor x_367_cast_fp16 = mul(x = x_361_cast_fp16, y = var_1707_to_fp16)[name = tensor("x_367_cast_fp16")]; tensor layers_9_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_9_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(302703872)))]; tensor x_369_cast_fp16 = mul(x = layers_9_input_layernorm_weight_to_fp16, y = x_367_cast_fp16)[name = tensor("x_369_cast_fp16")]; tensor layers_9_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_9_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(302704960)))]; tensor linear_63_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_9_self_attn_q_proj_weight_to_fp16, x = x_369_cast_fp16)[name = tensor("linear_63_cast_fp16")]; tensor var_1721 = const()[name = tensor("op_1721"), val = tensor([1, 1, 12, 128])]; tensor x_371_cast_fp16 = reshape(shape = var_1721, x = linear_63_cast_fp16)[name = tensor("x_371_cast_fp16")]; tensor x_371_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_371_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1691_promoted_1 = const()[name = tensor("op_1691_promoted_1"), val = tensor(0x1p+1)]; tensor x_371_cast_fp16_to_fp32 = cast(dtype = x_371_cast_fp16_to_fp32_dtype_0, x = x_371_cast_fp16)[name = tensor("cast_520")]; tensor var_1725 = pow(x = x_371_cast_fp16_to_fp32, y = var_1691_promoted_1)[name = tensor("op_1725")]; tensor var_75_axes_0 = const()[name = tensor("var_75_axes_0"), val = tensor([-1])]; tensor var_75_keep_dims_0 = const()[name = tensor("var_75_keep_dims_0"), val = tensor(true)]; tensor var_75_0 = reduce_mean(axes = var_75_axes_0, keep_dims = var_75_keep_dims_0, x = var_1725)[name = tensor("var_75")]; tensor var_75_to_fp16_dtype_0 = const()[name = tensor("var_75_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1729_to_fp16 = const()[name = tensor("op_1729_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_75_to_fp16 = cast(dtype = var_75_to_fp16_dtype_0, x = var_75_0)[name = tensor("cast_519")]; tensor var_1730_cast_fp16 = add(x = var_75_to_fp16, y = var_1729_to_fp16)[name = tensor("op_1730_cast_fp16")]; tensor var_1730_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1730_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1731_epsilon_0 = const()[name = tensor("op_1731_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1730_cast_fp16_to_fp32 = cast(dtype = var_1730_cast_fp16_to_fp32_dtype_0, x = var_1730_cast_fp16)[name = tensor("cast_518")]; tensor var_1731 = rsqrt(epsilon = var_1731_epsilon_0, x = var_1730_cast_fp16_to_fp32)[name = tensor("op_1731")]; tensor var_1731_to_fp16_dtype_0 = const()[name = tensor("op_1731_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1731_to_fp16 = cast(dtype = var_1731_to_fp16_dtype_0, x = var_1731)[name = tensor("cast_517")]; tensor x_377_cast_fp16 = mul(x = x_371_cast_fp16, y = var_1731_to_fp16)[name = tensor("x_377_cast_fp16")]; tensor layers_9_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_9_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(304277888)))]; tensor var_1733_cast_fp16 = mul(x = layers_9_self_attn_q_norm_weight_to_fp16, y = x_377_cast_fp16)[name = tensor("op_1733_cast_fp16")]; tensor q_37_perm_0 = const()[name = tensor("q_37_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_9_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_9_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(304278208)))]; tensor linear_64_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_9_self_attn_k_proj_weight_to_fp16, x = x_369_cast_fp16)[name = tensor("linear_64_cast_fp16")]; tensor var_1737 = const()[name = tensor("op_1737"), val = tensor([1, 1, 4, 128])]; tensor x_379_cast_fp16 = reshape(shape = var_1737, x = linear_64_cast_fp16)[name = tensor("x_379_cast_fp16")]; tensor x_379_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_379_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1691_promoted_2 = const()[name = tensor("op_1691_promoted_2"), val = tensor(0x1p+1)]; tensor x_379_cast_fp16_to_fp32 = cast(dtype = x_379_cast_fp16_to_fp32_dtype_0, x = x_379_cast_fp16)[name = tensor("cast_516")]; tensor var_1741 = pow(x = x_379_cast_fp16_to_fp32, y = var_1691_promoted_2)[name = tensor("op_1741")]; tensor var_77_axes_0 = const()[name = tensor("var_77_axes_0"), val = tensor([-1])]; tensor var_77_keep_dims_0 = const()[name = tensor("var_77_keep_dims_0"), val = tensor(true)]; tensor var_77 = reduce_mean(axes = var_77_axes_0, keep_dims = var_77_keep_dims_0, x = var_1741)[name = tensor("var_77")]; tensor var_77_to_fp16_dtype_0 = const()[name = tensor("var_77_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1745_to_fp16 = const()[name = tensor("op_1745_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_77_to_fp16 = cast(dtype = var_77_to_fp16_dtype_0, x = var_77)[name = tensor("cast_515")]; tensor var_1746_cast_fp16 = add(x = var_77_to_fp16, y = var_1745_to_fp16)[name = tensor("op_1746_cast_fp16")]; tensor var_1746_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1746_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1747_epsilon_0 = const()[name = tensor("op_1747_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1746_cast_fp16_to_fp32 = cast(dtype = var_1746_cast_fp16_to_fp32_dtype_0, x = var_1746_cast_fp16)[name = tensor("cast_514")]; tensor var_1747 = rsqrt(epsilon = var_1747_epsilon_0, x = var_1746_cast_fp16_to_fp32)[name = tensor("op_1747")]; tensor var_1747_to_fp16_dtype_0 = const()[name = tensor("op_1747_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1747_to_fp16 = cast(dtype = var_1747_to_fp16_dtype_0, x = var_1747)[name = tensor("cast_513")]; tensor x_385_cast_fp16 = mul(x = x_379_cast_fp16, y = var_1747_to_fp16)[name = tensor("x_385_cast_fp16")]; tensor layers_9_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_9_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(304802560)))]; tensor var_1749_cast_fp16 = mul(x = layers_9_self_attn_k_norm_weight_to_fp16, y = x_385_cast_fp16)[name = tensor("op_1749_cast_fp16")]; tensor k_37_perm_0 = const()[name = tensor("k_37_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_9_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_9_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(304802880)))]; tensor linear_65_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_9_self_attn_v_proj_weight_to_fp16, x = x_369_cast_fp16)[name = tensor("linear_65_cast_fp16")]; tensor var_1753 = const()[name = tensor("op_1753"), val = tensor([1, 1, 4, 128])]; tensor var_1754_cast_fp16 = reshape(shape = var_1753, x = linear_65_cast_fp16)[name = tensor("op_1754_cast_fp16")]; tensor v_19_perm_0 = const()[name = tensor("v_19_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_37_cast_fp16 = transpose(perm = q_37_perm_0, x = var_1733_cast_fp16)[name = tensor("transpose_75")]; tensor var_1758_cast_fp16 = mul(x = q_37_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1758_cast_fp16")]; tensor x1_37_begin_0 = const()[name = tensor("x1_37_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_37_end_0 = const()[name = tensor("x1_37_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_37_end_mask_0 = const()[name = tensor("x1_37_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_37_cast_fp16 = slice_by_index(begin = x1_37_begin_0, end = x1_37_end_0, end_mask = x1_37_end_mask_0, x = q_37_cast_fp16)[name = tensor("x1_37_cast_fp16")]; tensor x2_37_begin_0 = const()[name = tensor("x2_37_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_37_end_0 = const()[name = tensor("x2_37_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_37_end_mask_0 = const()[name = tensor("x2_37_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_37_cast_fp16 = slice_by_index(begin = x2_37_begin_0, end = x2_37_end_0, end_mask = x2_37_end_mask_0, x = q_37_cast_fp16)[name = tensor("x2_37_cast_fp16")]; tensor const_19_promoted_to_fp16 = const()[name = tensor("const_19_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1761_cast_fp16 = mul(x = x2_37_cast_fp16, y = const_19_promoted_to_fp16)[name = tensor("op_1761_cast_fp16")]; tensor var_1763_interleave_0 = const()[name = tensor("op_1763_interleave_0"), val = tensor(false)]; tensor var_1763_cast_fp16 = concat(axis = var_1692, interleave = var_1763_interleave_0, values = (var_1761_cast_fp16, x1_37_cast_fp16))[name = tensor("op_1763_cast_fp16")]; tensor var_1764_cast_fp16 = mul(x = var_1763_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1764_cast_fp16")]; tensor q_39_cast_fp16 = add(x = var_1758_cast_fp16, y = var_1764_cast_fp16)[name = tensor("q_39_cast_fp16")]; tensor k_37_cast_fp16 = transpose(perm = k_37_perm_0, x = var_1749_cast_fp16)[name = tensor("transpose_74")]; tensor var_1766_cast_fp16 = mul(x = k_37_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1766_cast_fp16")]; tensor x1_39_begin_0 = const()[name = tensor("x1_39_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_39_end_0 = const()[name = tensor("x1_39_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_39_end_mask_0 = const()[name = tensor("x1_39_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_39_cast_fp16 = slice_by_index(begin = x1_39_begin_0, end = x1_39_end_0, end_mask = x1_39_end_mask_0, x = k_37_cast_fp16)[name = tensor("x1_39_cast_fp16")]; tensor x2_39_begin_0 = const()[name = tensor("x2_39_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_39_end_0 = const()[name = tensor("x2_39_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_39_end_mask_0 = const()[name = tensor("x2_39_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_39_cast_fp16 = slice_by_index(begin = x2_39_begin_0, end = x2_39_end_0, end_mask = x2_39_end_mask_0, x = k_37_cast_fp16)[name = tensor("x2_39_cast_fp16")]; tensor const_20_promoted_to_fp16 = const()[name = tensor("const_20_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1769_cast_fp16 = mul(x = x2_39_cast_fp16, y = const_20_promoted_to_fp16)[name = tensor("op_1769_cast_fp16")]; tensor var_1771_interleave_0 = const()[name = tensor("op_1771_interleave_0"), val = tensor(false)]; tensor var_1771_cast_fp16 = concat(axis = var_1692, interleave = var_1771_interleave_0, values = (var_1769_cast_fp16, x1_39_cast_fp16))[name = tensor("op_1771_cast_fp16")]; tensor var_1772_cast_fp16 = mul(x = var_1771_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1772_cast_fp16")]; tensor k_39_cast_fp16 = add(x = var_1766_cast_fp16, y = var_1772_cast_fp16)[name = tensor("k_39_cast_fp16")]; tensor var_1775_cast_fp16 = mul(x = k_cache_19_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1775_cast_fp16")]; tensor var_1776_cast_fp16 = mul(x = k_39_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1776_cast_fp16")]; tensor k_full_19_cast_fp16 = add(x = var_1775_cast_fp16, y = var_1776_cast_fp16)[name = tensor("k_full_19_cast_fp16")]; tensor var_1779_cast_fp16 = mul(x = v_cache_19_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1779_cast_fp16")]; tensor v_19_cast_fp16 = transpose(perm = v_19_perm_0, x = var_1754_cast_fp16)[name = tensor("transpose_73")]; tensor var_1780_cast_fp16 = mul(x = v_19_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1780_cast_fp16")]; tensor v_full_19_cast_fp16 = add(x = var_1779_cast_fp16, y = var_1780_cast_fp16)[name = tensor("v_full_19_cast_fp16")]; tensor var_1782_axes_0 = const()[name = tensor("op_1782_axes_0"), val = tensor([2])]; tensor var_1782_cast_fp16 = expand_dims(axes = var_1782_axes_0, x = k_full_19_cast_fp16)[name = tensor("op_1782_cast_fp16")]; tensor var_1784_reps_0 = const()[name = tensor("op_1784_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1784_cast_fp16 = tile(reps = var_1784_reps_0, x = var_1782_cast_fp16)[name = tensor("op_1784_cast_fp16")]; tensor var_1785 = const()[name = tensor("op_1785"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_19_cast_fp16 = reshape(shape = var_1785, x = var_1784_cast_fp16)[name = tensor("k_rep_19_cast_fp16")]; tensor var_1787_axes_0 = const()[name = tensor("op_1787_axes_0"), val = tensor([2])]; tensor var_1787_cast_fp16 = expand_dims(axes = var_1787_axes_0, x = v_full_19_cast_fp16)[name = tensor("op_1787_cast_fp16")]; tensor var_1789_reps_0 = const()[name = tensor("op_1789_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1789_cast_fp16 = tile(reps = var_1789_reps_0, x = var_1787_cast_fp16)[name = tensor("op_1789_cast_fp16")]; tensor var_1790 = const()[name = tensor("op_1790"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_19_cast_fp16 = reshape(shape = var_1790, x = var_1789_cast_fp16)[name = tensor("v_rep_19_cast_fp16")]; tensor var_1793_transpose_x_1 = const()[name = tensor("op_1793_transpose_x_1"), val = tensor(false)]; tensor var_1793_transpose_y_1 = const()[name = tensor("op_1793_transpose_y_1"), val = tensor(true)]; tensor var_1793_cast_fp16 = matmul(transpose_x = var_1793_transpose_x_1, transpose_y = var_1793_transpose_y_1, x = q_39_cast_fp16, y = k_rep_19_cast_fp16)[name = tensor("op_1793_cast_fp16")]; tensor var_1794_to_fp16 = const()[name = tensor("op_1794_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_37_cast_fp16 = mul(x = var_1793_cast_fp16, y = var_1794_to_fp16)[name = tensor("attn_37_cast_fp16")]; tensor input_39_cast_fp16 = add(x = attn_37_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_39_cast_fp16")]; tensor input_39_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_39_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_39_cast_fp16_to_fp32 = cast(dtype = input_39_cast_fp16_to_fp32_dtype_0, x = input_39_cast_fp16)[name = tensor("cast_512")]; tensor attn_39 = softmax(axis = var_1692, x = input_39_cast_fp16_to_fp32)[name = tensor("attn_39")]; tensor out_19_transpose_x_0 = const()[name = tensor("out_19_transpose_x_0"), val = tensor(false)]; tensor out_19_transpose_y_0 = const()[name = tensor("out_19_transpose_y_0"), val = tensor(false)]; tensor attn_39_to_fp16_dtype_0 = const()[name = tensor("attn_39_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_39_to_fp16 = cast(dtype = attn_39_to_fp16_dtype_0, x = attn_39)[name = tensor("cast_511")]; tensor out_19_cast_fp16 = matmul(transpose_x = out_19_transpose_x_0, transpose_y = out_19_transpose_y_0, x = attn_39_to_fp16, y = v_rep_19_cast_fp16)[name = tensor("out_19_cast_fp16")]; tensor var_1799_perm_0 = const()[name = tensor("op_1799_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1801 = const()[name = tensor("op_1801"), val = tensor([1, 1, 1536])]; tensor var_1799_cast_fp16 = transpose(perm = var_1799_perm_0, x = out_19_cast_fp16)[name = tensor("transpose_72")]; tensor x_387_cast_fp16 = reshape(shape = var_1801, x = var_1799_cast_fp16)[name = tensor("x_387_cast_fp16")]; tensor layers_9_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_9_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(305327232)))]; tensor linear_66_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_9_self_attn_o_proj_weight_to_fp16, x = x_387_cast_fp16)[name = tensor("linear_66_cast_fp16")]; tensor x_389_cast_fp16 = add(x = x_361_cast_fp16, y = linear_66_cast_fp16)[name = tensor("x_389_cast_fp16")]; tensor x_389_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_389_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1691_promoted_3 = const()[name = tensor("op_1691_promoted_3"), val = tensor(0x1p+1)]; tensor x_389_cast_fp16_to_fp32 = cast(dtype = x_389_cast_fp16_to_fp32_dtype_0, x = x_389_cast_fp16)[name = tensor("cast_510")]; tensor var_1812 = pow(x = x_389_cast_fp16_to_fp32, y = var_1691_promoted_3)[name = tensor("op_1812")]; tensor var_79_axes_0 = const()[name = tensor("var_79_axes_0"), val = tensor([-1])]; tensor var_79_keep_dims_0 = const()[name = tensor("var_79_keep_dims_0"), val = tensor(true)]; tensor var_79 = reduce_mean(axes = var_79_axes_0, keep_dims = var_79_keep_dims_0, x = var_1812)[name = tensor("var_79")]; tensor var_79_to_fp16_dtype_0 = const()[name = tensor("var_79_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1816_to_fp16 = const()[name = tensor("op_1816_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_79_to_fp16 = cast(dtype = var_79_to_fp16_dtype_0, x = var_79)[name = tensor("cast_509")]; tensor var_1817_cast_fp16 = add(x = var_79_to_fp16, y = var_1816_to_fp16)[name = tensor("op_1817_cast_fp16")]; tensor var_1817_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1817_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1818_epsilon_0 = const()[name = tensor("op_1818_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1817_cast_fp16_to_fp32 = cast(dtype = var_1817_cast_fp16_to_fp32_dtype_0, x = var_1817_cast_fp16)[name = tensor("cast_508")]; tensor var_1818 = rsqrt(epsilon = var_1818_epsilon_0, x = var_1817_cast_fp16_to_fp32)[name = tensor("op_1818")]; tensor var_1818_to_fp16_dtype_0 = const()[name = tensor("op_1818_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1818_to_fp16 = cast(dtype = var_1818_to_fp16_dtype_0, x = var_1818)[name = tensor("cast_507")]; tensor x_395_cast_fp16 = mul(x = x_389_cast_fp16, y = var_1818_to_fp16)[name = tensor("x_395_cast_fp16")]; tensor layers_9_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_9_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(306900160)))]; tensor x_397_cast_fp16 = mul(x = layers_9_post_attention_layernorm_weight_to_fp16, y = x_395_cast_fp16)[name = tensor("x_397_cast_fp16")]; tensor layers_9_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_9_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(306901248)))]; tensor linear_67_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_9_mlp_gate_proj_weight_to_fp16, x = x_397_cast_fp16)[name = tensor("linear_67_cast_fp16")]; tensor var_1829_cast_fp16 = silu(x = linear_67_cast_fp16)[name = tensor("op_1829_cast_fp16")]; tensor layers_9_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_9_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(308474176)))]; tensor linear_68_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_9_mlp_up_proj_weight_to_fp16, x = x_397_cast_fp16)[name = tensor("linear_68_cast_fp16")]; tensor x_399_cast_fp16 = mul(x = var_1829_cast_fp16, y = linear_68_cast_fp16)[name = tensor("x_399_cast_fp16")]; tensor layers_9_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_9_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(310047104)))]; tensor linear_69_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_9_mlp_down_proj_weight_to_fp16, x = x_399_cast_fp16)[name = tensor("linear_69_cast_fp16")]; tensor x_401_cast_fp16 = add(x = x_389_cast_fp16, y = linear_69_cast_fp16)[name = tensor("x_401_cast_fp16")]; tensor x_401_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_401_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_21_begin_0 = const()[name = tensor("k_cache_21_begin_0"), val = tensor([10, 0, 0, 0, 0])]; tensor k_cache_21_end_0 = const()[name = tensor("k_cache_21_end_0"), val = tensor([11, 1, 4, 2048, 128])]; tensor k_cache_21_end_mask_0 = const()[name = tensor("k_cache_21_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_21_squeeze_mask_0 = const()[name = tensor("k_cache_21_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_21_cast_fp16 = slice_by_index(begin = k_cache_21_begin_0, end = k_cache_21_end_0, end_mask = k_cache_21_end_mask_0, squeeze_mask = k_cache_21_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_21_cast_fp16")]; tensor v_cache_21_begin_0 = const()[name = tensor("v_cache_21_begin_0"), val = tensor([10, 0, 0, 0, 0])]; tensor v_cache_21_end_0 = const()[name = tensor("v_cache_21_end_0"), val = tensor([11, 1, 4, 2048, 128])]; tensor v_cache_21_end_mask_0 = const()[name = tensor("v_cache_21_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_21_squeeze_mask_0 = const()[name = tensor("v_cache_21_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_21_cast_fp16 = slice_by_index(begin = v_cache_21_begin_0, end = v_cache_21_end_0, end_mask = v_cache_21_end_mask_0, squeeze_mask = v_cache_21_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_21_cast_fp16")]; tensor var_1861 = const()[name = tensor("op_1861"), val = tensor(-1)]; tensor var_1860_promoted = const()[name = tensor("op_1860_promoted"), val = tensor(0x1p+1)]; tensor x_401_cast_fp16_to_fp32 = cast(dtype = x_401_cast_fp16_to_fp32_dtype_0, x = x_401_cast_fp16)[name = tensor("cast_506")]; tensor var_1870 = pow(x = x_401_cast_fp16_to_fp32, y = var_1860_promoted)[name = tensor("op_1870")]; tensor var_81_axes_0 = const()[name = tensor("var_81_axes_0"), val = tensor([-1])]; tensor var_81_keep_dims_0 = const()[name = tensor("var_81_keep_dims_0"), val = tensor(true)]; tensor var_81 = reduce_mean(axes = var_81_axes_0, keep_dims = var_81_keep_dims_0, x = var_1870)[name = tensor("var_81")]; tensor var_81_to_fp16_dtype_0_0 = const()[name = tensor("var_81_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1874_to_fp16 = const()[name = tensor("op_1874_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_81_to_fp16 = cast(dtype = var_81_to_fp16_dtype_0_0, x = var_81)[name = tensor("cast_505")]; tensor var_1875_cast_fp16 = add(x = var_81_to_fp16, y = var_1874_to_fp16)[name = tensor("op_1875_cast_fp16")]; tensor var_1875_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1875_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1876_epsilon_0 = const()[name = tensor("op_1876_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1875_cast_fp16_to_fp32 = cast(dtype = var_1875_cast_fp16_to_fp32_dtype_0, x = var_1875_cast_fp16)[name = tensor("cast_504")]; tensor var_1876 = rsqrt(epsilon = var_1876_epsilon_0, x = var_1875_cast_fp16_to_fp32)[name = tensor("op_1876")]; tensor var_1876_to_fp16_dtype_0 = const()[name = tensor("op_1876_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1876_to_fp16 = cast(dtype = var_1876_to_fp16_dtype_0, x = var_1876)[name = tensor("cast_503")]; tensor x_407_cast_fp16 = mul(x = x_401_cast_fp16, y = var_1876_to_fp16)[name = tensor("x_407_cast_fp16")]; tensor layers_10_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_10_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(311620032)))]; tensor x_409_cast_fp16 = mul(x = layers_10_input_layernorm_weight_to_fp16, y = x_407_cast_fp16)[name = tensor("x_409_cast_fp16")]; tensor layers_10_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_10_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(311621120)))]; tensor linear_70_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_10_self_attn_q_proj_weight_to_fp16, x = x_409_cast_fp16)[name = tensor("linear_70_cast_fp16")]; tensor var_1890 = const()[name = tensor("op_1890"), val = tensor([1, 1, 12, 128])]; tensor x_411_cast_fp16 = reshape(shape = var_1890, x = linear_70_cast_fp16)[name = tensor("x_411_cast_fp16")]; tensor x_411_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_411_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1860_promoted_1 = const()[name = tensor("op_1860_promoted_1"), val = tensor(0x1p+1)]; tensor x_411_cast_fp16_to_fp32 = cast(dtype = x_411_cast_fp16_to_fp32_dtype_0, x = x_411_cast_fp16)[name = tensor("cast_502")]; tensor var_1894 = pow(x = x_411_cast_fp16_to_fp32, y = var_1860_promoted_1)[name = tensor("op_1894")]; tensor var_83_axes_0_0 = const()[name = tensor("var_83_axes_0"), val = tensor([-1])]; tensor var_83_keep_dims_0 = const()[name = tensor("var_83_keep_dims_0"), val = tensor(true)]; tensor var_83 = reduce_mean(axes = var_83_axes_0_0, keep_dims = var_83_keep_dims_0, x = var_1894)[name = tensor("var_83")]; tensor var_83_to_fp16_dtype_0 = const()[name = tensor("var_83_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1898_to_fp16 = const()[name = tensor("op_1898_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_83_to_fp16 = cast(dtype = var_83_to_fp16_dtype_0, x = var_83)[name = tensor("cast_501")]; tensor var_1899_cast_fp16 = add(x = var_83_to_fp16, y = var_1898_to_fp16)[name = tensor("op_1899_cast_fp16")]; tensor var_1899_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1899_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1900_epsilon_0 = const()[name = tensor("op_1900_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1899_cast_fp16_to_fp32 = cast(dtype = var_1899_cast_fp16_to_fp32_dtype_0, x = var_1899_cast_fp16)[name = tensor("cast_500")]; tensor var_1900 = rsqrt(epsilon = var_1900_epsilon_0, x = var_1899_cast_fp16_to_fp32)[name = tensor("op_1900")]; tensor var_1900_to_fp16_dtype_0 = const()[name = tensor("op_1900_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1900_to_fp16 = cast(dtype = var_1900_to_fp16_dtype_0, x = var_1900)[name = tensor("cast_499")]; tensor x_417_cast_fp16 = mul(x = x_411_cast_fp16, y = var_1900_to_fp16)[name = tensor("x_417_cast_fp16")]; tensor layers_10_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_10_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(313194048)))]; tensor var_1902_cast_fp16 = mul(x = layers_10_self_attn_q_norm_weight_to_fp16, y = x_417_cast_fp16)[name = tensor("op_1902_cast_fp16")]; tensor q_41_perm_0 = const()[name = tensor("q_41_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_10_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_10_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(313194368)))]; tensor linear_71_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_10_self_attn_k_proj_weight_to_fp16, x = x_409_cast_fp16)[name = tensor("linear_71_cast_fp16")]; tensor var_1906 = const()[name = tensor("op_1906"), val = tensor([1, 1, 4, 128])]; tensor x_419_cast_fp16 = reshape(shape = var_1906, x = linear_71_cast_fp16)[name = tensor("x_419_cast_fp16")]; tensor x_419_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_419_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1860_promoted_2 = const()[name = tensor("op_1860_promoted_2"), val = tensor(0x1p+1)]; tensor x_419_cast_fp16_to_fp32 = cast(dtype = x_419_cast_fp16_to_fp32_dtype_0, x = x_419_cast_fp16)[name = tensor("cast_498")]; tensor var_1910 = pow(x = x_419_cast_fp16_to_fp32, y = var_1860_promoted_2)[name = tensor("op_1910")]; tensor var_85_axes_0 = const()[name = tensor("var_85_axes_0"), val = tensor([-1])]; tensor var_85_keep_dims_0 = const()[name = tensor("var_85_keep_dims_0"), val = tensor(true)]; tensor var_85 = reduce_mean(axes = var_85_axes_0, keep_dims = var_85_keep_dims_0, x = var_1910)[name = tensor("var_85")]; tensor var_85_to_fp16_dtype_0 = const()[name = tensor("var_85_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1914_to_fp16 = const()[name = tensor("op_1914_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_85_to_fp16 = cast(dtype = var_85_to_fp16_dtype_0, x = var_85)[name = tensor("cast_497")]; tensor var_1915_cast_fp16 = add(x = var_85_to_fp16, y = var_1914_to_fp16)[name = tensor("op_1915_cast_fp16")]; tensor var_1915_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1915_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1916_epsilon_0 = const()[name = tensor("op_1916_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1915_cast_fp16_to_fp32 = cast(dtype = var_1915_cast_fp16_to_fp32_dtype_0, x = var_1915_cast_fp16)[name = tensor("cast_496")]; tensor var_1916 = rsqrt(epsilon = var_1916_epsilon_0, x = var_1915_cast_fp16_to_fp32)[name = tensor("op_1916")]; tensor var_1916_to_fp16_dtype_0 = const()[name = tensor("op_1916_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1916_to_fp16 = cast(dtype = var_1916_to_fp16_dtype_0, x = var_1916)[name = tensor("cast_495")]; tensor x_425_cast_fp16 = mul(x = x_419_cast_fp16, y = var_1916_to_fp16)[name = tensor("x_425_cast_fp16")]; tensor layers_10_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_10_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(313718720)))]; tensor var_1918_cast_fp16 = mul(x = layers_10_self_attn_k_norm_weight_to_fp16, y = x_425_cast_fp16)[name = tensor("op_1918_cast_fp16")]; tensor k_41_perm_0 = const()[name = tensor("k_41_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_10_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_10_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(313719040)))]; tensor linear_72_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_10_self_attn_v_proj_weight_to_fp16, x = x_409_cast_fp16)[name = tensor("linear_72_cast_fp16")]; tensor var_1922 = const()[name = tensor("op_1922"), val = tensor([1, 1, 4, 128])]; tensor var_1923_cast_fp16 = reshape(shape = var_1922, x = linear_72_cast_fp16)[name = tensor("op_1923_cast_fp16")]; tensor v_21_perm_0 = const()[name = tensor("v_21_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_41_cast_fp16 = transpose(perm = q_41_perm_0, x = var_1902_cast_fp16)[name = tensor("transpose_71")]; tensor var_1927_cast_fp16 = mul(x = q_41_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1927_cast_fp16")]; tensor x1_41_begin_0 = const()[name = tensor("x1_41_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_41_end_0 = const()[name = tensor("x1_41_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_41_end_mask_0 = const()[name = tensor("x1_41_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_41_cast_fp16 = slice_by_index(begin = x1_41_begin_0, end = x1_41_end_0, end_mask = x1_41_end_mask_0, x = q_41_cast_fp16)[name = tensor("x1_41_cast_fp16")]; tensor x2_41_begin_0 = const()[name = tensor("x2_41_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_41_end_0 = const()[name = tensor("x2_41_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_41_end_mask_0 = const()[name = tensor("x2_41_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_41_cast_fp16 = slice_by_index(begin = x2_41_begin_0, end = x2_41_end_0, end_mask = x2_41_end_mask_0, x = q_41_cast_fp16)[name = tensor("x2_41_cast_fp16")]; tensor const_21_promoted_to_fp16 = const()[name = tensor("const_21_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1930_cast_fp16 = mul(x = x2_41_cast_fp16, y = const_21_promoted_to_fp16)[name = tensor("op_1930_cast_fp16")]; tensor var_1932_interleave_0 = const()[name = tensor("op_1932_interleave_0"), val = tensor(false)]; tensor var_1932_cast_fp16 = concat(axis = var_1861, interleave = var_1932_interleave_0, values = (var_1930_cast_fp16, x1_41_cast_fp16))[name = tensor("op_1932_cast_fp16")]; tensor var_1933_cast_fp16 = mul(x = var_1932_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1933_cast_fp16")]; tensor q_43_cast_fp16 = add(x = var_1927_cast_fp16, y = var_1933_cast_fp16)[name = tensor("q_43_cast_fp16")]; tensor k_41_cast_fp16 = transpose(perm = k_41_perm_0, x = var_1918_cast_fp16)[name = tensor("transpose_70")]; tensor var_1935_cast_fp16 = mul(x = k_41_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_1935_cast_fp16")]; tensor x1_43_begin_0 = const()[name = tensor("x1_43_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_43_end_0 = const()[name = tensor("x1_43_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_43_end_mask_0 = const()[name = tensor("x1_43_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_43_cast_fp16 = slice_by_index(begin = x1_43_begin_0, end = x1_43_end_0, end_mask = x1_43_end_mask_0, x = k_41_cast_fp16)[name = tensor("x1_43_cast_fp16")]; tensor x2_43_begin_0 = const()[name = tensor("x2_43_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_43_end_0 = const()[name = tensor("x2_43_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_43_end_mask_0 = const()[name = tensor("x2_43_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_43_cast_fp16 = slice_by_index(begin = x2_43_begin_0, end = x2_43_end_0, end_mask = x2_43_end_mask_0, x = k_41_cast_fp16)[name = tensor("x2_43_cast_fp16")]; tensor const_22_promoted_to_fp16 = const()[name = tensor("const_22_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_1938_cast_fp16 = mul(x = x2_43_cast_fp16, y = const_22_promoted_to_fp16)[name = tensor("op_1938_cast_fp16")]; tensor var_1940_interleave_0 = const()[name = tensor("op_1940_interleave_0"), val = tensor(false)]; tensor var_1940_cast_fp16 = concat(axis = var_1861, interleave = var_1940_interleave_0, values = (var_1938_cast_fp16, x1_43_cast_fp16))[name = tensor("op_1940_cast_fp16")]; tensor var_1941_cast_fp16 = mul(x = var_1940_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_1941_cast_fp16")]; tensor k_43_cast_fp16 = add(x = var_1935_cast_fp16, y = var_1941_cast_fp16)[name = tensor("k_43_cast_fp16")]; tensor var_1944_cast_fp16 = mul(x = k_cache_21_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1944_cast_fp16")]; tensor var_1945_cast_fp16 = mul(x = k_43_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1945_cast_fp16")]; tensor k_full_21_cast_fp16 = add(x = var_1944_cast_fp16, y = var_1945_cast_fp16)[name = tensor("k_full_21_cast_fp16")]; tensor var_1948_cast_fp16 = mul(x = v_cache_21_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_1948_cast_fp16")]; tensor v_21_cast_fp16 = transpose(perm = v_21_perm_0, x = var_1923_cast_fp16)[name = tensor("transpose_69")]; tensor var_1949_cast_fp16 = mul(x = v_21_cast_fp16, y = var_107_to_fp16)[name = tensor("op_1949_cast_fp16")]; tensor v_full_21_cast_fp16 = add(x = var_1948_cast_fp16, y = var_1949_cast_fp16)[name = tensor("v_full_21_cast_fp16")]; tensor var_1951_axes_0 = const()[name = tensor("op_1951_axes_0"), val = tensor([2])]; tensor var_1951_cast_fp16 = expand_dims(axes = var_1951_axes_0, x = k_full_21_cast_fp16)[name = tensor("op_1951_cast_fp16")]; tensor var_1953_reps_0 = const()[name = tensor("op_1953_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1953_cast_fp16 = tile(reps = var_1953_reps_0, x = var_1951_cast_fp16)[name = tensor("op_1953_cast_fp16")]; tensor var_1954 = const()[name = tensor("op_1954"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_21_cast_fp16 = reshape(shape = var_1954, x = var_1953_cast_fp16)[name = tensor("k_rep_21_cast_fp16")]; tensor var_1956_axes_0 = const()[name = tensor("op_1956_axes_0"), val = tensor([2])]; tensor var_1956_cast_fp16 = expand_dims(axes = var_1956_axes_0, x = v_full_21_cast_fp16)[name = tensor("op_1956_cast_fp16")]; tensor var_1958_reps_0 = const()[name = tensor("op_1958_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_1958_cast_fp16 = tile(reps = var_1958_reps_0, x = var_1956_cast_fp16)[name = tensor("op_1958_cast_fp16")]; tensor var_1959 = const()[name = tensor("op_1959"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_21_cast_fp16 = reshape(shape = var_1959, x = var_1958_cast_fp16)[name = tensor("v_rep_21_cast_fp16")]; tensor var_1962_transpose_x_1 = const()[name = tensor("op_1962_transpose_x_1"), val = tensor(false)]; tensor var_1962_transpose_y_1 = const()[name = tensor("op_1962_transpose_y_1"), val = tensor(true)]; tensor var_1962_cast_fp16 = matmul(transpose_x = var_1962_transpose_x_1, transpose_y = var_1962_transpose_y_1, x = q_43_cast_fp16, y = k_rep_21_cast_fp16)[name = tensor("op_1962_cast_fp16")]; tensor var_1963_to_fp16 = const()[name = tensor("op_1963_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_41_cast_fp16 = mul(x = var_1962_cast_fp16, y = var_1963_to_fp16)[name = tensor("attn_41_cast_fp16")]; tensor input_43_cast_fp16 = add(x = attn_41_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_43_cast_fp16")]; tensor input_43_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_43_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_43_cast_fp16_to_fp32 = cast(dtype = input_43_cast_fp16_to_fp32_dtype_0, x = input_43_cast_fp16)[name = tensor("cast_494")]; tensor attn_43 = softmax(axis = var_1861, x = input_43_cast_fp16_to_fp32)[name = tensor("attn_43")]; tensor out_21_transpose_x_0 = const()[name = tensor("out_21_transpose_x_0"), val = tensor(false)]; tensor out_21_transpose_y_0 = const()[name = tensor("out_21_transpose_y_0"), val = tensor(false)]; tensor attn_43_to_fp16_dtype_0 = const()[name = tensor("attn_43_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_43_to_fp16 = cast(dtype = attn_43_to_fp16_dtype_0, x = attn_43)[name = tensor("cast_493")]; tensor out_21_cast_fp16 = matmul(transpose_x = out_21_transpose_x_0, transpose_y = out_21_transpose_y_0, x = attn_43_to_fp16, y = v_rep_21_cast_fp16)[name = tensor("out_21_cast_fp16")]; tensor var_1968_perm_0 = const()[name = tensor("op_1968_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1970 = const()[name = tensor("op_1970"), val = tensor([1, 1, 1536])]; tensor var_1968_cast_fp16 = transpose(perm = var_1968_perm_0, x = out_21_cast_fp16)[name = tensor("transpose_68")]; tensor x_427_cast_fp16 = reshape(shape = var_1970, x = var_1968_cast_fp16)[name = tensor("x_427_cast_fp16")]; tensor layers_10_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_10_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(314243392)))]; tensor linear_73_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_10_self_attn_o_proj_weight_to_fp16, x = x_427_cast_fp16)[name = tensor("linear_73_cast_fp16")]; tensor x_429_cast_fp16 = add(x = x_401_cast_fp16, y = linear_73_cast_fp16)[name = tensor("x_429_cast_fp16")]; tensor x_429_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_429_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1860_promoted_3 = const()[name = tensor("op_1860_promoted_3"), val = tensor(0x1p+1)]; tensor x_429_cast_fp16_to_fp32 = cast(dtype = x_429_cast_fp16_to_fp32_dtype_0, x = x_429_cast_fp16)[name = tensor("cast_492")]; tensor var_1981 = pow(x = x_429_cast_fp16_to_fp32, y = var_1860_promoted_3)[name = tensor("op_1981")]; tensor var_87_axes_0 = const()[name = tensor("var_87_axes_0"), val = tensor([-1])]; tensor var_87_keep_dims_0 = const()[name = tensor("var_87_keep_dims_0"), val = tensor(true)]; tensor var_87 = reduce_mean(axes = var_87_axes_0, keep_dims = var_87_keep_dims_0, x = var_1981)[name = tensor("var_87")]; tensor var_87_to_fp16_dtype_0 = const()[name = tensor("var_87_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1985_to_fp16 = const()[name = tensor("op_1985_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_87_to_fp16 = cast(dtype = var_87_to_fp16_dtype_0, x = var_87)[name = tensor("cast_491")]; tensor var_1986_cast_fp16 = add(x = var_87_to_fp16, y = var_1985_to_fp16)[name = tensor("op_1986_cast_fp16")]; tensor var_1986_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_1986_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_1987_epsilon_0 = const()[name = tensor("op_1987_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_1986_cast_fp16_to_fp32 = cast(dtype = var_1986_cast_fp16_to_fp32_dtype_0, x = var_1986_cast_fp16)[name = tensor("cast_490")]; tensor var_1987 = rsqrt(epsilon = var_1987_epsilon_0, x = var_1986_cast_fp16_to_fp32)[name = tensor("op_1987")]; tensor var_1987_to_fp16_dtype_0 = const()[name = tensor("op_1987_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_1987_to_fp16 = cast(dtype = var_1987_to_fp16_dtype_0, x = var_1987)[name = tensor("cast_489")]; tensor x_435_cast_fp16 = mul(x = x_429_cast_fp16, y = var_1987_to_fp16)[name = tensor("x_435_cast_fp16")]; tensor layers_10_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_10_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(315816320)))]; tensor x_437_cast_fp16 = mul(x = layers_10_post_attention_layernorm_weight_to_fp16, y = x_435_cast_fp16)[name = tensor("x_437_cast_fp16")]; tensor layers_10_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_10_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(315817408)))]; tensor linear_74_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_10_mlp_gate_proj_weight_to_fp16, x = x_437_cast_fp16)[name = tensor("linear_74_cast_fp16")]; tensor var_1998_cast_fp16 = silu(x = linear_74_cast_fp16)[name = tensor("op_1998_cast_fp16")]; tensor layers_10_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_10_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(317390336)))]; tensor linear_75_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_10_mlp_up_proj_weight_to_fp16, x = x_437_cast_fp16)[name = tensor("linear_75_cast_fp16")]; tensor x_439_cast_fp16 = mul(x = var_1998_cast_fp16, y = linear_75_cast_fp16)[name = tensor("x_439_cast_fp16")]; tensor layers_10_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_10_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(318963264)))]; tensor linear_76_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_10_mlp_down_proj_weight_to_fp16, x = x_439_cast_fp16)[name = tensor("linear_76_cast_fp16")]; tensor x_441_cast_fp16 = add(x = x_429_cast_fp16, y = linear_76_cast_fp16)[name = tensor("x_441_cast_fp16")]; tensor x_441_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_441_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_23_begin_0 = const()[name = tensor("k_cache_23_begin_0"), val = tensor([11, 0, 0, 0, 0])]; tensor k_cache_23_end_0 = const()[name = tensor("k_cache_23_end_0"), val = tensor([12, 1, 4, 2048, 128])]; tensor k_cache_23_end_mask_0 = const()[name = tensor("k_cache_23_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_23_squeeze_mask_0 = const()[name = tensor("k_cache_23_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_23_cast_fp16 = slice_by_index(begin = k_cache_23_begin_0, end = k_cache_23_end_0, end_mask = k_cache_23_end_mask_0, squeeze_mask = k_cache_23_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_23_cast_fp16")]; tensor v_cache_23_begin_0 = const()[name = tensor("v_cache_23_begin_0"), val = tensor([11, 0, 0, 0, 0])]; tensor v_cache_23_end_0 = const()[name = tensor("v_cache_23_end_0"), val = tensor([12, 1, 4, 2048, 128])]; tensor v_cache_23_end_mask_0 = const()[name = tensor("v_cache_23_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_23_squeeze_mask_0 = const()[name = tensor("v_cache_23_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_23_cast_fp16 = slice_by_index(begin = v_cache_23_begin_0, end = v_cache_23_end_0, end_mask = v_cache_23_end_mask_0, squeeze_mask = v_cache_23_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_23_cast_fp16")]; tensor var_2030 = const()[name = tensor("op_2030"), val = tensor(-1)]; tensor var_2029_promoted = const()[name = tensor("op_2029_promoted"), val = tensor(0x1p+1)]; tensor x_441_cast_fp16_to_fp32 = cast(dtype = x_441_cast_fp16_to_fp32_dtype_0, x = x_441_cast_fp16)[name = tensor("cast_488")]; tensor var_2039 = pow(x = x_441_cast_fp16_to_fp32, y = var_2029_promoted)[name = tensor("op_2039")]; tensor var_89_axes_0 = const()[name = tensor("var_89_axes_0"), val = tensor([-1])]; tensor var_89_keep_dims_0 = const()[name = tensor("var_89_keep_dims_0"), val = tensor(true)]; tensor var_89 = reduce_mean(axes = var_89_axes_0, keep_dims = var_89_keep_dims_0, x = var_2039)[name = tensor("var_89")]; tensor var_89_to_fp16_dtype_0 = const()[name = tensor("var_89_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2043_to_fp16 = const()[name = tensor("op_2043_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_89_to_fp16 = cast(dtype = var_89_to_fp16_dtype_0, x = var_89)[name = tensor("cast_487")]; tensor var_2044_cast_fp16 = add(x = var_89_to_fp16, y = var_2043_to_fp16)[name = tensor("op_2044_cast_fp16")]; tensor var_2044_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2044_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2045_epsilon_0 = const()[name = tensor("op_2045_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2044_cast_fp16_to_fp32 = cast(dtype = var_2044_cast_fp16_to_fp32_dtype_0, x = var_2044_cast_fp16)[name = tensor("cast_486")]; tensor var_2045 = rsqrt(epsilon = var_2045_epsilon_0, x = var_2044_cast_fp16_to_fp32)[name = tensor("op_2045")]; tensor var_2045_to_fp16_dtype_0 = const()[name = tensor("op_2045_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2045_to_fp16 = cast(dtype = var_2045_to_fp16_dtype_0, x = var_2045)[name = tensor("cast_485")]; tensor x_447_cast_fp16 = mul(x = x_441_cast_fp16, y = var_2045_to_fp16)[name = tensor("x_447_cast_fp16")]; tensor layers_11_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_11_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(320536192)))]; tensor x_449_cast_fp16 = mul(x = layers_11_input_layernorm_weight_to_fp16, y = x_447_cast_fp16)[name = tensor("x_449_cast_fp16")]; tensor layers_11_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_11_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(320537280)))]; tensor linear_77_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_11_self_attn_q_proj_weight_to_fp16, x = x_449_cast_fp16)[name = tensor("linear_77_cast_fp16")]; tensor var_2059 = const()[name = tensor("op_2059"), val = tensor([1, 1, 12, 128])]; tensor x_451_cast_fp16 = reshape(shape = var_2059, x = linear_77_cast_fp16)[name = tensor("x_451_cast_fp16")]; tensor x_451_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_451_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2029_promoted_1 = const()[name = tensor("op_2029_promoted_1"), val = tensor(0x1p+1)]; tensor x_451_cast_fp16_to_fp32 = cast(dtype = x_451_cast_fp16_to_fp32_dtype_0, x = x_451_cast_fp16)[name = tensor("cast_484")]; tensor var_2063 = pow(x = x_451_cast_fp16_to_fp32, y = var_2029_promoted_1)[name = tensor("op_2063")]; tensor var_91_axes_0 = const()[name = tensor("var_91_axes_0"), val = tensor([-1])]; tensor var_91_keep_dims_0 = const()[name = tensor("var_91_keep_dims_0"), val = tensor(true)]; tensor var_91_0 = reduce_mean(axes = var_91_axes_0, keep_dims = var_91_keep_dims_0, x = var_2063)[name = tensor("var_91")]; tensor var_91_to_fp16_dtype_0 = const()[name = tensor("var_91_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2067_to_fp16 = const()[name = tensor("op_2067_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_91_to_fp16 = cast(dtype = var_91_to_fp16_dtype_0, x = var_91_0)[name = tensor("cast_483")]; tensor var_2068_cast_fp16 = add(x = var_91_to_fp16, y = var_2067_to_fp16)[name = tensor("op_2068_cast_fp16")]; tensor var_2068_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2068_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2069_epsilon_0 = const()[name = tensor("op_2069_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2068_cast_fp16_to_fp32 = cast(dtype = var_2068_cast_fp16_to_fp32_dtype_0, x = var_2068_cast_fp16)[name = tensor("cast_482")]; tensor var_2069 = rsqrt(epsilon = var_2069_epsilon_0, x = var_2068_cast_fp16_to_fp32)[name = tensor("op_2069")]; tensor var_2069_to_fp16_dtype_0 = const()[name = tensor("op_2069_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2069_to_fp16 = cast(dtype = var_2069_to_fp16_dtype_0, x = var_2069)[name = tensor("cast_481")]; tensor x_457_cast_fp16 = mul(x = x_451_cast_fp16, y = var_2069_to_fp16)[name = tensor("x_457_cast_fp16")]; tensor layers_11_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_11_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322110208)))]; tensor var_2071_cast_fp16 = mul(x = layers_11_self_attn_q_norm_weight_to_fp16, y = x_457_cast_fp16)[name = tensor("op_2071_cast_fp16")]; tensor q_45_perm_0 = const()[name = tensor("q_45_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_11_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_11_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322110528)))]; tensor linear_78_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_11_self_attn_k_proj_weight_to_fp16, x = x_449_cast_fp16)[name = tensor("linear_78_cast_fp16")]; tensor var_2075 = const()[name = tensor("op_2075"), val = tensor([1, 1, 4, 128])]; tensor x_459_cast_fp16 = reshape(shape = var_2075, x = linear_78_cast_fp16)[name = tensor("x_459_cast_fp16")]; tensor x_459_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_459_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2029_promoted_2 = const()[name = tensor("op_2029_promoted_2"), val = tensor(0x1p+1)]; tensor x_459_cast_fp16_to_fp32 = cast(dtype = x_459_cast_fp16_to_fp32_dtype_0, x = x_459_cast_fp16)[name = tensor("cast_480")]; tensor var_2079 = pow(x = x_459_cast_fp16_to_fp32, y = var_2029_promoted_2)[name = tensor("op_2079")]; tensor var_93_axes_0 = const()[name = tensor("var_93_axes_0"), val = tensor([-1])]; tensor var_93_keep_dims_0 = const()[name = tensor("var_93_keep_dims_0"), val = tensor(true)]; tensor var_93 = reduce_mean(axes = var_93_axes_0, keep_dims = var_93_keep_dims_0, x = var_2079)[name = tensor("var_93")]; tensor var_93_to_fp16_dtype_0 = const()[name = tensor("var_93_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2083_to_fp16 = const()[name = tensor("op_2083_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_93_to_fp16 = cast(dtype = var_93_to_fp16_dtype_0, x = var_93)[name = tensor("cast_479")]; tensor var_2084_cast_fp16 = add(x = var_93_to_fp16, y = var_2083_to_fp16)[name = tensor("op_2084_cast_fp16")]; tensor var_2084_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2084_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2085_epsilon_0 = const()[name = tensor("op_2085_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2084_cast_fp16_to_fp32 = cast(dtype = var_2084_cast_fp16_to_fp32_dtype_0, x = var_2084_cast_fp16)[name = tensor("cast_478")]; tensor var_2085 = rsqrt(epsilon = var_2085_epsilon_0, x = var_2084_cast_fp16_to_fp32)[name = tensor("op_2085")]; tensor var_2085_to_fp16_dtype_0 = const()[name = tensor("op_2085_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2085_to_fp16 = cast(dtype = var_2085_to_fp16_dtype_0, x = var_2085)[name = tensor("cast_477")]; tensor x_465_cast_fp16 = mul(x = x_459_cast_fp16, y = var_2085_to_fp16)[name = tensor("x_465_cast_fp16")]; tensor layers_11_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_11_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322634880)))]; tensor var_2087_cast_fp16 = mul(x = layers_11_self_attn_k_norm_weight_to_fp16, y = x_465_cast_fp16)[name = tensor("op_2087_cast_fp16")]; tensor k_45_perm_0 = const()[name = tensor("k_45_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_11_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_11_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322635200)))]; tensor linear_79_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_11_self_attn_v_proj_weight_to_fp16, x = x_449_cast_fp16)[name = tensor("linear_79_cast_fp16")]; tensor var_2091 = const()[name = tensor("op_2091"), val = tensor([1, 1, 4, 128])]; tensor var_2092_cast_fp16 = reshape(shape = var_2091, x = linear_79_cast_fp16)[name = tensor("op_2092_cast_fp16")]; tensor v_23_perm_0 = const()[name = tensor("v_23_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_45_cast_fp16 = transpose(perm = q_45_perm_0, x = var_2071_cast_fp16)[name = tensor("transpose_67")]; tensor var_2096_cast_fp16 = mul(x = q_45_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2096_cast_fp16")]; tensor x1_45_begin_0 = const()[name = tensor("x1_45_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_45_end_0 = const()[name = tensor("x1_45_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_45_end_mask_0 = const()[name = tensor("x1_45_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_45_cast_fp16 = slice_by_index(begin = x1_45_begin_0, end = x1_45_end_0, end_mask = x1_45_end_mask_0, x = q_45_cast_fp16)[name = tensor("x1_45_cast_fp16")]; tensor x2_45_begin_0 = const()[name = tensor("x2_45_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_45_end_0 = const()[name = tensor("x2_45_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_45_end_mask_0 = const()[name = tensor("x2_45_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_45_cast_fp16 = slice_by_index(begin = x2_45_begin_0, end = x2_45_end_0, end_mask = x2_45_end_mask_0, x = q_45_cast_fp16)[name = tensor("x2_45_cast_fp16")]; tensor const_23_promoted_to_fp16 = const()[name = tensor("const_23_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2099_cast_fp16 = mul(x = x2_45_cast_fp16, y = const_23_promoted_to_fp16)[name = tensor("op_2099_cast_fp16")]; tensor var_2101_interleave_0 = const()[name = tensor("op_2101_interleave_0"), val = tensor(false)]; tensor var_2101_cast_fp16 = concat(axis = var_2030, interleave = var_2101_interleave_0, values = (var_2099_cast_fp16, x1_45_cast_fp16))[name = tensor("op_2101_cast_fp16")]; tensor var_2102_cast_fp16 = mul(x = var_2101_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2102_cast_fp16")]; tensor q_47_cast_fp16 = add(x = var_2096_cast_fp16, y = var_2102_cast_fp16)[name = tensor("q_47_cast_fp16")]; tensor k_45_cast_fp16 = transpose(perm = k_45_perm_0, x = var_2087_cast_fp16)[name = tensor("transpose_66")]; tensor var_2104_cast_fp16 = mul(x = k_45_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2104_cast_fp16")]; tensor x1_47_begin_0 = const()[name = tensor("x1_47_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_47_end_0 = const()[name = tensor("x1_47_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_47_end_mask_0 = const()[name = tensor("x1_47_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_47_cast_fp16 = slice_by_index(begin = x1_47_begin_0, end = x1_47_end_0, end_mask = x1_47_end_mask_0, x = k_45_cast_fp16)[name = tensor("x1_47_cast_fp16")]; tensor x2_47_begin_0 = const()[name = tensor("x2_47_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_47_end_0 = const()[name = tensor("x2_47_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_47_end_mask_0 = const()[name = tensor("x2_47_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_47_cast_fp16 = slice_by_index(begin = x2_47_begin_0, end = x2_47_end_0, end_mask = x2_47_end_mask_0, x = k_45_cast_fp16)[name = tensor("x2_47_cast_fp16")]; tensor const_24_promoted_to_fp16 = const()[name = tensor("const_24_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2107_cast_fp16 = mul(x = x2_47_cast_fp16, y = const_24_promoted_to_fp16)[name = tensor("op_2107_cast_fp16")]; tensor var_2109_interleave_0 = const()[name = tensor("op_2109_interleave_0"), val = tensor(false)]; tensor var_2109_cast_fp16 = concat(axis = var_2030, interleave = var_2109_interleave_0, values = (var_2107_cast_fp16, x1_47_cast_fp16))[name = tensor("op_2109_cast_fp16")]; tensor var_2110_cast_fp16 = mul(x = var_2109_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2110_cast_fp16")]; tensor k_47_cast_fp16 = add(x = var_2104_cast_fp16, y = var_2110_cast_fp16)[name = tensor("k_47_cast_fp16")]; tensor var_2113_cast_fp16 = mul(x = k_cache_23_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2113_cast_fp16")]; tensor var_2114_cast_fp16 = mul(x = k_47_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2114_cast_fp16")]; tensor k_full_23_cast_fp16 = add(x = var_2113_cast_fp16, y = var_2114_cast_fp16)[name = tensor("k_full_23_cast_fp16")]; tensor var_2117_cast_fp16 = mul(x = v_cache_23_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2117_cast_fp16")]; tensor v_23_cast_fp16 = transpose(perm = v_23_perm_0, x = var_2092_cast_fp16)[name = tensor("transpose_65")]; tensor var_2118_cast_fp16 = mul(x = v_23_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2118_cast_fp16")]; tensor v_full_23_cast_fp16 = add(x = var_2117_cast_fp16, y = var_2118_cast_fp16)[name = tensor("v_full_23_cast_fp16")]; tensor var_2120_axes_0 = const()[name = tensor("op_2120_axes_0"), val = tensor([2])]; tensor var_2120_cast_fp16 = expand_dims(axes = var_2120_axes_0, x = k_full_23_cast_fp16)[name = tensor("op_2120_cast_fp16")]; tensor var_2122_reps_0 = const()[name = tensor("op_2122_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2122_cast_fp16 = tile(reps = var_2122_reps_0, x = var_2120_cast_fp16)[name = tensor("op_2122_cast_fp16")]; tensor var_2123 = const()[name = tensor("op_2123"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_23_cast_fp16 = reshape(shape = var_2123, x = var_2122_cast_fp16)[name = tensor("k_rep_23_cast_fp16")]; tensor var_2125_axes_0 = const()[name = tensor("op_2125_axes_0"), val = tensor([2])]; tensor var_2125_cast_fp16 = expand_dims(axes = var_2125_axes_0, x = v_full_23_cast_fp16)[name = tensor("op_2125_cast_fp16")]; tensor var_2127_reps_0 = const()[name = tensor("op_2127_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2127_cast_fp16 = tile(reps = var_2127_reps_0, x = var_2125_cast_fp16)[name = tensor("op_2127_cast_fp16")]; tensor var_2128 = const()[name = tensor("op_2128"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_23_cast_fp16 = reshape(shape = var_2128, x = var_2127_cast_fp16)[name = tensor("v_rep_23_cast_fp16")]; tensor var_2131_transpose_x_1 = const()[name = tensor("op_2131_transpose_x_1"), val = tensor(false)]; tensor var_2131_transpose_y_1 = const()[name = tensor("op_2131_transpose_y_1"), val = tensor(true)]; tensor var_2131_cast_fp16 = matmul(transpose_x = var_2131_transpose_x_1, transpose_y = var_2131_transpose_y_1, x = q_47_cast_fp16, y = k_rep_23_cast_fp16)[name = tensor("op_2131_cast_fp16")]; tensor var_2132_to_fp16 = const()[name = tensor("op_2132_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_45_cast_fp16 = mul(x = var_2131_cast_fp16, y = var_2132_to_fp16)[name = tensor("attn_45_cast_fp16")]; tensor input_47_cast_fp16 = add(x = attn_45_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_47_cast_fp16")]; tensor input_47_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_47_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_47_cast_fp16_to_fp32 = cast(dtype = input_47_cast_fp16_to_fp32_dtype_0, x = input_47_cast_fp16)[name = tensor("cast_476")]; tensor attn_47 = softmax(axis = var_2030, x = input_47_cast_fp16_to_fp32)[name = tensor("attn_47")]; tensor out_23_transpose_x_0 = const()[name = tensor("out_23_transpose_x_0"), val = tensor(false)]; tensor out_23_transpose_y_0 = const()[name = tensor("out_23_transpose_y_0"), val = tensor(false)]; tensor attn_47_to_fp16_dtype_0 = const()[name = tensor("attn_47_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_47_to_fp16 = cast(dtype = attn_47_to_fp16_dtype_0, x = attn_47)[name = tensor("cast_475")]; tensor out_23_cast_fp16 = matmul(transpose_x = out_23_transpose_x_0, transpose_y = out_23_transpose_y_0, x = attn_47_to_fp16, y = v_rep_23_cast_fp16)[name = tensor("out_23_cast_fp16")]; tensor var_2137_perm_0 = const()[name = tensor("op_2137_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2139 = const()[name = tensor("op_2139"), val = tensor([1, 1, 1536])]; tensor var_2137_cast_fp16 = transpose(perm = var_2137_perm_0, x = out_23_cast_fp16)[name = tensor("transpose_64")]; tensor x_467_cast_fp16 = reshape(shape = var_2139, x = var_2137_cast_fp16)[name = tensor("x_467_cast_fp16")]; tensor layers_11_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_11_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(323159552)))]; tensor linear_80_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_11_self_attn_o_proj_weight_to_fp16, x = x_467_cast_fp16)[name = tensor("linear_80_cast_fp16")]; tensor x_469_cast_fp16 = add(x = x_441_cast_fp16, y = linear_80_cast_fp16)[name = tensor("x_469_cast_fp16")]; tensor x_469_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_469_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2029_promoted_3 = const()[name = tensor("op_2029_promoted_3"), val = tensor(0x1p+1)]; tensor x_469_cast_fp16_to_fp32 = cast(dtype = x_469_cast_fp16_to_fp32_dtype_0, x = x_469_cast_fp16)[name = tensor("cast_474")]; tensor var_2150 = pow(x = x_469_cast_fp16_to_fp32, y = var_2029_promoted_3)[name = tensor("op_2150")]; tensor var_95_axes_0 = const()[name = tensor("var_95_axes_0"), val = tensor([-1])]; tensor var_95_keep_dims_0 = const()[name = tensor("var_95_keep_dims_0"), val = tensor(true)]; tensor var_95 = reduce_mean(axes = var_95_axes_0, keep_dims = var_95_keep_dims_0, x = var_2150)[name = tensor("var_95")]; tensor var_95_to_fp16_dtype_0 = const()[name = tensor("var_95_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2154_to_fp16 = const()[name = tensor("op_2154_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_95_to_fp16 = cast(dtype = var_95_to_fp16_dtype_0, x = var_95)[name = tensor("cast_473")]; tensor var_2155_cast_fp16 = add(x = var_95_to_fp16, y = var_2154_to_fp16)[name = tensor("op_2155_cast_fp16")]; tensor var_2155_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2155_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2156_epsilon_0 = const()[name = tensor("op_2156_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2155_cast_fp16_to_fp32 = cast(dtype = var_2155_cast_fp16_to_fp32_dtype_0, x = var_2155_cast_fp16)[name = tensor("cast_472")]; tensor var_2156 = rsqrt(epsilon = var_2156_epsilon_0, x = var_2155_cast_fp16_to_fp32)[name = tensor("op_2156")]; tensor var_2156_to_fp16_dtype_0 = const()[name = tensor("op_2156_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2156_to_fp16 = cast(dtype = var_2156_to_fp16_dtype_0, x = var_2156)[name = tensor("cast_471")]; tensor x_475_cast_fp16 = mul(x = x_469_cast_fp16, y = var_2156_to_fp16)[name = tensor("x_475_cast_fp16")]; tensor layers_11_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_11_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(324732480)))]; tensor x_477_cast_fp16 = mul(x = layers_11_post_attention_layernorm_weight_to_fp16, y = x_475_cast_fp16)[name = tensor("x_477_cast_fp16")]; tensor layers_11_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_11_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(324733568)))]; tensor linear_81_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_11_mlp_gate_proj_weight_to_fp16, x = x_477_cast_fp16)[name = tensor("linear_81_cast_fp16")]; tensor var_2167_cast_fp16 = silu(x = linear_81_cast_fp16)[name = tensor("op_2167_cast_fp16")]; tensor layers_11_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_11_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(326306496)))]; tensor linear_82_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_11_mlp_up_proj_weight_to_fp16, x = x_477_cast_fp16)[name = tensor("linear_82_cast_fp16")]; tensor x_479_cast_fp16 = mul(x = var_2167_cast_fp16, y = linear_82_cast_fp16)[name = tensor("x_479_cast_fp16")]; tensor layers_11_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_11_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(327879424)))]; tensor linear_83_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_11_mlp_down_proj_weight_to_fp16, x = x_479_cast_fp16)[name = tensor("linear_83_cast_fp16")]; tensor x_481_cast_fp16 = add(x = x_469_cast_fp16, y = linear_83_cast_fp16)[name = tensor("x_481_cast_fp16")]; tensor x_481_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_481_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_25_begin_0 = const()[name = tensor("k_cache_25_begin_0"), val = tensor([12, 0, 0, 0, 0])]; tensor k_cache_25_end_0 = const()[name = tensor("k_cache_25_end_0"), val = tensor([13, 1, 4, 2048, 128])]; tensor k_cache_25_end_mask_0 = const()[name = tensor("k_cache_25_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_25_squeeze_mask_0 = const()[name = tensor("k_cache_25_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_25_cast_fp16 = slice_by_index(begin = k_cache_25_begin_0, end = k_cache_25_end_0, end_mask = k_cache_25_end_mask_0, squeeze_mask = k_cache_25_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_25_cast_fp16")]; tensor v_cache_25_begin_0 = const()[name = tensor("v_cache_25_begin_0"), val = tensor([12, 0, 0, 0, 0])]; tensor v_cache_25_end_0 = const()[name = tensor("v_cache_25_end_0"), val = tensor([13, 1, 4, 2048, 128])]; tensor v_cache_25_end_mask_0 = const()[name = tensor("v_cache_25_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_25_squeeze_mask_0 = const()[name = tensor("v_cache_25_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_25_cast_fp16 = slice_by_index(begin = v_cache_25_begin_0, end = v_cache_25_end_0, end_mask = v_cache_25_end_mask_0, squeeze_mask = v_cache_25_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_25_cast_fp16")]; tensor var_2199 = const()[name = tensor("op_2199"), val = tensor(-1)]; tensor var_2198_promoted = const()[name = tensor("op_2198_promoted"), val = tensor(0x1p+1)]; tensor x_481_cast_fp16_to_fp32 = cast(dtype = x_481_cast_fp16_to_fp32_dtype_0, x = x_481_cast_fp16)[name = tensor("cast_470")]; tensor var_2208 = pow(x = x_481_cast_fp16_to_fp32, y = var_2198_promoted)[name = tensor("op_2208")]; tensor var_97_axes_0 = const()[name = tensor("var_97_axes_0"), val = tensor([-1])]; tensor var_97_keep_dims_0 = const()[name = tensor("var_97_keep_dims_0"), val = tensor(true)]; tensor var_97 = reduce_mean(axes = var_97_axes_0, keep_dims = var_97_keep_dims_0, x = var_2208)[name = tensor("var_97")]; tensor var_97_to_fp16_dtype_0 = const()[name = tensor("var_97_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2212_to_fp16 = const()[name = tensor("op_2212_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_97_to_fp16 = cast(dtype = var_97_to_fp16_dtype_0, x = var_97)[name = tensor("cast_469")]; tensor var_2213_cast_fp16 = add(x = var_97_to_fp16, y = var_2212_to_fp16)[name = tensor("op_2213_cast_fp16")]; tensor var_2213_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2213_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2214_epsilon_0 = const()[name = tensor("op_2214_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2213_cast_fp16_to_fp32 = cast(dtype = var_2213_cast_fp16_to_fp32_dtype_0, x = var_2213_cast_fp16)[name = tensor("cast_468")]; tensor var_2214 = rsqrt(epsilon = var_2214_epsilon_0, x = var_2213_cast_fp16_to_fp32)[name = tensor("op_2214")]; tensor var_2214_to_fp16_dtype_0 = const()[name = tensor("op_2214_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2214_to_fp16 = cast(dtype = var_2214_to_fp16_dtype_0, x = var_2214)[name = tensor("cast_467")]; tensor x_487_cast_fp16 = mul(x = x_481_cast_fp16, y = var_2214_to_fp16)[name = tensor("x_487_cast_fp16")]; tensor layers_12_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_12_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(329452352)))]; tensor x_489_cast_fp16 = mul(x = layers_12_input_layernorm_weight_to_fp16, y = x_487_cast_fp16)[name = tensor("x_489_cast_fp16")]; tensor layers_12_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_12_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(329453440)))]; tensor linear_84_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_12_self_attn_q_proj_weight_to_fp16, x = x_489_cast_fp16)[name = tensor("linear_84_cast_fp16")]; tensor var_2228 = const()[name = tensor("op_2228"), val = tensor([1, 1, 12, 128])]; tensor x_491_cast_fp16 = reshape(shape = var_2228, x = linear_84_cast_fp16)[name = tensor("x_491_cast_fp16")]; tensor x_491_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_491_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2198_promoted_1 = const()[name = tensor("op_2198_promoted_1"), val = tensor(0x1p+1)]; tensor x_491_cast_fp16_to_fp32 = cast(dtype = x_491_cast_fp16_to_fp32_dtype_0, x = x_491_cast_fp16)[name = tensor("cast_466")]; tensor var_2232 = pow(x = x_491_cast_fp16_to_fp32, y = var_2198_promoted_1)[name = tensor("op_2232")]; tensor var_99_axes_0 = const()[name = tensor("var_99_axes_0"), val = tensor([-1])]; tensor var_99_keep_dims_0 = const()[name = tensor("var_99_keep_dims_0"), val = tensor(true)]; tensor var_99 = reduce_mean(axes = var_99_axes_0, keep_dims = var_99_keep_dims_0, x = var_2232)[name = tensor("var_99")]; tensor var_99_to_fp16_dtype_0 = const()[name = tensor("var_99_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2236_to_fp16 = const()[name = tensor("op_2236_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_99_to_fp16 = cast(dtype = var_99_to_fp16_dtype_0, x = var_99)[name = tensor("cast_465")]; tensor var_2237_cast_fp16 = add(x = var_99_to_fp16, y = var_2236_to_fp16)[name = tensor("op_2237_cast_fp16")]; tensor var_2237_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2237_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2238_epsilon_0 = const()[name = tensor("op_2238_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2237_cast_fp16_to_fp32 = cast(dtype = var_2237_cast_fp16_to_fp32_dtype_0, x = var_2237_cast_fp16)[name = tensor("cast_464")]; tensor var_2238 = rsqrt(epsilon = var_2238_epsilon_0, x = var_2237_cast_fp16_to_fp32)[name = tensor("op_2238")]; tensor var_2238_to_fp16_dtype_0 = const()[name = tensor("op_2238_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2238_to_fp16 = cast(dtype = var_2238_to_fp16_dtype_0, x = var_2238)[name = tensor("cast_463")]; tensor x_497_cast_fp16 = mul(x = x_491_cast_fp16, y = var_2238_to_fp16)[name = tensor("x_497_cast_fp16")]; tensor layers_12_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_12_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(331026368)))]; tensor var_2240_cast_fp16 = mul(x = layers_12_self_attn_q_norm_weight_to_fp16, y = x_497_cast_fp16)[name = tensor("op_2240_cast_fp16")]; tensor q_49_perm_0 = const()[name = tensor("q_49_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_12_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_12_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(331026688)))]; tensor linear_85_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_12_self_attn_k_proj_weight_to_fp16, x = x_489_cast_fp16)[name = tensor("linear_85_cast_fp16")]; tensor var_2244 = const()[name = tensor("op_2244"), val = tensor([1, 1, 4, 128])]; tensor x_499_cast_fp16 = reshape(shape = var_2244, x = linear_85_cast_fp16)[name = tensor("x_499_cast_fp16")]; tensor x_499_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_499_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2198_promoted_2 = const()[name = tensor("op_2198_promoted_2"), val = tensor(0x1p+1)]; tensor x_499_cast_fp16_to_fp32 = cast(dtype = x_499_cast_fp16_to_fp32_dtype_0, x = x_499_cast_fp16)[name = tensor("cast_462")]; tensor var_2248 = pow(x = x_499_cast_fp16_to_fp32, y = var_2198_promoted_2)[name = tensor("op_2248")]; tensor var_101_axes_0 = const()[name = tensor("var_101_axes_0"), val = tensor([-1])]; tensor var_101_keep_dims_0 = const()[name = tensor("var_101_keep_dims_0"), val = tensor(true)]; tensor var_101 = reduce_mean(axes = var_101_axes_0, keep_dims = var_101_keep_dims_0, x = var_2248)[name = tensor("var_101")]; tensor var_101_to_fp16_dtype_0 = const()[name = tensor("var_101_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2252_to_fp16 = const()[name = tensor("op_2252_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_101_to_fp16 = cast(dtype = var_101_to_fp16_dtype_0, x = var_101)[name = tensor("cast_461")]; tensor var_2253_cast_fp16 = add(x = var_101_to_fp16, y = var_2252_to_fp16)[name = tensor("op_2253_cast_fp16")]; tensor var_2253_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2253_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2254_epsilon_0 = const()[name = tensor("op_2254_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2253_cast_fp16_to_fp32 = cast(dtype = var_2253_cast_fp16_to_fp32_dtype_0, x = var_2253_cast_fp16)[name = tensor("cast_460")]; tensor var_2254 = rsqrt(epsilon = var_2254_epsilon_0, x = var_2253_cast_fp16_to_fp32)[name = tensor("op_2254")]; tensor var_2254_to_fp16_dtype_0 = const()[name = tensor("op_2254_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2254_to_fp16 = cast(dtype = var_2254_to_fp16_dtype_0, x = var_2254)[name = tensor("cast_459")]; tensor x_505_cast_fp16 = mul(x = x_499_cast_fp16, y = var_2254_to_fp16)[name = tensor("x_505_cast_fp16")]; tensor layers_12_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_12_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(331551040)))]; tensor var_2256_cast_fp16 = mul(x = layers_12_self_attn_k_norm_weight_to_fp16, y = x_505_cast_fp16)[name = tensor("op_2256_cast_fp16")]; tensor k_49_perm_0 = const()[name = tensor("k_49_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_12_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_12_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(331551360)))]; tensor linear_86_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_12_self_attn_v_proj_weight_to_fp16, x = x_489_cast_fp16)[name = tensor("linear_86_cast_fp16")]; tensor var_2260 = const()[name = tensor("op_2260"), val = tensor([1, 1, 4, 128])]; tensor var_2261_cast_fp16 = reshape(shape = var_2260, x = linear_86_cast_fp16)[name = tensor("op_2261_cast_fp16")]; tensor v_25_perm_0 = const()[name = tensor("v_25_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_49_cast_fp16 = transpose(perm = q_49_perm_0, x = var_2240_cast_fp16)[name = tensor("transpose_63")]; tensor var_2265_cast_fp16 = mul(x = q_49_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2265_cast_fp16")]; tensor x1_49_begin_0 = const()[name = tensor("x1_49_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_49_end_0 = const()[name = tensor("x1_49_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_49_end_mask_0 = const()[name = tensor("x1_49_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_49_cast_fp16 = slice_by_index(begin = x1_49_begin_0, end = x1_49_end_0, end_mask = x1_49_end_mask_0, x = q_49_cast_fp16)[name = tensor("x1_49_cast_fp16")]; tensor x2_49_begin_0 = const()[name = tensor("x2_49_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_49_end_0 = const()[name = tensor("x2_49_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_49_end_mask_0 = const()[name = tensor("x2_49_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_49_cast_fp16 = slice_by_index(begin = x2_49_begin_0, end = x2_49_end_0, end_mask = x2_49_end_mask_0, x = q_49_cast_fp16)[name = tensor("x2_49_cast_fp16")]; tensor const_25_promoted_to_fp16 = const()[name = tensor("const_25_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2268_cast_fp16 = mul(x = x2_49_cast_fp16, y = const_25_promoted_to_fp16)[name = tensor("op_2268_cast_fp16")]; tensor var_2270_interleave_0 = const()[name = tensor("op_2270_interleave_0"), val = tensor(false)]; tensor var_2270_cast_fp16 = concat(axis = var_2199, interleave = var_2270_interleave_0, values = (var_2268_cast_fp16, x1_49_cast_fp16))[name = tensor("op_2270_cast_fp16")]; tensor var_2271_cast_fp16 = mul(x = var_2270_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2271_cast_fp16")]; tensor q_51_cast_fp16 = add(x = var_2265_cast_fp16, y = var_2271_cast_fp16)[name = tensor("q_51_cast_fp16")]; tensor k_49_cast_fp16 = transpose(perm = k_49_perm_0, x = var_2256_cast_fp16)[name = tensor("transpose_62")]; tensor var_2273_cast_fp16 = mul(x = k_49_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2273_cast_fp16")]; tensor x1_51_begin_0 = const()[name = tensor("x1_51_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_51_end_0 = const()[name = tensor("x1_51_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_51_end_mask_0 = const()[name = tensor("x1_51_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_51_cast_fp16 = slice_by_index(begin = x1_51_begin_0, end = x1_51_end_0, end_mask = x1_51_end_mask_0, x = k_49_cast_fp16)[name = tensor("x1_51_cast_fp16")]; tensor x2_51_begin_0 = const()[name = tensor("x2_51_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_51_end_0 = const()[name = tensor("x2_51_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_51_end_mask_0 = const()[name = tensor("x2_51_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_51_cast_fp16 = slice_by_index(begin = x2_51_begin_0, end = x2_51_end_0, end_mask = x2_51_end_mask_0, x = k_49_cast_fp16)[name = tensor("x2_51_cast_fp16")]; tensor const_26_promoted_to_fp16 = const()[name = tensor("const_26_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2276_cast_fp16 = mul(x = x2_51_cast_fp16, y = const_26_promoted_to_fp16)[name = tensor("op_2276_cast_fp16")]; tensor var_2278_interleave_0 = const()[name = tensor("op_2278_interleave_0"), val = tensor(false)]; tensor var_2278_cast_fp16 = concat(axis = var_2199, interleave = var_2278_interleave_0, values = (var_2276_cast_fp16, x1_51_cast_fp16))[name = tensor("op_2278_cast_fp16")]; tensor var_2279_cast_fp16 = mul(x = var_2278_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2279_cast_fp16")]; tensor k_51_cast_fp16 = add(x = var_2273_cast_fp16, y = var_2279_cast_fp16)[name = tensor("k_51_cast_fp16")]; tensor var_2282_cast_fp16 = mul(x = k_cache_25_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2282_cast_fp16")]; tensor var_2283_cast_fp16 = mul(x = k_51_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2283_cast_fp16")]; tensor k_full_25_cast_fp16 = add(x = var_2282_cast_fp16, y = var_2283_cast_fp16)[name = tensor("k_full_25_cast_fp16")]; tensor var_2286_cast_fp16 = mul(x = v_cache_25_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2286_cast_fp16")]; tensor v_25_cast_fp16 = transpose(perm = v_25_perm_0, x = var_2261_cast_fp16)[name = tensor("transpose_61")]; tensor var_2287_cast_fp16 = mul(x = v_25_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2287_cast_fp16")]; tensor v_full_25_cast_fp16 = add(x = var_2286_cast_fp16, y = var_2287_cast_fp16)[name = tensor("v_full_25_cast_fp16")]; tensor var_2289_axes_0 = const()[name = tensor("op_2289_axes_0"), val = tensor([2])]; tensor var_2289_cast_fp16 = expand_dims(axes = var_2289_axes_0, x = k_full_25_cast_fp16)[name = tensor("op_2289_cast_fp16")]; tensor var_2291_reps_0 = const()[name = tensor("op_2291_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2291_cast_fp16 = tile(reps = var_2291_reps_0, x = var_2289_cast_fp16)[name = tensor("op_2291_cast_fp16")]; tensor var_2292 = const()[name = tensor("op_2292"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_25_cast_fp16 = reshape(shape = var_2292, x = var_2291_cast_fp16)[name = tensor("k_rep_25_cast_fp16")]; tensor var_2294_axes_0 = const()[name = tensor("op_2294_axes_0"), val = tensor([2])]; tensor var_2294_cast_fp16 = expand_dims(axes = var_2294_axes_0, x = v_full_25_cast_fp16)[name = tensor("op_2294_cast_fp16")]; tensor var_2296_reps_0 = const()[name = tensor("op_2296_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2296_cast_fp16 = tile(reps = var_2296_reps_0, x = var_2294_cast_fp16)[name = tensor("op_2296_cast_fp16")]; tensor var_2297 = const()[name = tensor("op_2297"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_25_cast_fp16 = reshape(shape = var_2297, x = var_2296_cast_fp16)[name = tensor("v_rep_25_cast_fp16")]; tensor var_2300_transpose_x_1 = const()[name = tensor("op_2300_transpose_x_1"), val = tensor(false)]; tensor var_2300_transpose_y_1 = const()[name = tensor("op_2300_transpose_y_1"), val = tensor(true)]; tensor var_2300_cast_fp16 = matmul(transpose_x = var_2300_transpose_x_1, transpose_y = var_2300_transpose_y_1, x = q_51_cast_fp16, y = k_rep_25_cast_fp16)[name = tensor("op_2300_cast_fp16")]; tensor var_2301_to_fp16 = const()[name = tensor("op_2301_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_49_cast_fp16 = mul(x = var_2300_cast_fp16, y = var_2301_to_fp16)[name = tensor("attn_49_cast_fp16")]; tensor input_51_cast_fp16 = add(x = attn_49_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_51_cast_fp16")]; tensor input_51_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_51_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_51_cast_fp16_to_fp32 = cast(dtype = input_51_cast_fp16_to_fp32_dtype_0, x = input_51_cast_fp16)[name = tensor("cast_458")]; tensor attn_51 = softmax(axis = var_2199, x = input_51_cast_fp16_to_fp32)[name = tensor("attn_51")]; tensor out_25_transpose_x_0 = const()[name = tensor("out_25_transpose_x_0"), val = tensor(false)]; tensor out_25_transpose_y_0 = const()[name = tensor("out_25_transpose_y_0"), val = tensor(false)]; tensor attn_51_to_fp16_dtype_0 = const()[name = tensor("attn_51_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_51_to_fp16 = cast(dtype = attn_51_to_fp16_dtype_0, x = attn_51)[name = tensor("cast_457")]; tensor out_25_cast_fp16 = matmul(transpose_x = out_25_transpose_x_0, transpose_y = out_25_transpose_y_0, x = attn_51_to_fp16, y = v_rep_25_cast_fp16)[name = tensor("out_25_cast_fp16")]; tensor var_2306_perm_0 = const()[name = tensor("op_2306_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2308 = const()[name = tensor("op_2308"), val = tensor([1, 1, 1536])]; tensor var_2306_cast_fp16 = transpose(perm = var_2306_perm_0, x = out_25_cast_fp16)[name = tensor("transpose_60")]; tensor x_507_cast_fp16 = reshape(shape = var_2308, x = var_2306_cast_fp16)[name = tensor("x_507_cast_fp16")]; tensor layers_12_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_12_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(332075712)))]; tensor linear_87_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_12_self_attn_o_proj_weight_to_fp16, x = x_507_cast_fp16)[name = tensor("linear_87_cast_fp16")]; tensor x_509_cast_fp16 = add(x = x_481_cast_fp16, y = linear_87_cast_fp16)[name = tensor("x_509_cast_fp16")]; tensor x_509_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_509_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2198_promoted_3 = const()[name = tensor("op_2198_promoted_3"), val = tensor(0x1p+1)]; tensor x_509_cast_fp16_to_fp32 = cast(dtype = x_509_cast_fp16_to_fp32_dtype_0, x = x_509_cast_fp16)[name = tensor("cast_456")]; tensor var_2319 = pow(x = x_509_cast_fp16_to_fp32, y = var_2198_promoted_3)[name = tensor("op_2319")]; tensor var_103_axes_0 = const()[name = tensor("var_103_axes_0"), val = tensor([-1])]; tensor var_103_keep_dims_0 = const()[name = tensor("var_103_keep_dims_0"), val = tensor(true)]; tensor var_103 = reduce_mean(axes = var_103_axes_0, keep_dims = var_103_keep_dims_0, x = var_2319)[name = tensor("var_103")]; tensor var_103_to_fp16_dtype_0 = const()[name = tensor("var_103_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2323_to_fp16 = const()[name = tensor("op_2323_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_103_to_fp16 = cast(dtype = var_103_to_fp16_dtype_0, x = var_103)[name = tensor("cast_455")]; tensor var_2324_cast_fp16 = add(x = var_103_to_fp16, y = var_2323_to_fp16)[name = tensor("op_2324_cast_fp16")]; tensor var_2324_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2324_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2325_epsilon_0 = const()[name = tensor("op_2325_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2324_cast_fp16_to_fp32 = cast(dtype = var_2324_cast_fp16_to_fp32_dtype_0, x = var_2324_cast_fp16)[name = tensor("cast_454")]; tensor var_2325 = rsqrt(epsilon = var_2325_epsilon_0, x = var_2324_cast_fp16_to_fp32)[name = tensor("op_2325")]; tensor var_2325_to_fp16_dtype_0 = const()[name = tensor("op_2325_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2325_to_fp16 = cast(dtype = var_2325_to_fp16_dtype_0, x = var_2325)[name = tensor("cast_453")]; tensor x_515_cast_fp16 = mul(x = x_509_cast_fp16, y = var_2325_to_fp16)[name = tensor("x_515_cast_fp16")]; tensor layers_12_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_12_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(333648640)))]; tensor x_517_cast_fp16 = mul(x = layers_12_post_attention_layernorm_weight_to_fp16, y = x_515_cast_fp16)[name = tensor("x_517_cast_fp16")]; tensor layers_12_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_12_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(333649728)))]; tensor linear_88_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_12_mlp_gate_proj_weight_to_fp16, x = x_517_cast_fp16)[name = tensor("linear_88_cast_fp16")]; tensor var_2336_cast_fp16 = silu(x = linear_88_cast_fp16)[name = tensor("op_2336_cast_fp16")]; tensor layers_12_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_12_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(335222656)))]; tensor linear_89_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_12_mlp_up_proj_weight_to_fp16, x = x_517_cast_fp16)[name = tensor("linear_89_cast_fp16")]; tensor x_519_cast_fp16 = mul(x = var_2336_cast_fp16, y = linear_89_cast_fp16)[name = tensor("x_519_cast_fp16")]; tensor layers_12_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_12_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(336795584)))]; tensor linear_90_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_12_mlp_down_proj_weight_to_fp16, x = x_519_cast_fp16)[name = tensor("linear_90_cast_fp16")]; tensor x_521_cast_fp16 = add(x = x_509_cast_fp16, y = linear_90_cast_fp16)[name = tensor("x_521_cast_fp16")]; tensor x_521_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_521_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_27_begin_0 = const()[name = tensor("k_cache_27_begin_0"), val = tensor([13, 0, 0, 0, 0])]; tensor k_cache_27_end_0 = const()[name = tensor("k_cache_27_end_0"), val = tensor([14, 1, 4, 2048, 128])]; tensor k_cache_27_end_mask_0 = const()[name = tensor("k_cache_27_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_27_squeeze_mask_0 = const()[name = tensor("k_cache_27_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_27_cast_fp16 = slice_by_index(begin = k_cache_27_begin_0, end = k_cache_27_end_0, end_mask = k_cache_27_end_mask_0, squeeze_mask = k_cache_27_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_27_cast_fp16")]; tensor v_cache_27_begin_0 = const()[name = tensor("v_cache_27_begin_0"), val = tensor([13, 0, 0, 0, 0])]; tensor v_cache_27_end_0 = const()[name = tensor("v_cache_27_end_0"), val = tensor([14, 1, 4, 2048, 128])]; tensor v_cache_27_end_mask_0 = const()[name = tensor("v_cache_27_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_27_squeeze_mask_0 = const()[name = tensor("v_cache_27_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_27_cast_fp16 = slice_by_index(begin = v_cache_27_begin_0, end = v_cache_27_end_0, end_mask = v_cache_27_end_mask_0, squeeze_mask = v_cache_27_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_27_cast_fp16")]; tensor var_2368 = const()[name = tensor("op_2368"), val = tensor(-1)]; tensor var_2367_promoted = const()[name = tensor("op_2367_promoted"), val = tensor(0x1p+1)]; tensor x_521_cast_fp16_to_fp32 = cast(dtype = x_521_cast_fp16_to_fp32_dtype_0, x = x_521_cast_fp16)[name = tensor("cast_452")]; tensor var_2377 = pow(x = x_521_cast_fp16_to_fp32, y = var_2367_promoted)[name = tensor("op_2377")]; tensor var_105_axes_0 = const()[name = tensor("var_105_axes_0"), val = tensor([-1])]; tensor var_105_keep_dims_0 = const()[name = tensor("var_105_keep_dims_0"), val = tensor(true)]; tensor var_105_0 = reduce_mean(axes = var_105_axes_0, keep_dims = var_105_keep_dims_0, x = var_2377)[name = tensor("var_105")]; tensor var_105_to_fp16_dtype_0 = const()[name = tensor("var_105_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2381_to_fp16 = const()[name = tensor("op_2381_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_105_to_fp16 = cast(dtype = var_105_to_fp16_dtype_0, x = var_105_0)[name = tensor("cast_451")]; tensor var_2382_cast_fp16 = add(x = var_105_to_fp16, y = var_2381_to_fp16)[name = tensor("op_2382_cast_fp16")]; tensor var_2382_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2382_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2383_epsilon_0 = const()[name = tensor("op_2383_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2382_cast_fp16_to_fp32 = cast(dtype = var_2382_cast_fp16_to_fp32_dtype_0, x = var_2382_cast_fp16)[name = tensor("cast_450")]; tensor var_2383 = rsqrt(epsilon = var_2383_epsilon_0, x = var_2382_cast_fp16_to_fp32)[name = tensor("op_2383")]; tensor var_2383_to_fp16_dtype_0 = const()[name = tensor("op_2383_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2383_to_fp16 = cast(dtype = var_2383_to_fp16_dtype_0, x = var_2383)[name = tensor("cast_449")]; tensor x_527_cast_fp16 = mul(x = x_521_cast_fp16, y = var_2383_to_fp16)[name = tensor("x_527_cast_fp16")]; tensor layers_13_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_13_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(338368512)))]; tensor x_529_cast_fp16 = mul(x = layers_13_input_layernorm_weight_to_fp16, y = x_527_cast_fp16)[name = tensor("x_529_cast_fp16")]; tensor layers_13_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_13_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(338369600)))]; tensor linear_91_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_13_self_attn_q_proj_weight_to_fp16, x = x_529_cast_fp16)[name = tensor("linear_91_cast_fp16")]; tensor var_2397 = const()[name = tensor("op_2397"), val = tensor([1, 1, 12, 128])]; tensor x_531_cast_fp16 = reshape(shape = var_2397, x = linear_91_cast_fp16)[name = tensor("x_531_cast_fp16")]; tensor x_531_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_531_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2367_promoted_1 = const()[name = tensor("op_2367_promoted_1"), val = tensor(0x1p+1)]; tensor x_531_cast_fp16_to_fp32 = cast(dtype = x_531_cast_fp16_to_fp32_dtype_0, x = x_531_cast_fp16)[name = tensor("cast_448")]; tensor var_2401 = pow(x = x_531_cast_fp16_to_fp32, y = var_2367_promoted_1)[name = tensor("op_2401")]; tensor var_107_axes_0 = const()[name = tensor("var_107_axes_0"), val = tensor([-1])]; tensor var_107_keep_dims_0 = const()[name = tensor("var_107_keep_dims_0"), val = tensor(true)]; tensor var_107_0 = reduce_mean(axes = var_107_axes_0, keep_dims = var_107_keep_dims_0, x = var_2401)[name = tensor("var_107")]; tensor var_107_to_fp16_dtype_0 = const()[name = tensor("var_107_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2405_to_fp16 = const()[name = tensor("op_2405_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_107_to_fp16_0 = cast(dtype = var_107_to_fp16_dtype_0, x = var_107_0)[name = tensor("cast_447")]; tensor var_2406_cast_fp16 = add(x = var_107_to_fp16_0, y = var_2405_to_fp16)[name = tensor("op_2406_cast_fp16")]; tensor var_2406_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2406_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2407_epsilon_0 = const()[name = tensor("op_2407_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2406_cast_fp16_to_fp32 = cast(dtype = var_2406_cast_fp16_to_fp32_dtype_0, x = var_2406_cast_fp16)[name = tensor("cast_446")]; tensor var_2407 = rsqrt(epsilon = var_2407_epsilon_0, x = var_2406_cast_fp16_to_fp32)[name = tensor("op_2407")]; tensor var_2407_to_fp16_dtype_0 = const()[name = tensor("op_2407_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2407_to_fp16 = cast(dtype = var_2407_to_fp16_dtype_0, x = var_2407)[name = tensor("cast_445")]; tensor x_537_cast_fp16 = mul(x = x_531_cast_fp16, y = var_2407_to_fp16)[name = tensor("x_537_cast_fp16")]; tensor layers_13_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_13_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(339942528)))]; tensor var_2409_cast_fp16 = mul(x = layers_13_self_attn_q_norm_weight_to_fp16, y = x_537_cast_fp16)[name = tensor("op_2409_cast_fp16")]; tensor q_53_perm_0 = const()[name = tensor("q_53_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_13_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_13_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(339942848)))]; tensor linear_92_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_13_self_attn_k_proj_weight_to_fp16, x = x_529_cast_fp16)[name = tensor("linear_92_cast_fp16")]; tensor var_2413 = const()[name = tensor("op_2413"), val = tensor([1, 1, 4, 128])]; tensor x_539_cast_fp16 = reshape(shape = var_2413, x = linear_92_cast_fp16)[name = tensor("x_539_cast_fp16")]; tensor x_539_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_539_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2367_promoted_2 = const()[name = tensor("op_2367_promoted_2"), val = tensor(0x1p+1)]; tensor x_539_cast_fp16_to_fp32 = cast(dtype = x_539_cast_fp16_to_fp32_dtype_0, x = x_539_cast_fp16)[name = tensor("cast_444")]; tensor var_2417 = pow(x = x_539_cast_fp16_to_fp32, y = var_2367_promoted_2)[name = tensor("op_2417")]; tensor var_109_axes_0 = const()[name = tensor("var_109_axes_0"), val = tensor([-1])]; tensor var_109_keep_dims_0 = const()[name = tensor("var_109_keep_dims_0"), val = tensor(true)]; tensor var_109 = reduce_mean(axes = var_109_axes_0, keep_dims = var_109_keep_dims_0, x = var_2417)[name = tensor("var_109")]; tensor var_109_to_fp16_dtype_0 = const()[name = tensor("var_109_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2421_to_fp16 = const()[name = tensor("op_2421_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_109_to_fp16 = cast(dtype = var_109_to_fp16_dtype_0, x = var_109)[name = tensor("cast_443")]; tensor var_2422_cast_fp16 = add(x = var_109_to_fp16, y = var_2421_to_fp16)[name = tensor("op_2422_cast_fp16")]; tensor var_2422_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2422_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2423_epsilon_0 = const()[name = tensor("op_2423_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2422_cast_fp16_to_fp32 = cast(dtype = var_2422_cast_fp16_to_fp32_dtype_0, x = var_2422_cast_fp16)[name = tensor("cast_442")]; tensor var_2423 = rsqrt(epsilon = var_2423_epsilon_0, x = var_2422_cast_fp16_to_fp32)[name = tensor("op_2423")]; tensor var_2423_to_fp16_dtype_0 = const()[name = tensor("op_2423_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2423_to_fp16 = cast(dtype = var_2423_to_fp16_dtype_0, x = var_2423)[name = tensor("cast_441")]; tensor x_545_cast_fp16 = mul(x = x_539_cast_fp16, y = var_2423_to_fp16)[name = tensor("x_545_cast_fp16")]; tensor layers_13_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_13_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(340467200)))]; tensor var_2425_cast_fp16 = mul(x = layers_13_self_attn_k_norm_weight_to_fp16, y = x_545_cast_fp16)[name = tensor("op_2425_cast_fp16")]; tensor k_53_perm_0 = const()[name = tensor("k_53_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_13_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_13_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(340467520)))]; tensor linear_93_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_13_self_attn_v_proj_weight_to_fp16, x = x_529_cast_fp16)[name = tensor("linear_93_cast_fp16")]; tensor var_2429 = const()[name = tensor("op_2429"), val = tensor([1, 1, 4, 128])]; tensor var_2430_cast_fp16 = reshape(shape = var_2429, x = linear_93_cast_fp16)[name = tensor("op_2430_cast_fp16")]; tensor v_27_perm_0 = const()[name = tensor("v_27_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_53_cast_fp16 = transpose(perm = q_53_perm_0, x = var_2409_cast_fp16)[name = tensor("transpose_59")]; tensor var_2434_cast_fp16 = mul(x = q_53_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2434_cast_fp16")]; tensor x1_53_begin_0 = const()[name = tensor("x1_53_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_53_end_0 = const()[name = tensor("x1_53_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_53_end_mask_0 = const()[name = tensor("x1_53_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_53_cast_fp16 = slice_by_index(begin = x1_53_begin_0, end = x1_53_end_0, end_mask = x1_53_end_mask_0, x = q_53_cast_fp16)[name = tensor("x1_53_cast_fp16")]; tensor x2_53_begin_0 = const()[name = tensor("x2_53_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_53_end_0 = const()[name = tensor("x2_53_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_53_end_mask_0 = const()[name = tensor("x2_53_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_53_cast_fp16 = slice_by_index(begin = x2_53_begin_0, end = x2_53_end_0, end_mask = x2_53_end_mask_0, x = q_53_cast_fp16)[name = tensor("x2_53_cast_fp16")]; tensor const_27_promoted_to_fp16 = const()[name = tensor("const_27_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2437_cast_fp16 = mul(x = x2_53_cast_fp16, y = const_27_promoted_to_fp16)[name = tensor("op_2437_cast_fp16")]; tensor var_2439_interleave_0 = const()[name = tensor("op_2439_interleave_0"), val = tensor(false)]; tensor var_2439_cast_fp16 = concat(axis = var_2368, interleave = var_2439_interleave_0, values = (var_2437_cast_fp16, x1_53_cast_fp16))[name = tensor("op_2439_cast_fp16")]; tensor var_2440_cast_fp16 = mul(x = var_2439_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2440_cast_fp16")]; tensor q_55_cast_fp16 = add(x = var_2434_cast_fp16, y = var_2440_cast_fp16)[name = tensor("q_55_cast_fp16")]; tensor k_53_cast_fp16 = transpose(perm = k_53_perm_0, x = var_2425_cast_fp16)[name = tensor("transpose_58")]; tensor var_2442_cast_fp16 = mul(x = k_53_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2442_cast_fp16")]; tensor x1_55_begin_0 = const()[name = tensor("x1_55_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_55_end_0 = const()[name = tensor("x1_55_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_55_end_mask_0 = const()[name = tensor("x1_55_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_55_cast_fp16 = slice_by_index(begin = x1_55_begin_0, end = x1_55_end_0, end_mask = x1_55_end_mask_0, x = k_53_cast_fp16)[name = tensor("x1_55_cast_fp16")]; tensor x2_55_begin_0 = const()[name = tensor("x2_55_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_55_end_0 = const()[name = tensor("x2_55_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_55_end_mask_0 = const()[name = tensor("x2_55_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_55_cast_fp16 = slice_by_index(begin = x2_55_begin_0, end = x2_55_end_0, end_mask = x2_55_end_mask_0, x = k_53_cast_fp16)[name = tensor("x2_55_cast_fp16")]; tensor const_28_promoted_to_fp16 = const()[name = tensor("const_28_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2445_cast_fp16 = mul(x = x2_55_cast_fp16, y = const_28_promoted_to_fp16)[name = tensor("op_2445_cast_fp16")]; tensor var_2447_interleave_0 = const()[name = tensor("op_2447_interleave_0"), val = tensor(false)]; tensor var_2447_cast_fp16 = concat(axis = var_2368, interleave = var_2447_interleave_0, values = (var_2445_cast_fp16, x1_55_cast_fp16))[name = tensor("op_2447_cast_fp16")]; tensor var_2448_cast_fp16 = mul(x = var_2447_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2448_cast_fp16")]; tensor k_55_cast_fp16 = add(x = var_2442_cast_fp16, y = var_2448_cast_fp16)[name = tensor("k_55_cast_fp16")]; tensor var_2451_cast_fp16 = mul(x = k_cache_27_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2451_cast_fp16")]; tensor var_2452_cast_fp16 = mul(x = k_55_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2452_cast_fp16")]; tensor k_full_27_cast_fp16 = add(x = var_2451_cast_fp16, y = var_2452_cast_fp16)[name = tensor("k_full_27_cast_fp16")]; tensor var_2455_cast_fp16 = mul(x = v_cache_27_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2455_cast_fp16")]; tensor v_27_cast_fp16 = transpose(perm = v_27_perm_0, x = var_2430_cast_fp16)[name = tensor("transpose_57")]; tensor var_2456_cast_fp16 = mul(x = v_27_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2456_cast_fp16")]; tensor v_full_27_cast_fp16 = add(x = var_2455_cast_fp16, y = var_2456_cast_fp16)[name = tensor("v_full_27_cast_fp16")]; tensor var_2458_axes_0 = const()[name = tensor("op_2458_axes_0"), val = tensor([2])]; tensor var_2458_cast_fp16 = expand_dims(axes = var_2458_axes_0, x = k_full_27_cast_fp16)[name = tensor("op_2458_cast_fp16")]; tensor var_2460_reps_0 = const()[name = tensor("op_2460_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2460_cast_fp16 = tile(reps = var_2460_reps_0, x = var_2458_cast_fp16)[name = tensor("op_2460_cast_fp16")]; tensor var_2461 = const()[name = tensor("op_2461"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_27_cast_fp16 = reshape(shape = var_2461, x = var_2460_cast_fp16)[name = tensor("k_rep_27_cast_fp16")]; tensor var_2463_axes_0 = const()[name = tensor("op_2463_axes_0"), val = tensor([2])]; tensor var_2463_cast_fp16 = expand_dims(axes = var_2463_axes_0, x = v_full_27_cast_fp16)[name = tensor("op_2463_cast_fp16")]; tensor var_2465_reps_0 = const()[name = tensor("op_2465_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2465_cast_fp16 = tile(reps = var_2465_reps_0, x = var_2463_cast_fp16)[name = tensor("op_2465_cast_fp16")]; tensor var_2466 = const()[name = tensor("op_2466"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_27_cast_fp16 = reshape(shape = var_2466, x = var_2465_cast_fp16)[name = tensor("v_rep_27_cast_fp16")]; tensor var_2469_transpose_x_1 = const()[name = tensor("op_2469_transpose_x_1"), val = tensor(false)]; tensor var_2469_transpose_y_1 = const()[name = tensor("op_2469_transpose_y_1"), val = tensor(true)]; tensor var_2469_cast_fp16 = matmul(transpose_x = var_2469_transpose_x_1, transpose_y = var_2469_transpose_y_1, x = q_55_cast_fp16, y = k_rep_27_cast_fp16)[name = tensor("op_2469_cast_fp16")]; tensor var_2470_to_fp16 = const()[name = tensor("op_2470_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_53_cast_fp16 = mul(x = var_2469_cast_fp16, y = var_2470_to_fp16)[name = tensor("attn_53_cast_fp16")]; tensor input_55_cast_fp16 = add(x = attn_53_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_55_cast_fp16")]; tensor input_55_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_55_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_55_cast_fp16_to_fp32 = cast(dtype = input_55_cast_fp16_to_fp32_dtype_0, x = input_55_cast_fp16)[name = tensor("cast_440")]; tensor attn_55 = softmax(axis = var_2368, x = input_55_cast_fp16_to_fp32)[name = tensor("attn_55")]; tensor out_27_transpose_x_0 = const()[name = tensor("out_27_transpose_x_0"), val = tensor(false)]; tensor out_27_transpose_y_0 = const()[name = tensor("out_27_transpose_y_0"), val = tensor(false)]; tensor attn_55_to_fp16_dtype_0 = const()[name = tensor("attn_55_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_55_to_fp16 = cast(dtype = attn_55_to_fp16_dtype_0, x = attn_55)[name = tensor("cast_439")]; tensor out_27_cast_fp16 = matmul(transpose_x = out_27_transpose_x_0, transpose_y = out_27_transpose_y_0, x = attn_55_to_fp16, y = v_rep_27_cast_fp16)[name = tensor("out_27_cast_fp16")]; tensor var_2475_perm_0 = const()[name = tensor("op_2475_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2477 = const()[name = tensor("op_2477"), val = tensor([1, 1, 1536])]; tensor var_2475_cast_fp16 = transpose(perm = var_2475_perm_0, x = out_27_cast_fp16)[name = tensor("transpose_56")]; tensor x_547_cast_fp16 = reshape(shape = var_2477, x = var_2475_cast_fp16)[name = tensor("x_547_cast_fp16")]; tensor layers_13_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_13_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(340991872)))]; tensor linear_94_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_13_self_attn_o_proj_weight_to_fp16, x = x_547_cast_fp16)[name = tensor("linear_94_cast_fp16")]; tensor x_549_cast_fp16 = add(x = x_521_cast_fp16, y = linear_94_cast_fp16)[name = tensor("x_549_cast_fp16")]; tensor x_549_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_549_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2367_promoted_3 = const()[name = tensor("op_2367_promoted_3"), val = tensor(0x1p+1)]; tensor x_549_cast_fp16_to_fp32 = cast(dtype = x_549_cast_fp16_to_fp32_dtype_0, x = x_549_cast_fp16)[name = tensor("cast_438")]; tensor var_2488 = pow(x = x_549_cast_fp16_to_fp32, y = var_2367_promoted_3)[name = tensor("op_2488")]; tensor var_111_axes_0 = const()[name = tensor("var_111_axes_0"), val = tensor([-1])]; tensor var_111_keep_dims_0 = const()[name = tensor("var_111_keep_dims_0"), val = tensor(true)]; tensor var_111 = reduce_mean(axes = var_111_axes_0, keep_dims = var_111_keep_dims_0, x = var_2488)[name = tensor("var_111")]; tensor var_111_to_fp16_dtype_0 = const()[name = tensor("var_111_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2492_to_fp16 = const()[name = tensor("op_2492_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_111_to_fp16 = cast(dtype = var_111_to_fp16_dtype_0, x = var_111)[name = tensor("cast_437")]; tensor var_2493_cast_fp16 = add(x = var_111_to_fp16, y = var_2492_to_fp16)[name = tensor("op_2493_cast_fp16")]; tensor var_2493_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2493_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2494_epsilon_0 = const()[name = tensor("op_2494_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2493_cast_fp16_to_fp32 = cast(dtype = var_2493_cast_fp16_to_fp32_dtype_0, x = var_2493_cast_fp16)[name = tensor("cast_436")]; tensor var_2494 = rsqrt(epsilon = var_2494_epsilon_0, x = var_2493_cast_fp16_to_fp32)[name = tensor("op_2494")]; tensor var_2494_to_fp16_dtype_0 = const()[name = tensor("op_2494_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2494_to_fp16 = cast(dtype = var_2494_to_fp16_dtype_0, x = var_2494)[name = tensor("cast_435")]; tensor x_555_cast_fp16 = mul(x = x_549_cast_fp16, y = var_2494_to_fp16)[name = tensor("x_555_cast_fp16")]; tensor layers_13_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_13_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(342564800)))]; tensor x_557_cast_fp16 = mul(x = layers_13_post_attention_layernorm_weight_to_fp16, y = x_555_cast_fp16)[name = tensor("x_557_cast_fp16")]; tensor layers_13_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_13_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(342565888)))]; tensor linear_95_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_13_mlp_gate_proj_weight_to_fp16, x = x_557_cast_fp16)[name = tensor("linear_95_cast_fp16")]; tensor var_2505_cast_fp16 = silu(x = linear_95_cast_fp16)[name = tensor("op_2505_cast_fp16")]; tensor layers_13_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_13_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(344138816)))]; tensor linear_96_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_13_mlp_up_proj_weight_to_fp16, x = x_557_cast_fp16)[name = tensor("linear_96_cast_fp16")]; tensor x_559_cast_fp16 = mul(x = var_2505_cast_fp16, y = linear_96_cast_fp16)[name = tensor("x_559_cast_fp16")]; tensor layers_13_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_13_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(345711744)))]; tensor linear_97_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_13_mlp_down_proj_weight_to_fp16, x = x_559_cast_fp16)[name = tensor("linear_97_cast_fp16")]; tensor x_561_cast_fp16 = add(x = x_549_cast_fp16, y = linear_97_cast_fp16)[name = tensor("x_561_cast_fp16")]; tensor x_561_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_561_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_29_begin_0 = const()[name = tensor("k_cache_29_begin_0"), val = tensor([14, 0, 0, 0, 0])]; tensor k_cache_29_end_0 = const()[name = tensor("k_cache_29_end_0"), val = tensor([15, 1, 4, 2048, 128])]; tensor k_cache_29_end_mask_0 = const()[name = tensor("k_cache_29_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_29_squeeze_mask_0 = const()[name = tensor("k_cache_29_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_29_cast_fp16 = slice_by_index(begin = k_cache_29_begin_0, end = k_cache_29_end_0, end_mask = k_cache_29_end_mask_0, squeeze_mask = k_cache_29_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_29_cast_fp16")]; tensor v_cache_29_begin_0 = const()[name = tensor("v_cache_29_begin_0"), val = tensor([14, 0, 0, 0, 0])]; tensor v_cache_29_end_0 = const()[name = tensor("v_cache_29_end_0"), val = tensor([15, 1, 4, 2048, 128])]; tensor v_cache_29_end_mask_0 = const()[name = tensor("v_cache_29_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_29_squeeze_mask_0 = const()[name = tensor("v_cache_29_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_29_cast_fp16 = slice_by_index(begin = v_cache_29_begin_0, end = v_cache_29_end_0, end_mask = v_cache_29_end_mask_0, squeeze_mask = v_cache_29_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_29_cast_fp16")]; tensor var_2537 = const()[name = tensor("op_2537"), val = tensor(-1)]; tensor var_2536_promoted = const()[name = tensor("op_2536_promoted"), val = tensor(0x1p+1)]; tensor x_561_cast_fp16_to_fp32 = cast(dtype = x_561_cast_fp16_to_fp32_dtype_0, x = x_561_cast_fp16)[name = tensor("cast_434")]; tensor var_2546 = pow(x = x_561_cast_fp16_to_fp32, y = var_2536_promoted)[name = tensor("op_2546")]; tensor var_113_axes_0 = const()[name = tensor("var_113_axes_0"), val = tensor([-1])]; tensor var_113_keep_dims_0 = const()[name = tensor("var_113_keep_dims_0"), val = tensor(true)]; tensor var_113 = reduce_mean(axes = var_113_axes_0, keep_dims = var_113_keep_dims_0, x = var_2546)[name = tensor("var_113")]; tensor var_113_to_fp16_dtype_0 = const()[name = tensor("var_113_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2550_to_fp16 = const()[name = tensor("op_2550_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_113_to_fp16 = cast(dtype = var_113_to_fp16_dtype_0, x = var_113)[name = tensor("cast_433")]; tensor var_2551_cast_fp16 = add(x = var_113_to_fp16, y = var_2550_to_fp16)[name = tensor("op_2551_cast_fp16")]; tensor var_2551_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2551_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2552_epsilon_0 = const()[name = tensor("op_2552_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2551_cast_fp16_to_fp32 = cast(dtype = var_2551_cast_fp16_to_fp32_dtype_0, x = var_2551_cast_fp16)[name = tensor("cast_432")]; tensor var_2552 = rsqrt(epsilon = var_2552_epsilon_0, x = var_2551_cast_fp16_to_fp32)[name = tensor("op_2552")]; tensor var_2552_to_fp16_dtype_0 = const()[name = tensor("op_2552_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2552_to_fp16 = cast(dtype = var_2552_to_fp16_dtype_0, x = var_2552)[name = tensor("cast_431")]; tensor x_567_cast_fp16 = mul(x = x_561_cast_fp16, y = var_2552_to_fp16)[name = tensor("x_567_cast_fp16")]; tensor layers_14_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_14_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(347284672)))]; tensor x_569_cast_fp16 = mul(x = layers_14_input_layernorm_weight_to_fp16, y = x_567_cast_fp16)[name = tensor("x_569_cast_fp16")]; tensor layers_14_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_14_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(347285760)))]; tensor linear_98_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_14_self_attn_q_proj_weight_to_fp16, x = x_569_cast_fp16)[name = tensor("linear_98_cast_fp16")]; tensor var_2566 = const()[name = tensor("op_2566"), val = tensor([1, 1, 12, 128])]; tensor x_571_cast_fp16 = reshape(shape = var_2566, x = linear_98_cast_fp16)[name = tensor("x_571_cast_fp16")]; tensor x_571_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_571_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2536_promoted_1 = const()[name = tensor("op_2536_promoted_1"), val = tensor(0x1p+1)]; tensor x_571_cast_fp16_to_fp32 = cast(dtype = x_571_cast_fp16_to_fp32_dtype_0, x = x_571_cast_fp16)[name = tensor("cast_430")]; tensor var_2570 = pow(x = x_571_cast_fp16_to_fp32, y = var_2536_promoted_1)[name = tensor("op_2570")]; tensor var_115_axes_0 = const()[name = tensor("var_115_axes_0"), val = tensor([-1])]; tensor var_115_keep_dims_0 = const()[name = tensor("var_115_keep_dims_0"), val = tensor(true)]; tensor var_115 = reduce_mean(axes = var_115_axes_0, keep_dims = var_115_keep_dims_0, x = var_2570)[name = tensor("var_115")]; tensor var_115_to_fp16_dtype_0 = const()[name = tensor("var_115_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2574_to_fp16 = const()[name = tensor("op_2574_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_115_to_fp16 = cast(dtype = var_115_to_fp16_dtype_0, x = var_115)[name = tensor("cast_429")]; tensor var_2575_cast_fp16 = add(x = var_115_to_fp16, y = var_2574_to_fp16)[name = tensor("op_2575_cast_fp16")]; tensor var_2575_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2575_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2576_epsilon_0 = const()[name = tensor("op_2576_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2575_cast_fp16_to_fp32 = cast(dtype = var_2575_cast_fp16_to_fp32_dtype_0, x = var_2575_cast_fp16)[name = tensor("cast_428")]; tensor var_2576 = rsqrt(epsilon = var_2576_epsilon_0, x = var_2575_cast_fp16_to_fp32)[name = tensor("op_2576")]; tensor var_2576_to_fp16_dtype_0 = const()[name = tensor("op_2576_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2576_to_fp16 = cast(dtype = var_2576_to_fp16_dtype_0, x = var_2576)[name = tensor("cast_427")]; tensor x_577_cast_fp16 = mul(x = x_571_cast_fp16, y = var_2576_to_fp16)[name = tensor("x_577_cast_fp16")]; tensor layers_14_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_14_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(348858688)))]; tensor var_2578_cast_fp16 = mul(x = layers_14_self_attn_q_norm_weight_to_fp16, y = x_577_cast_fp16)[name = tensor("op_2578_cast_fp16")]; tensor q_57_perm_0 = const()[name = tensor("q_57_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_14_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_14_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(348859008)))]; tensor linear_99_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_14_self_attn_k_proj_weight_to_fp16, x = x_569_cast_fp16)[name = tensor("linear_99_cast_fp16")]; tensor var_2582 = const()[name = tensor("op_2582"), val = tensor([1, 1, 4, 128])]; tensor x_579_cast_fp16 = reshape(shape = var_2582, x = linear_99_cast_fp16)[name = tensor("x_579_cast_fp16")]; tensor x_579_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_579_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2536_promoted_2 = const()[name = tensor("op_2536_promoted_2"), val = tensor(0x1p+1)]; tensor x_579_cast_fp16_to_fp32 = cast(dtype = x_579_cast_fp16_to_fp32_dtype_0, x = x_579_cast_fp16)[name = tensor("cast_426")]; tensor var_2586 = pow(x = x_579_cast_fp16_to_fp32, y = var_2536_promoted_2)[name = tensor("op_2586")]; tensor var_117_axes_0 = const()[name = tensor("var_117_axes_0"), val = tensor([-1])]; tensor var_117_keep_dims_0 = const()[name = tensor("var_117_keep_dims_0"), val = tensor(true)]; tensor var_117 = reduce_mean(axes = var_117_axes_0, keep_dims = var_117_keep_dims_0, x = var_2586)[name = tensor("var_117")]; tensor var_117_to_fp16_dtype_0 = const()[name = tensor("var_117_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2590_to_fp16 = const()[name = tensor("op_2590_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_117_to_fp16 = cast(dtype = var_117_to_fp16_dtype_0, x = var_117)[name = tensor("cast_425")]; tensor var_2591_cast_fp16 = add(x = var_117_to_fp16, y = var_2590_to_fp16)[name = tensor("op_2591_cast_fp16")]; tensor var_2591_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2591_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2592_epsilon_0 = const()[name = tensor("op_2592_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2591_cast_fp16_to_fp32 = cast(dtype = var_2591_cast_fp16_to_fp32_dtype_0, x = var_2591_cast_fp16)[name = tensor("cast_424")]; tensor var_2592 = rsqrt(epsilon = var_2592_epsilon_0, x = var_2591_cast_fp16_to_fp32)[name = tensor("op_2592")]; tensor var_2592_to_fp16_dtype_0 = const()[name = tensor("op_2592_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2592_to_fp16 = cast(dtype = var_2592_to_fp16_dtype_0, x = var_2592)[name = tensor("cast_423")]; tensor x_585_cast_fp16 = mul(x = x_579_cast_fp16, y = var_2592_to_fp16)[name = tensor("x_585_cast_fp16")]; tensor layers_14_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_14_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(349383360)))]; tensor var_2594_cast_fp16 = mul(x = layers_14_self_attn_k_norm_weight_to_fp16, y = x_585_cast_fp16)[name = tensor("op_2594_cast_fp16")]; tensor k_57_perm_0 = const()[name = tensor("k_57_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_14_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_14_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(349383680)))]; tensor linear_100_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_14_self_attn_v_proj_weight_to_fp16, x = x_569_cast_fp16)[name = tensor("linear_100_cast_fp16")]; tensor var_2598 = const()[name = tensor("op_2598"), val = tensor([1, 1, 4, 128])]; tensor var_2599_cast_fp16 = reshape(shape = var_2598, x = linear_100_cast_fp16)[name = tensor("op_2599_cast_fp16")]; tensor v_29_perm_0 = const()[name = tensor("v_29_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_57_cast_fp16 = transpose(perm = q_57_perm_0, x = var_2578_cast_fp16)[name = tensor("transpose_55")]; tensor var_2603_cast_fp16 = mul(x = q_57_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2603_cast_fp16")]; tensor x1_57_begin_0 = const()[name = tensor("x1_57_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_57_end_0 = const()[name = tensor("x1_57_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_57_end_mask_0 = const()[name = tensor("x1_57_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_57_cast_fp16 = slice_by_index(begin = x1_57_begin_0, end = x1_57_end_0, end_mask = x1_57_end_mask_0, x = q_57_cast_fp16)[name = tensor("x1_57_cast_fp16")]; tensor x2_57_begin_0 = const()[name = tensor("x2_57_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_57_end_0 = const()[name = tensor("x2_57_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_57_end_mask_0 = const()[name = tensor("x2_57_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_57_cast_fp16 = slice_by_index(begin = x2_57_begin_0, end = x2_57_end_0, end_mask = x2_57_end_mask_0, x = q_57_cast_fp16)[name = tensor("x2_57_cast_fp16")]; tensor const_29_promoted_to_fp16 = const()[name = tensor("const_29_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2606_cast_fp16 = mul(x = x2_57_cast_fp16, y = const_29_promoted_to_fp16)[name = tensor("op_2606_cast_fp16")]; tensor var_2608_interleave_0 = const()[name = tensor("op_2608_interleave_0"), val = tensor(false)]; tensor var_2608_cast_fp16 = concat(axis = var_2537, interleave = var_2608_interleave_0, values = (var_2606_cast_fp16, x1_57_cast_fp16))[name = tensor("op_2608_cast_fp16")]; tensor var_2609_cast_fp16 = mul(x = var_2608_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2609_cast_fp16")]; tensor q_59_cast_fp16 = add(x = var_2603_cast_fp16, y = var_2609_cast_fp16)[name = tensor("q_59_cast_fp16")]; tensor k_57_cast_fp16 = transpose(perm = k_57_perm_0, x = var_2594_cast_fp16)[name = tensor("transpose_54")]; tensor var_2611_cast_fp16 = mul(x = k_57_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2611_cast_fp16")]; tensor x1_59_begin_0 = const()[name = tensor("x1_59_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_59_end_0 = const()[name = tensor("x1_59_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_59_end_mask_0 = const()[name = tensor("x1_59_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_59_cast_fp16 = slice_by_index(begin = x1_59_begin_0, end = x1_59_end_0, end_mask = x1_59_end_mask_0, x = k_57_cast_fp16)[name = tensor("x1_59_cast_fp16")]; tensor x2_59_begin_0 = const()[name = tensor("x2_59_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_59_end_0 = const()[name = tensor("x2_59_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_59_end_mask_0 = const()[name = tensor("x2_59_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_59_cast_fp16 = slice_by_index(begin = x2_59_begin_0, end = x2_59_end_0, end_mask = x2_59_end_mask_0, x = k_57_cast_fp16)[name = tensor("x2_59_cast_fp16")]; tensor const_30_promoted_to_fp16 = const()[name = tensor("const_30_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2614_cast_fp16 = mul(x = x2_59_cast_fp16, y = const_30_promoted_to_fp16)[name = tensor("op_2614_cast_fp16")]; tensor var_2616_interleave_0 = const()[name = tensor("op_2616_interleave_0"), val = tensor(false)]; tensor var_2616_cast_fp16 = concat(axis = var_2537, interleave = var_2616_interleave_0, values = (var_2614_cast_fp16, x1_59_cast_fp16))[name = tensor("op_2616_cast_fp16")]; tensor var_2617_cast_fp16 = mul(x = var_2616_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2617_cast_fp16")]; tensor k_59_cast_fp16 = add(x = var_2611_cast_fp16, y = var_2617_cast_fp16)[name = tensor("k_59_cast_fp16")]; tensor var_2620_cast_fp16 = mul(x = k_cache_29_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2620_cast_fp16")]; tensor var_2621_cast_fp16 = mul(x = k_59_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2621_cast_fp16")]; tensor k_full_29_cast_fp16 = add(x = var_2620_cast_fp16, y = var_2621_cast_fp16)[name = tensor("k_full_29_cast_fp16")]; tensor var_2624_cast_fp16 = mul(x = v_cache_29_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2624_cast_fp16")]; tensor v_29_cast_fp16 = transpose(perm = v_29_perm_0, x = var_2599_cast_fp16)[name = tensor("transpose_53")]; tensor var_2625_cast_fp16 = mul(x = v_29_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2625_cast_fp16")]; tensor v_full_29_cast_fp16 = add(x = var_2624_cast_fp16, y = var_2625_cast_fp16)[name = tensor("v_full_29_cast_fp16")]; tensor var_2627_axes_0 = const()[name = tensor("op_2627_axes_0"), val = tensor([2])]; tensor var_2627_cast_fp16 = expand_dims(axes = var_2627_axes_0, x = k_full_29_cast_fp16)[name = tensor("op_2627_cast_fp16")]; tensor var_2629_reps_0 = const()[name = tensor("op_2629_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2629_cast_fp16 = tile(reps = var_2629_reps_0, x = var_2627_cast_fp16)[name = tensor("op_2629_cast_fp16")]; tensor var_2630 = const()[name = tensor("op_2630"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_29_cast_fp16 = reshape(shape = var_2630, x = var_2629_cast_fp16)[name = tensor("k_rep_29_cast_fp16")]; tensor var_2632_axes_0 = const()[name = tensor("op_2632_axes_0"), val = tensor([2])]; tensor var_2632_cast_fp16 = expand_dims(axes = var_2632_axes_0, x = v_full_29_cast_fp16)[name = tensor("op_2632_cast_fp16")]; tensor var_2634_reps_0 = const()[name = tensor("op_2634_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2634_cast_fp16 = tile(reps = var_2634_reps_0, x = var_2632_cast_fp16)[name = tensor("op_2634_cast_fp16")]; tensor var_2635 = const()[name = tensor("op_2635"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_29_cast_fp16 = reshape(shape = var_2635, x = var_2634_cast_fp16)[name = tensor("v_rep_29_cast_fp16")]; tensor var_2638_transpose_x_1 = const()[name = tensor("op_2638_transpose_x_1"), val = tensor(false)]; tensor var_2638_transpose_y_1 = const()[name = tensor("op_2638_transpose_y_1"), val = tensor(true)]; tensor var_2638_cast_fp16 = matmul(transpose_x = var_2638_transpose_x_1, transpose_y = var_2638_transpose_y_1, x = q_59_cast_fp16, y = k_rep_29_cast_fp16)[name = tensor("op_2638_cast_fp16")]; tensor var_2639_to_fp16 = const()[name = tensor("op_2639_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_57_cast_fp16 = mul(x = var_2638_cast_fp16, y = var_2639_to_fp16)[name = tensor("attn_57_cast_fp16")]; tensor input_59_cast_fp16 = add(x = attn_57_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_59_cast_fp16")]; tensor input_59_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_59_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_59_cast_fp16_to_fp32 = cast(dtype = input_59_cast_fp16_to_fp32_dtype_0, x = input_59_cast_fp16)[name = tensor("cast_422")]; tensor attn_59 = softmax(axis = var_2537, x = input_59_cast_fp16_to_fp32)[name = tensor("attn_59")]; tensor out_29_transpose_x_0 = const()[name = tensor("out_29_transpose_x_0"), val = tensor(false)]; tensor out_29_transpose_y_0 = const()[name = tensor("out_29_transpose_y_0"), val = tensor(false)]; tensor attn_59_to_fp16_dtype_0 = const()[name = tensor("attn_59_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_59_to_fp16 = cast(dtype = attn_59_to_fp16_dtype_0, x = attn_59)[name = tensor("cast_421")]; tensor out_29_cast_fp16 = matmul(transpose_x = out_29_transpose_x_0, transpose_y = out_29_transpose_y_0, x = attn_59_to_fp16, y = v_rep_29_cast_fp16)[name = tensor("out_29_cast_fp16")]; tensor var_2644_perm_0 = const()[name = tensor("op_2644_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2646 = const()[name = tensor("op_2646"), val = tensor([1, 1, 1536])]; tensor var_2644_cast_fp16 = transpose(perm = var_2644_perm_0, x = out_29_cast_fp16)[name = tensor("transpose_52")]; tensor x_587_cast_fp16 = reshape(shape = var_2646, x = var_2644_cast_fp16)[name = tensor("x_587_cast_fp16")]; tensor layers_14_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_14_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(349908032)))]; tensor linear_101_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_14_self_attn_o_proj_weight_to_fp16, x = x_587_cast_fp16)[name = tensor("linear_101_cast_fp16")]; tensor x_589_cast_fp16 = add(x = x_561_cast_fp16, y = linear_101_cast_fp16)[name = tensor("x_589_cast_fp16")]; tensor x_589_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_589_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2536_promoted_3 = const()[name = tensor("op_2536_promoted_3"), val = tensor(0x1p+1)]; tensor x_589_cast_fp16_to_fp32 = cast(dtype = x_589_cast_fp16_to_fp32_dtype_0, x = x_589_cast_fp16)[name = tensor("cast_420")]; tensor var_2657 = pow(x = x_589_cast_fp16_to_fp32, y = var_2536_promoted_3)[name = tensor("op_2657")]; tensor var_119_axes_0 = const()[name = tensor("var_119_axes_0"), val = tensor([-1])]; tensor var_119_keep_dims_0 = const()[name = tensor("var_119_keep_dims_0"), val = tensor(true)]; tensor var_119 = reduce_mean(axes = var_119_axes_0, keep_dims = var_119_keep_dims_0, x = var_2657)[name = tensor("var_119")]; tensor var_119_to_fp16_dtype_0 = const()[name = tensor("var_119_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2661_to_fp16 = const()[name = tensor("op_2661_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_119_to_fp16 = cast(dtype = var_119_to_fp16_dtype_0, x = var_119)[name = tensor("cast_419")]; tensor var_2662_cast_fp16 = add(x = var_119_to_fp16, y = var_2661_to_fp16)[name = tensor("op_2662_cast_fp16")]; tensor var_2662_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2662_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2663_epsilon_0 = const()[name = tensor("op_2663_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2662_cast_fp16_to_fp32 = cast(dtype = var_2662_cast_fp16_to_fp32_dtype_0, x = var_2662_cast_fp16)[name = tensor("cast_418")]; tensor var_2663 = rsqrt(epsilon = var_2663_epsilon_0, x = var_2662_cast_fp16_to_fp32)[name = tensor("op_2663")]; tensor var_2663_to_fp16_dtype_0 = const()[name = tensor("op_2663_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2663_to_fp16 = cast(dtype = var_2663_to_fp16_dtype_0, x = var_2663)[name = tensor("cast_417")]; tensor x_595_cast_fp16 = mul(x = x_589_cast_fp16, y = var_2663_to_fp16)[name = tensor("x_595_cast_fp16")]; tensor layers_14_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_14_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351480960)))]; tensor x_597_cast_fp16 = mul(x = layers_14_post_attention_layernorm_weight_to_fp16, y = x_595_cast_fp16)[name = tensor("x_597_cast_fp16")]; tensor layers_14_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_14_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351482048)))]; tensor linear_102_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_14_mlp_gate_proj_weight_to_fp16, x = x_597_cast_fp16)[name = tensor("linear_102_cast_fp16")]; tensor var_2674_cast_fp16 = silu(x = linear_102_cast_fp16)[name = tensor("op_2674_cast_fp16")]; tensor layers_14_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_14_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(353054976)))]; tensor linear_103_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_14_mlp_up_proj_weight_to_fp16, x = x_597_cast_fp16)[name = tensor("linear_103_cast_fp16")]; tensor x_599_cast_fp16 = mul(x = var_2674_cast_fp16, y = linear_103_cast_fp16)[name = tensor("x_599_cast_fp16")]; tensor layers_14_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_14_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(354627904)))]; tensor linear_104_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_14_mlp_down_proj_weight_to_fp16, x = x_599_cast_fp16)[name = tensor("linear_104_cast_fp16")]; tensor x_601_cast_fp16 = add(x = x_589_cast_fp16, y = linear_104_cast_fp16)[name = tensor("x_601_cast_fp16")]; tensor x_601_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_601_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_31_begin_0 = const()[name = tensor("k_cache_31_begin_0"), val = tensor([15, 0, 0, 0, 0])]; tensor k_cache_31_end_0 = const()[name = tensor("k_cache_31_end_0"), val = tensor([16, 1, 4, 2048, 128])]; tensor k_cache_31_end_mask_0 = const()[name = tensor("k_cache_31_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_31_squeeze_mask_0 = const()[name = tensor("k_cache_31_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_31_cast_fp16 = slice_by_index(begin = k_cache_31_begin_0, end = k_cache_31_end_0, end_mask = k_cache_31_end_mask_0, squeeze_mask = k_cache_31_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_31_cast_fp16")]; tensor v_cache_31_begin_0 = const()[name = tensor("v_cache_31_begin_0"), val = tensor([15, 0, 0, 0, 0])]; tensor v_cache_31_end_0 = const()[name = tensor("v_cache_31_end_0"), val = tensor([16, 1, 4, 2048, 128])]; tensor v_cache_31_end_mask_0 = const()[name = tensor("v_cache_31_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_31_squeeze_mask_0 = const()[name = tensor("v_cache_31_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_31_cast_fp16 = slice_by_index(begin = v_cache_31_begin_0, end = v_cache_31_end_0, end_mask = v_cache_31_end_mask_0, squeeze_mask = v_cache_31_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_31_cast_fp16")]; tensor var_2706 = const()[name = tensor("op_2706"), val = tensor(-1)]; tensor var_2705_promoted = const()[name = tensor("op_2705_promoted"), val = tensor(0x1p+1)]; tensor x_601_cast_fp16_to_fp32 = cast(dtype = x_601_cast_fp16_to_fp32_dtype_0, x = x_601_cast_fp16)[name = tensor("cast_416")]; tensor var_2715 = pow(x = x_601_cast_fp16_to_fp32, y = var_2705_promoted)[name = tensor("op_2715")]; tensor var_121_axes_0 = const()[name = tensor("var_121_axes_0"), val = tensor([-1])]; tensor var_121_keep_dims_0 = const()[name = tensor("var_121_keep_dims_0"), val = tensor(true)]; tensor var_121 = reduce_mean(axes = var_121_axes_0, keep_dims = var_121_keep_dims_0, x = var_2715)[name = tensor("var_121")]; tensor var_121_to_fp16_dtype_0 = const()[name = tensor("var_121_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2719_to_fp16 = const()[name = tensor("op_2719_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_121_to_fp16 = cast(dtype = var_121_to_fp16_dtype_0, x = var_121)[name = tensor("cast_415")]; tensor var_2720_cast_fp16 = add(x = var_121_to_fp16, y = var_2719_to_fp16)[name = tensor("op_2720_cast_fp16")]; tensor var_2720_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2720_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2721_epsilon_0 = const()[name = tensor("op_2721_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2720_cast_fp16_to_fp32 = cast(dtype = var_2720_cast_fp16_to_fp32_dtype_0, x = var_2720_cast_fp16)[name = tensor("cast_414")]; tensor var_2721 = rsqrt(epsilon = var_2721_epsilon_0, x = var_2720_cast_fp16_to_fp32)[name = tensor("op_2721")]; tensor var_2721_to_fp16_dtype_0 = const()[name = tensor("op_2721_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2721_to_fp16 = cast(dtype = var_2721_to_fp16_dtype_0, x = var_2721)[name = tensor("cast_413")]; tensor x_607_cast_fp16 = mul(x = x_601_cast_fp16, y = var_2721_to_fp16)[name = tensor("x_607_cast_fp16")]; tensor layers_15_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_15_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356200832)))]; tensor x_609_cast_fp16 = mul(x = layers_15_input_layernorm_weight_to_fp16, y = x_607_cast_fp16)[name = tensor("x_609_cast_fp16")]; tensor layers_15_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_15_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356201920)))]; tensor linear_105_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_15_self_attn_q_proj_weight_to_fp16, x = x_609_cast_fp16)[name = tensor("linear_105_cast_fp16")]; tensor var_2735 = const()[name = tensor("op_2735"), val = tensor([1, 1, 12, 128])]; tensor x_611_cast_fp16 = reshape(shape = var_2735, x = linear_105_cast_fp16)[name = tensor("x_611_cast_fp16")]; tensor x_611_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_611_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2705_promoted_1 = const()[name = tensor("op_2705_promoted_1"), val = tensor(0x1p+1)]; tensor x_611_cast_fp16_to_fp32 = cast(dtype = x_611_cast_fp16_to_fp32_dtype_0, x = x_611_cast_fp16)[name = tensor("cast_412")]; tensor var_2739 = pow(x = x_611_cast_fp16_to_fp32, y = var_2705_promoted_1)[name = tensor("op_2739")]; tensor var_123_axes_0 = const()[name = tensor("var_123_axes_0"), val = tensor([-1])]; tensor var_123_keep_dims_0 = const()[name = tensor("var_123_keep_dims_0"), val = tensor(true)]; tensor var_123 = reduce_mean(axes = var_123_axes_0, keep_dims = var_123_keep_dims_0, x = var_2739)[name = tensor("var_123")]; tensor var_123_to_fp16_dtype_0 = const()[name = tensor("var_123_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2743_to_fp16 = const()[name = tensor("op_2743_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_123_to_fp16 = cast(dtype = var_123_to_fp16_dtype_0, x = var_123)[name = tensor("cast_411")]; tensor var_2744_cast_fp16 = add(x = var_123_to_fp16, y = var_2743_to_fp16)[name = tensor("op_2744_cast_fp16")]; tensor var_2744_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2744_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2745_epsilon_0 = const()[name = tensor("op_2745_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2744_cast_fp16_to_fp32 = cast(dtype = var_2744_cast_fp16_to_fp32_dtype_0, x = var_2744_cast_fp16)[name = tensor("cast_410")]; tensor var_2745 = rsqrt(epsilon = var_2745_epsilon_0, x = var_2744_cast_fp16_to_fp32)[name = tensor("op_2745")]; tensor var_2745_to_fp16_dtype_0 = const()[name = tensor("op_2745_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2745_to_fp16 = cast(dtype = var_2745_to_fp16_dtype_0, x = var_2745)[name = tensor("cast_409")]; tensor x_617_cast_fp16 = mul(x = x_611_cast_fp16, y = var_2745_to_fp16)[name = tensor("x_617_cast_fp16")]; tensor layers_15_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_15_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(357774848)))]; tensor var_2747_cast_fp16 = mul(x = layers_15_self_attn_q_norm_weight_to_fp16, y = x_617_cast_fp16)[name = tensor("op_2747_cast_fp16")]; tensor q_61_perm_0 = const()[name = tensor("q_61_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_15_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_15_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(357775168)))]; tensor linear_106_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_15_self_attn_k_proj_weight_to_fp16, x = x_609_cast_fp16)[name = tensor("linear_106_cast_fp16")]; tensor var_2751 = const()[name = tensor("op_2751"), val = tensor([1, 1, 4, 128])]; tensor x_619_cast_fp16 = reshape(shape = var_2751, x = linear_106_cast_fp16)[name = tensor("x_619_cast_fp16")]; tensor x_619_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_619_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2705_promoted_2 = const()[name = tensor("op_2705_promoted_2"), val = tensor(0x1p+1)]; tensor x_619_cast_fp16_to_fp32 = cast(dtype = x_619_cast_fp16_to_fp32_dtype_0, x = x_619_cast_fp16)[name = tensor("cast_408")]; tensor var_2755 = pow(x = x_619_cast_fp16_to_fp32, y = var_2705_promoted_2)[name = tensor("op_2755")]; tensor var_125_axes_0 = const()[name = tensor("var_125_axes_0"), val = tensor([-1])]; tensor var_125_keep_dims_0 = const()[name = tensor("var_125_keep_dims_0"), val = tensor(true)]; tensor var_125 = reduce_mean(axes = var_125_axes_0, keep_dims = var_125_keep_dims_0, x = var_2755)[name = tensor("var_125")]; tensor var_125_to_fp16_dtype_0 = const()[name = tensor("var_125_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2759_to_fp16 = const()[name = tensor("op_2759_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_125_to_fp16 = cast(dtype = var_125_to_fp16_dtype_0, x = var_125)[name = tensor("cast_407")]; tensor var_2760_cast_fp16 = add(x = var_125_to_fp16, y = var_2759_to_fp16)[name = tensor("op_2760_cast_fp16")]; tensor var_2760_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2760_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2761_epsilon_0 = const()[name = tensor("op_2761_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2760_cast_fp16_to_fp32 = cast(dtype = var_2760_cast_fp16_to_fp32_dtype_0, x = var_2760_cast_fp16)[name = tensor("cast_406")]; tensor var_2761 = rsqrt(epsilon = var_2761_epsilon_0, x = var_2760_cast_fp16_to_fp32)[name = tensor("op_2761")]; tensor var_2761_to_fp16_dtype_0 = const()[name = tensor("op_2761_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2761_to_fp16 = cast(dtype = var_2761_to_fp16_dtype_0, x = var_2761)[name = tensor("cast_405")]; tensor x_625_cast_fp16 = mul(x = x_619_cast_fp16, y = var_2761_to_fp16)[name = tensor("x_625_cast_fp16")]; tensor layers_15_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_15_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(358299520)))]; tensor var_2763_cast_fp16 = mul(x = layers_15_self_attn_k_norm_weight_to_fp16, y = x_625_cast_fp16)[name = tensor("op_2763_cast_fp16")]; tensor k_61_perm_0 = const()[name = tensor("k_61_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_15_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_15_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(358299840)))]; tensor linear_107_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_15_self_attn_v_proj_weight_to_fp16, x = x_609_cast_fp16)[name = tensor("linear_107_cast_fp16")]; tensor var_2767 = const()[name = tensor("op_2767"), val = tensor([1, 1, 4, 128])]; tensor var_2768_cast_fp16 = reshape(shape = var_2767, x = linear_107_cast_fp16)[name = tensor("op_2768_cast_fp16")]; tensor v_31_perm_0 = const()[name = tensor("v_31_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_61_cast_fp16 = transpose(perm = q_61_perm_0, x = var_2747_cast_fp16)[name = tensor("transpose_51")]; tensor var_2772_cast_fp16 = mul(x = q_61_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2772_cast_fp16")]; tensor x1_61_begin_0 = const()[name = tensor("x1_61_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_61_end_0 = const()[name = tensor("x1_61_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_61_end_mask_0 = const()[name = tensor("x1_61_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_61_cast_fp16 = slice_by_index(begin = x1_61_begin_0, end = x1_61_end_0, end_mask = x1_61_end_mask_0, x = q_61_cast_fp16)[name = tensor("x1_61_cast_fp16")]; tensor x2_61_begin_0 = const()[name = tensor("x2_61_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_61_end_0 = const()[name = tensor("x2_61_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_61_end_mask_0 = const()[name = tensor("x2_61_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_61_cast_fp16 = slice_by_index(begin = x2_61_begin_0, end = x2_61_end_0, end_mask = x2_61_end_mask_0, x = q_61_cast_fp16)[name = tensor("x2_61_cast_fp16")]; tensor const_31_promoted_to_fp16 = const()[name = tensor("const_31_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2775_cast_fp16 = mul(x = x2_61_cast_fp16, y = const_31_promoted_to_fp16)[name = tensor("op_2775_cast_fp16")]; tensor var_2777_interleave_0 = const()[name = tensor("op_2777_interleave_0"), val = tensor(false)]; tensor var_2777_cast_fp16 = concat(axis = var_2706, interleave = var_2777_interleave_0, values = (var_2775_cast_fp16, x1_61_cast_fp16))[name = tensor("op_2777_cast_fp16")]; tensor var_2778_cast_fp16 = mul(x = var_2777_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2778_cast_fp16")]; tensor q_63_cast_fp16 = add(x = var_2772_cast_fp16, y = var_2778_cast_fp16)[name = tensor("q_63_cast_fp16")]; tensor k_61_cast_fp16 = transpose(perm = k_61_perm_0, x = var_2763_cast_fp16)[name = tensor("transpose_50")]; tensor var_2780_cast_fp16 = mul(x = k_61_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2780_cast_fp16")]; tensor x1_63_begin_0 = const()[name = tensor("x1_63_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_63_end_0 = const()[name = tensor("x1_63_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_63_end_mask_0 = const()[name = tensor("x1_63_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_63_cast_fp16 = slice_by_index(begin = x1_63_begin_0, end = x1_63_end_0, end_mask = x1_63_end_mask_0, x = k_61_cast_fp16)[name = tensor("x1_63_cast_fp16")]; tensor x2_63_begin_0 = const()[name = tensor("x2_63_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_63_end_0 = const()[name = tensor("x2_63_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_63_end_mask_0 = const()[name = tensor("x2_63_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_63_cast_fp16 = slice_by_index(begin = x2_63_begin_0, end = x2_63_end_0, end_mask = x2_63_end_mask_0, x = k_61_cast_fp16)[name = tensor("x2_63_cast_fp16")]; tensor const_32_promoted_to_fp16 = const()[name = tensor("const_32_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2783_cast_fp16 = mul(x = x2_63_cast_fp16, y = const_32_promoted_to_fp16)[name = tensor("op_2783_cast_fp16")]; tensor var_2785_interleave_0 = const()[name = tensor("op_2785_interleave_0"), val = tensor(false)]; tensor var_2785_cast_fp16 = concat(axis = var_2706, interleave = var_2785_interleave_0, values = (var_2783_cast_fp16, x1_63_cast_fp16))[name = tensor("op_2785_cast_fp16")]; tensor var_2786_cast_fp16 = mul(x = var_2785_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2786_cast_fp16")]; tensor k_63_cast_fp16 = add(x = var_2780_cast_fp16, y = var_2786_cast_fp16)[name = tensor("k_63_cast_fp16")]; tensor var_2789_cast_fp16 = mul(x = k_cache_31_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2789_cast_fp16")]; tensor var_2790_cast_fp16 = mul(x = k_63_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2790_cast_fp16")]; tensor k_full_31_cast_fp16 = add(x = var_2789_cast_fp16, y = var_2790_cast_fp16)[name = tensor("k_full_31_cast_fp16")]; tensor var_2793_cast_fp16 = mul(x = v_cache_31_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2793_cast_fp16")]; tensor v_31_cast_fp16 = transpose(perm = v_31_perm_0, x = var_2768_cast_fp16)[name = tensor("transpose_49")]; tensor var_2794_cast_fp16 = mul(x = v_31_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2794_cast_fp16")]; tensor v_full_31_cast_fp16 = add(x = var_2793_cast_fp16, y = var_2794_cast_fp16)[name = tensor("v_full_31_cast_fp16")]; tensor var_2796_axes_0 = const()[name = tensor("op_2796_axes_0"), val = tensor([2])]; tensor var_2796_cast_fp16 = expand_dims(axes = var_2796_axes_0, x = k_full_31_cast_fp16)[name = tensor("op_2796_cast_fp16")]; tensor var_2798_reps_0 = const()[name = tensor("op_2798_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2798_cast_fp16 = tile(reps = var_2798_reps_0, x = var_2796_cast_fp16)[name = tensor("op_2798_cast_fp16")]; tensor var_2799 = const()[name = tensor("op_2799"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_31_cast_fp16 = reshape(shape = var_2799, x = var_2798_cast_fp16)[name = tensor("k_rep_31_cast_fp16")]; tensor var_2801_axes_0 = const()[name = tensor("op_2801_axes_0"), val = tensor([2])]; tensor var_2801_cast_fp16 = expand_dims(axes = var_2801_axes_0, x = v_full_31_cast_fp16)[name = tensor("op_2801_cast_fp16")]; tensor var_2803_reps_0 = const()[name = tensor("op_2803_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2803_cast_fp16 = tile(reps = var_2803_reps_0, x = var_2801_cast_fp16)[name = tensor("op_2803_cast_fp16")]; tensor var_2804 = const()[name = tensor("op_2804"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_31_cast_fp16 = reshape(shape = var_2804, x = var_2803_cast_fp16)[name = tensor("v_rep_31_cast_fp16")]; tensor var_2807_transpose_x_1 = const()[name = tensor("op_2807_transpose_x_1"), val = tensor(false)]; tensor var_2807_transpose_y_1 = const()[name = tensor("op_2807_transpose_y_1"), val = tensor(true)]; tensor var_2807_cast_fp16 = matmul(transpose_x = var_2807_transpose_x_1, transpose_y = var_2807_transpose_y_1, x = q_63_cast_fp16, y = k_rep_31_cast_fp16)[name = tensor("op_2807_cast_fp16")]; tensor var_2808_to_fp16 = const()[name = tensor("op_2808_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_61_cast_fp16 = mul(x = var_2807_cast_fp16, y = var_2808_to_fp16)[name = tensor("attn_61_cast_fp16")]; tensor input_63_cast_fp16 = add(x = attn_61_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_63_cast_fp16")]; tensor input_63_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_63_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_63_cast_fp16_to_fp32 = cast(dtype = input_63_cast_fp16_to_fp32_dtype_0, x = input_63_cast_fp16)[name = tensor("cast_404")]; tensor attn_63 = softmax(axis = var_2706, x = input_63_cast_fp16_to_fp32)[name = tensor("attn_63")]; tensor out_31_transpose_x_0 = const()[name = tensor("out_31_transpose_x_0"), val = tensor(false)]; tensor out_31_transpose_y_0 = const()[name = tensor("out_31_transpose_y_0"), val = tensor(false)]; tensor attn_63_to_fp16_dtype_0 = const()[name = tensor("attn_63_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_63_to_fp16 = cast(dtype = attn_63_to_fp16_dtype_0, x = attn_63)[name = tensor("cast_403")]; tensor out_31_cast_fp16 = matmul(transpose_x = out_31_transpose_x_0, transpose_y = out_31_transpose_y_0, x = attn_63_to_fp16, y = v_rep_31_cast_fp16)[name = tensor("out_31_cast_fp16")]; tensor var_2813_perm_0 = const()[name = tensor("op_2813_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2815 = const()[name = tensor("op_2815"), val = tensor([1, 1, 1536])]; tensor var_2813_cast_fp16 = transpose(perm = var_2813_perm_0, x = out_31_cast_fp16)[name = tensor("transpose_48")]; tensor x_627_cast_fp16 = reshape(shape = var_2815, x = var_2813_cast_fp16)[name = tensor("x_627_cast_fp16")]; tensor layers_15_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_15_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(358824192)))]; tensor linear_108_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_15_self_attn_o_proj_weight_to_fp16, x = x_627_cast_fp16)[name = tensor("linear_108_cast_fp16")]; tensor x_629_cast_fp16 = add(x = x_601_cast_fp16, y = linear_108_cast_fp16)[name = tensor("x_629_cast_fp16")]; tensor x_629_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_629_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2705_promoted_3 = const()[name = tensor("op_2705_promoted_3"), val = tensor(0x1p+1)]; tensor x_629_cast_fp16_to_fp32 = cast(dtype = x_629_cast_fp16_to_fp32_dtype_0, x = x_629_cast_fp16)[name = tensor("cast_402")]; tensor var_2826 = pow(x = x_629_cast_fp16_to_fp32, y = var_2705_promoted_3)[name = tensor("op_2826")]; tensor var_127_axes_0 = const()[name = tensor("var_127_axes_0"), val = tensor([-1])]; tensor var_127_keep_dims_0 = const()[name = tensor("var_127_keep_dims_0"), val = tensor(true)]; tensor var_127 = reduce_mean(axes = var_127_axes_0, keep_dims = var_127_keep_dims_0, x = var_2826)[name = tensor("var_127")]; tensor var_127_to_fp16_dtype_0 = const()[name = tensor("var_127_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2830_to_fp16 = const()[name = tensor("op_2830_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_127_to_fp16 = cast(dtype = var_127_to_fp16_dtype_0, x = var_127)[name = tensor("cast_401")]; tensor var_2831_cast_fp16 = add(x = var_127_to_fp16, y = var_2830_to_fp16)[name = tensor("op_2831_cast_fp16")]; tensor var_2831_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2831_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2832_epsilon_0 = const()[name = tensor("op_2832_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2831_cast_fp16_to_fp32 = cast(dtype = var_2831_cast_fp16_to_fp32_dtype_0, x = var_2831_cast_fp16)[name = tensor("cast_400")]; tensor var_2832 = rsqrt(epsilon = var_2832_epsilon_0, x = var_2831_cast_fp16_to_fp32)[name = tensor("op_2832")]; tensor var_2832_to_fp16_dtype_0 = const()[name = tensor("op_2832_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2832_to_fp16 = cast(dtype = var_2832_to_fp16_dtype_0, x = var_2832)[name = tensor("cast_399")]; tensor x_635_cast_fp16 = mul(x = x_629_cast_fp16, y = var_2832_to_fp16)[name = tensor("x_635_cast_fp16")]; tensor layers_15_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_15_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(360397120)))]; tensor x_637_cast_fp16 = mul(x = layers_15_post_attention_layernorm_weight_to_fp16, y = x_635_cast_fp16)[name = tensor("x_637_cast_fp16")]; tensor layers_15_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_15_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(360398208)))]; tensor linear_109_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_15_mlp_gate_proj_weight_to_fp16, x = x_637_cast_fp16)[name = tensor("linear_109_cast_fp16")]; tensor var_2843_cast_fp16 = silu(x = linear_109_cast_fp16)[name = tensor("op_2843_cast_fp16")]; tensor layers_15_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_15_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(361971136)))]; tensor linear_110_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_15_mlp_up_proj_weight_to_fp16, x = x_637_cast_fp16)[name = tensor("linear_110_cast_fp16")]; tensor x_639_cast_fp16 = mul(x = var_2843_cast_fp16, y = linear_110_cast_fp16)[name = tensor("x_639_cast_fp16")]; tensor layers_15_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_15_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(363544064)))]; tensor linear_111_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_15_mlp_down_proj_weight_to_fp16, x = x_639_cast_fp16)[name = tensor("linear_111_cast_fp16")]; tensor x_641_cast_fp16 = add(x = x_629_cast_fp16, y = linear_111_cast_fp16)[name = tensor("x_641_cast_fp16")]; tensor x_641_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_641_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_33_begin_0 = const()[name = tensor("k_cache_33_begin_0"), val = tensor([16, 0, 0, 0, 0])]; tensor k_cache_33_end_0 = const()[name = tensor("k_cache_33_end_0"), val = tensor([17, 1, 4, 2048, 128])]; tensor k_cache_33_end_mask_0 = const()[name = tensor("k_cache_33_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_33_squeeze_mask_0 = const()[name = tensor("k_cache_33_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_33_cast_fp16 = slice_by_index(begin = k_cache_33_begin_0, end = k_cache_33_end_0, end_mask = k_cache_33_end_mask_0, squeeze_mask = k_cache_33_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_33_cast_fp16")]; tensor v_cache_33_begin_0 = const()[name = tensor("v_cache_33_begin_0"), val = tensor([16, 0, 0, 0, 0])]; tensor v_cache_33_end_0 = const()[name = tensor("v_cache_33_end_0"), val = tensor([17, 1, 4, 2048, 128])]; tensor v_cache_33_end_mask_0 = const()[name = tensor("v_cache_33_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_33_squeeze_mask_0 = const()[name = tensor("v_cache_33_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_33_cast_fp16 = slice_by_index(begin = v_cache_33_begin_0, end = v_cache_33_end_0, end_mask = v_cache_33_end_mask_0, squeeze_mask = v_cache_33_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_33_cast_fp16")]; tensor var_2875 = const()[name = tensor("op_2875"), val = tensor(-1)]; tensor var_2874_promoted = const()[name = tensor("op_2874_promoted"), val = tensor(0x1p+1)]; tensor x_641_cast_fp16_to_fp32 = cast(dtype = x_641_cast_fp16_to_fp32_dtype_0, x = x_641_cast_fp16)[name = tensor("cast_398")]; tensor var_2884 = pow(x = x_641_cast_fp16_to_fp32, y = var_2874_promoted)[name = tensor("op_2884")]; tensor var_129_axes_0 = const()[name = tensor("var_129_axes_0"), val = tensor([-1])]; tensor var_129_keep_dims_0 = const()[name = tensor("var_129_keep_dims_0"), val = tensor(true)]; tensor var_129 = reduce_mean(axes = var_129_axes_0, keep_dims = var_129_keep_dims_0, x = var_2884)[name = tensor("var_129")]; tensor var_129_to_fp16_dtype_0 = const()[name = tensor("var_129_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2888_to_fp16 = const()[name = tensor("op_2888_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_129_to_fp16 = cast(dtype = var_129_to_fp16_dtype_0, x = var_129)[name = tensor("cast_397")]; tensor var_2889_cast_fp16 = add(x = var_129_to_fp16, y = var_2888_to_fp16)[name = tensor("op_2889_cast_fp16")]; tensor var_2889_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2889_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2890_epsilon_0 = const()[name = tensor("op_2890_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2889_cast_fp16_to_fp32 = cast(dtype = var_2889_cast_fp16_to_fp32_dtype_0, x = var_2889_cast_fp16)[name = tensor("cast_396")]; tensor var_2890 = rsqrt(epsilon = var_2890_epsilon_0, x = var_2889_cast_fp16_to_fp32)[name = tensor("op_2890")]; tensor var_2890_to_fp16_dtype_0 = const()[name = tensor("op_2890_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2890_to_fp16 = cast(dtype = var_2890_to_fp16_dtype_0, x = var_2890)[name = tensor("cast_395")]; tensor x_647_cast_fp16 = mul(x = x_641_cast_fp16, y = var_2890_to_fp16)[name = tensor("x_647_cast_fp16")]; tensor layers_16_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_16_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(365116992)))]; tensor x_649_cast_fp16 = mul(x = layers_16_input_layernorm_weight_to_fp16, y = x_647_cast_fp16)[name = tensor("x_649_cast_fp16")]; tensor layers_16_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_16_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(365118080)))]; tensor linear_112_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_16_self_attn_q_proj_weight_to_fp16, x = x_649_cast_fp16)[name = tensor("linear_112_cast_fp16")]; tensor var_2904 = const()[name = tensor("op_2904"), val = tensor([1, 1, 12, 128])]; tensor x_651_cast_fp16 = reshape(shape = var_2904, x = linear_112_cast_fp16)[name = tensor("x_651_cast_fp16")]; tensor x_651_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_651_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2874_promoted_1 = const()[name = tensor("op_2874_promoted_1"), val = tensor(0x1p+1)]; tensor x_651_cast_fp16_to_fp32 = cast(dtype = x_651_cast_fp16_to_fp32_dtype_0, x = x_651_cast_fp16)[name = tensor("cast_394")]; tensor var_2908 = pow(x = x_651_cast_fp16_to_fp32, y = var_2874_promoted_1)[name = tensor("op_2908")]; tensor var_131_axes_0 = const()[name = tensor("var_131_axes_0"), val = tensor([-1])]; tensor var_131_keep_dims_0 = const()[name = tensor("var_131_keep_dims_0"), val = tensor(true)]; tensor var_131 = reduce_mean(axes = var_131_axes_0, keep_dims = var_131_keep_dims_0, x = var_2908)[name = tensor("var_131")]; tensor var_131_to_fp16_dtype_0 = const()[name = tensor("var_131_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2912_to_fp16 = const()[name = tensor("op_2912_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_131_to_fp16 = cast(dtype = var_131_to_fp16_dtype_0, x = var_131)[name = tensor("cast_393")]; tensor var_2913_cast_fp16 = add(x = var_131_to_fp16, y = var_2912_to_fp16)[name = tensor("op_2913_cast_fp16")]; tensor var_2913_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2913_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2914_epsilon_0 = const()[name = tensor("op_2914_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2913_cast_fp16_to_fp32 = cast(dtype = var_2913_cast_fp16_to_fp32_dtype_0, x = var_2913_cast_fp16)[name = tensor("cast_392")]; tensor var_2914 = rsqrt(epsilon = var_2914_epsilon_0, x = var_2913_cast_fp16_to_fp32)[name = tensor("op_2914")]; tensor var_2914_to_fp16_dtype_0 = const()[name = tensor("op_2914_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2914_to_fp16 = cast(dtype = var_2914_to_fp16_dtype_0, x = var_2914)[name = tensor("cast_391")]; tensor x_657_cast_fp16 = mul(x = x_651_cast_fp16, y = var_2914_to_fp16)[name = tensor("x_657_cast_fp16")]; tensor layers_16_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_16_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(366691008)))]; tensor var_2916_cast_fp16 = mul(x = layers_16_self_attn_q_norm_weight_to_fp16, y = x_657_cast_fp16)[name = tensor("op_2916_cast_fp16")]; tensor q_65_perm_0 = const()[name = tensor("q_65_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_16_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_16_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(366691328)))]; tensor linear_113_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_16_self_attn_k_proj_weight_to_fp16, x = x_649_cast_fp16)[name = tensor("linear_113_cast_fp16")]; tensor var_2920 = const()[name = tensor("op_2920"), val = tensor([1, 1, 4, 128])]; tensor x_659_cast_fp16 = reshape(shape = var_2920, x = linear_113_cast_fp16)[name = tensor("x_659_cast_fp16")]; tensor x_659_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_659_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2874_promoted_2 = const()[name = tensor("op_2874_promoted_2"), val = tensor(0x1p+1)]; tensor x_659_cast_fp16_to_fp32 = cast(dtype = x_659_cast_fp16_to_fp32_dtype_0, x = x_659_cast_fp16)[name = tensor("cast_390")]; tensor var_2924 = pow(x = x_659_cast_fp16_to_fp32, y = var_2874_promoted_2)[name = tensor("op_2924")]; tensor var_133_axes_0 = const()[name = tensor("var_133_axes_0"), val = tensor([-1])]; tensor var_133_keep_dims_0 = const()[name = tensor("var_133_keep_dims_0"), val = tensor(true)]; tensor var_133 = reduce_mean(axes = var_133_axes_0, keep_dims = var_133_keep_dims_0, x = var_2924)[name = tensor("var_133")]; tensor var_133_to_fp16_dtype_0 = const()[name = tensor("var_133_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2928_to_fp16 = const()[name = tensor("op_2928_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_133_to_fp16 = cast(dtype = var_133_to_fp16_dtype_0, x = var_133)[name = tensor("cast_389")]; tensor var_2929_cast_fp16 = add(x = var_133_to_fp16, y = var_2928_to_fp16)[name = tensor("op_2929_cast_fp16")]; tensor var_2929_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_2929_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2930_epsilon_0 = const()[name = tensor("op_2930_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_2929_cast_fp16_to_fp32 = cast(dtype = var_2929_cast_fp16_to_fp32_dtype_0, x = var_2929_cast_fp16)[name = tensor("cast_388")]; tensor var_2930 = rsqrt(epsilon = var_2930_epsilon_0, x = var_2929_cast_fp16_to_fp32)[name = tensor("op_2930")]; tensor var_2930_to_fp16_dtype_0 = const()[name = tensor("op_2930_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2930_to_fp16 = cast(dtype = var_2930_to_fp16_dtype_0, x = var_2930)[name = tensor("cast_387")]; tensor x_665_cast_fp16 = mul(x = x_659_cast_fp16, y = var_2930_to_fp16)[name = tensor("x_665_cast_fp16")]; tensor layers_16_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_16_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(367215680)))]; tensor var_2932_cast_fp16 = mul(x = layers_16_self_attn_k_norm_weight_to_fp16, y = x_665_cast_fp16)[name = tensor("op_2932_cast_fp16")]; tensor k_65_perm_0 = const()[name = tensor("k_65_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_16_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_16_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(367216000)))]; tensor linear_114_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_16_self_attn_v_proj_weight_to_fp16, x = x_649_cast_fp16)[name = tensor("linear_114_cast_fp16")]; tensor var_2936 = const()[name = tensor("op_2936"), val = tensor([1, 1, 4, 128])]; tensor var_2937_cast_fp16 = reshape(shape = var_2936, x = linear_114_cast_fp16)[name = tensor("op_2937_cast_fp16")]; tensor v_33_perm_0 = const()[name = tensor("v_33_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_65_cast_fp16 = transpose(perm = q_65_perm_0, x = var_2916_cast_fp16)[name = tensor("transpose_47")]; tensor var_2941_cast_fp16 = mul(x = q_65_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2941_cast_fp16")]; tensor x1_65_begin_0 = const()[name = tensor("x1_65_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_65_end_0 = const()[name = tensor("x1_65_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_65_end_mask_0 = const()[name = tensor("x1_65_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_65_cast_fp16 = slice_by_index(begin = x1_65_begin_0, end = x1_65_end_0, end_mask = x1_65_end_mask_0, x = q_65_cast_fp16)[name = tensor("x1_65_cast_fp16")]; tensor x2_65_begin_0 = const()[name = tensor("x2_65_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_65_end_0 = const()[name = tensor("x2_65_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_65_end_mask_0 = const()[name = tensor("x2_65_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_65_cast_fp16 = slice_by_index(begin = x2_65_begin_0, end = x2_65_end_0, end_mask = x2_65_end_mask_0, x = q_65_cast_fp16)[name = tensor("x2_65_cast_fp16")]; tensor const_33_promoted_to_fp16 = const()[name = tensor("const_33_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2944_cast_fp16 = mul(x = x2_65_cast_fp16, y = const_33_promoted_to_fp16)[name = tensor("op_2944_cast_fp16")]; tensor var_2946_interleave_0 = const()[name = tensor("op_2946_interleave_0"), val = tensor(false)]; tensor var_2946_cast_fp16 = concat(axis = var_2875, interleave = var_2946_interleave_0, values = (var_2944_cast_fp16, x1_65_cast_fp16))[name = tensor("op_2946_cast_fp16")]; tensor var_2947_cast_fp16 = mul(x = var_2946_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2947_cast_fp16")]; tensor q_67_cast_fp16 = add(x = var_2941_cast_fp16, y = var_2947_cast_fp16)[name = tensor("q_67_cast_fp16")]; tensor k_65_cast_fp16 = transpose(perm = k_65_perm_0, x = var_2932_cast_fp16)[name = tensor("transpose_46")]; tensor var_2949_cast_fp16 = mul(x = k_65_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_2949_cast_fp16")]; tensor x1_67_begin_0 = const()[name = tensor("x1_67_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_67_end_0 = const()[name = tensor("x1_67_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_67_end_mask_0 = const()[name = tensor("x1_67_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_67_cast_fp16 = slice_by_index(begin = x1_67_begin_0, end = x1_67_end_0, end_mask = x1_67_end_mask_0, x = k_65_cast_fp16)[name = tensor("x1_67_cast_fp16")]; tensor x2_67_begin_0 = const()[name = tensor("x2_67_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_67_end_0 = const()[name = tensor("x2_67_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_67_end_mask_0 = const()[name = tensor("x2_67_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_67_cast_fp16 = slice_by_index(begin = x2_67_begin_0, end = x2_67_end_0, end_mask = x2_67_end_mask_0, x = k_65_cast_fp16)[name = tensor("x2_67_cast_fp16")]; tensor const_34_promoted_to_fp16 = const()[name = tensor("const_34_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_2952_cast_fp16 = mul(x = x2_67_cast_fp16, y = const_34_promoted_to_fp16)[name = tensor("op_2952_cast_fp16")]; tensor var_2954_interleave_0 = const()[name = tensor("op_2954_interleave_0"), val = tensor(false)]; tensor var_2954_cast_fp16 = concat(axis = var_2875, interleave = var_2954_interleave_0, values = (var_2952_cast_fp16, x1_67_cast_fp16))[name = tensor("op_2954_cast_fp16")]; tensor var_2955_cast_fp16 = mul(x = var_2954_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_2955_cast_fp16")]; tensor k_67_cast_fp16 = add(x = var_2949_cast_fp16, y = var_2955_cast_fp16)[name = tensor("k_67_cast_fp16")]; tensor var_2958_cast_fp16 = mul(x = k_cache_33_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2958_cast_fp16")]; tensor var_2959_cast_fp16 = mul(x = k_67_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2959_cast_fp16")]; tensor k_full_33_cast_fp16 = add(x = var_2958_cast_fp16, y = var_2959_cast_fp16)[name = tensor("k_full_33_cast_fp16")]; tensor var_2962_cast_fp16 = mul(x = v_cache_33_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_2962_cast_fp16")]; tensor v_33_cast_fp16 = transpose(perm = v_33_perm_0, x = var_2937_cast_fp16)[name = tensor("transpose_45")]; tensor var_2963_cast_fp16 = mul(x = v_33_cast_fp16, y = var_107_to_fp16)[name = tensor("op_2963_cast_fp16")]; tensor v_full_33_cast_fp16 = add(x = var_2962_cast_fp16, y = var_2963_cast_fp16)[name = tensor("v_full_33_cast_fp16")]; tensor var_2965_axes_0 = const()[name = tensor("op_2965_axes_0"), val = tensor([2])]; tensor var_2965_cast_fp16 = expand_dims(axes = var_2965_axes_0, x = k_full_33_cast_fp16)[name = tensor("op_2965_cast_fp16")]; tensor var_2967_reps_0 = const()[name = tensor("op_2967_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2967_cast_fp16 = tile(reps = var_2967_reps_0, x = var_2965_cast_fp16)[name = tensor("op_2967_cast_fp16")]; tensor var_2968 = const()[name = tensor("op_2968"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_33_cast_fp16 = reshape(shape = var_2968, x = var_2967_cast_fp16)[name = tensor("k_rep_33_cast_fp16")]; tensor var_2970_axes_0 = const()[name = tensor("op_2970_axes_0"), val = tensor([2])]; tensor var_2970_cast_fp16 = expand_dims(axes = var_2970_axes_0, x = v_full_33_cast_fp16)[name = tensor("op_2970_cast_fp16")]; tensor var_2972_reps_0 = const()[name = tensor("op_2972_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_2972_cast_fp16 = tile(reps = var_2972_reps_0, x = var_2970_cast_fp16)[name = tensor("op_2972_cast_fp16")]; tensor var_2973 = const()[name = tensor("op_2973"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_33_cast_fp16 = reshape(shape = var_2973, x = var_2972_cast_fp16)[name = tensor("v_rep_33_cast_fp16")]; tensor var_2976_transpose_x_1 = const()[name = tensor("op_2976_transpose_x_1"), val = tensor(false)]; tensor var_2976_transpose_y_1 = const()[name = tensor("op_2976_transpose_y_1"), val = tensor(true)]; tensor var_2976_cast_fp16 = matmul(transpose_x = var_2976_transpose_x_1, transpose_y = var_2976_transpose_y_1, x = q_67_cast_fp16, y = k_rep_33_cast_fp16)[name = tensor("op_2976_cast_fp16")]; tensor var_2977_to_fp16 = const()[name = tensor("op_2977_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_65_cast_fp16 = mul(x = var_2976_cast_fp16, y = var_2977_to_fp16)[name = tensor("attn_65_cast_fp16")]; tensor input_67_cast_fp16 = add(x = attn_65_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_67_cast_fp16")]; tensor input_67_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_67_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_67_cast_fp16_to_fp32 = cast(dtype = input_67_cast_fp16_to_fp32_dtype_0, x = input_67_cast_fp16)[name = tensor("cast_386")]; tensor attn_67 = softmax(axis = var_2875, x = input_67_cast_fp16_to_fp32)[name = tensor("attn_67")]; tensor out_33_transpose_x_0 = const()[name = tensor("out_33_transpose_x_0"), val = tensor(false)]; tensor out_33_transpose_y_0 = const()[name = tensor("out_33_transpose_y_0"), val = tensor(false)]; tensor attn_67_to_fp16_dtype_0 = const()[name = tensor("attn_67_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_67_to_fp16 = cast(dtype = attn_67_to_fp16_dtype_0, x = attn_67)[name = tensor("cast_385")]; tensor out_33_cast_fp16 = matmul(transpose_x = out_33_transpose_x_0, transpose_y = out_33_transpose_y_0, x = attn_67_to_fp16, y = v_rep_33_cast_fp16)[name = tensor("out_33_cast_fp16")]; tensor var_2982_perm_0 = const()[name = tensor("op_2982_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2984 = const()[name = tensor("op_2984"), val = tensor([1, 1, 1536])]; tensor var_2982_cast_fp16 = transpose(perm = var_2982_perm_0, x = out_33_cast_fp16)[name = tensor("transpose_44")]; tensor x_667_cast_fp16 = reshape(shape = var_2984, x = var_2982_cast_fp16)[name = tensor("x_667_cast_fp16")]; tensor layers_16_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_16_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(367740352)))]; tensor linear_115_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_16_self_attn_o_proj_weight_to_fp16, x = x_667_cast_fp16)[name = tensor("linear_115_cast_fp16")]; tensor x_669_cast_fp16 = add(x = x_641_cast_fp16, y = linear_115_cast_fp16)[name = tensor("x_669_cast_fp16")]; tensor x_669_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_669_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_2874_promoted_3 = const()[name = tensor("op_2874_promoted_3"), val = tensor(0x1p+1)]; tensor x_669_cast_fp16_to_fp32 = cast(dtype = x_669_cast_fp16_to_fp32_dtype_0, x = x_669_cast_fp16)[name = tensor("cast_384")]; tensor var_2995 = pow(x = x_669_cast_fp16_to_fp32, y = var_2874_promoted_3)[name = tensor("op_2995")]; tensor var_135_axes_0 = const()[name = tensor("var_135_axes_0"), val = tensor([-1])]; tensor var_135_keep_dims_0 = const()[name = tensor("var_135_keep_dims_0"), val = tensor(true)]; tensor var_135 = reduce_mean(axes = var_135_axes_0, keep_dims = var_135_keep_dims_0, x = var_2995)[name = tensor("var_135")]; tensor var_135_to_fp16_dtype_0 = const()[name = tensor("var_135_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_2999_to_fp16 = const()[name = tensor("op_2999_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_135_to_fp16 = cast(dtype = var_135_to_fp16_dtype_0, x = var_135)[name = tensor("cast_383")]; tensor var_3000_cast_fp16 = add(x = var_135_to_fp16, y = var_2999_to_fp16)[name = tensor("op_3000_cast_fp16")]; tensor var_3000_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3000_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3001_epsilon_0 = const()[name = tensor("op_3001_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3000_cast_fp16_to_fp32 = cast(dtype = var_3000_cast_fp16_to_fp32_dtype_0, x = var_3000_cast_fp16)[name = tensor("cast_382")]; tensor var_3001 = rsqrt(epsilon = var_3001_epsilon_0, x = var_3000_cast_fp16_to_fp32)[name = tensor("op_3001")]; tensor var_3001_to_fp16_dtype_0 = const()[name = tensor("op_3001_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3001_to_fp16 = cast(dtype = var_3001_to_fp16_dtype_0, x = var_3001)[name = tensor("cast_381")]; tensor x_675_cast_fp16 = mul(x = x_669_cast_fp16, y = var_3001_to_fp16)[name = tensor("x_675_cast_fp16")]; tensor layers_16_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_16_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(369313280)))]; tensor x_677_cast_fp16 = mul(x = layers_16_post_attention_layernorm_weight_to_fp16, y = x_675_cast_fp16)[name = tensor("x_677_cast_fp16")]; tensor layers_16_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_16_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(369314368)))]; tensor linear_116_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_16_mlp_gate_proj_weight_to_fp16, x = x_677_cast_fp16)[name = tensor("linear_116_cast_fp16")]; tensor var_3012_cast_fp16 = silu(x = linear_116_cast_fp16)[name = tensor("op_3012_cast_fp16")]; tensor layers_16_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_16_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(370887296)))]; tensor linear_117_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_16_mlp_up_proj_weight_to_fp16, x = x_677_cast_fp16)[name = tensor("linear_117_cast_fp16")]; tensor x_679_cast_fp16 = mul(x = var_3012_cast_fp16, y = linear_117_cast_fp16)[name = tensor("x_679_cast_fp16")]; tensor layers_16_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_16_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(372460224)))]; tensor linear_118_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_16_mlp_down_proj_weight_to_fp16, x = x_679_cast_fp16)[name = tensor("linear_118_cast_fp16")]; tensor x_681_cast_fp16 = add(x = x_669_cast_fp16, y = linear_118_cast_fp16)[name = tensor("x_681_cast_fp16")]; tensor x_681_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_681_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_35_begin_0 = const()[name = tensor("k_cache_35_begin_0"), val = tensor([17, 0, 0, 0, 0])]; tensor k_cache_35_end_0 = const()[name = tensor("k_cache_35_end_0"), val = tensor([18, 1, 4, 2048, 128])]; tensor k_cache_35_end_mask_0 = const()[name = tensor("k_cache_35_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_35_squeeze_mask_0 = const()[name = tensor("k_cache_35_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_35_cast_fp16 = slice_by_index(begin = k_cache_35_begin_0, end = k_cache_35_end_0, end_mask = k_cache_35_end_mask_0, squeeze_mask = k_cache_35_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_35_cast_fp16")]; tensor v_cache_35_begin_0 = const()[name = tensor("v_cache_35_begin_0"), val = tensor([17, 0, 0, 0, 0])]; tensor v_cache_35_end_0 = const()[name = tensor("v_cache_35_end_0"), val = tensor([18, 1, 4, 2048, 128])]; tensor v_cache_35_end_mask_0 = const()[name = tensor("v_cache_35_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_35_squeeze_mask_0 = const()[name = tensor("v_cache_35_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_35_cast_fp16 = slice_by_index(begin = v_cache_35_begin_0, end = v_cache_35_end_0, end_mask = v_cache_35_end_mask_0, squeeze_mask = v_cache_35_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_35_cast_fp16")]; tensor var_3044 = const()[name = tensor("op_3044"), val = tensor(-1)]; tensor var_3043_promoted = const()[name = tensor("op_3043_promoted"), val = tensor(0x1p+1)]; tensor x_681_cast_fp16_to_fp32 = cast(dtype = x_681_cast_fp16_to_fp32_dtype_0, x = x_681_cast_fp16)[name = tensor("cast_380")]; tensor var_3053 = pow(x = x_681_cast_fp16_to_fp32, y = var_3043_promoted)[name = tensor("op_3053")]; tensor var_137_axes_0 = const()[name = tensor("var_137_axes_0"), val = tensor([-1])]; tensor var_137_keep_dims_0 = const()[name = tensor("var_137_keep_dims_0"), val = tensor(true)]; tensor var_137 = reduce_mean(axes = var_137_axes_0, keep_dims = var_137_keep_dims_0, x = var_3053)[name = tensor("var_137")]; tensor var_137_to_fp16_dtype_0 = const()[name = tensor("var_137_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3057_to_fp16 = const()[name = tensor("op_3057_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_137_to_fp16 = cast(dtype = var_137_to_fp16_dtype_0, x = var_137)[name = tensor("cast_379")]; tensor var_3058_cast_fp16 = add(x = var_137_to_fp16, y = var_3057_to_fp16)[name = tensor("op_3058_cast_fp16")]; tensor var_3058_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3058_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3059_epsilon_0 = const()[name = tensor("op_3059_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3058_cast_fp16_to_fp32 = cast(dtype = var_3058_cast_fp16_to_fp32_dtype_0, x = var_3058_cast_fp16)[name = tensor("cast_378")]; tensor var_3059 = rsqrt(epsilon = var_3059_epsilon_0, x = var_3058_cast_fp16_to_fp32)[name = tensor("op_3059")]; tensor var_3059_to_fp16_dtype_0 = const()[name = tensor("op_3059_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3059_to_fp16 = cast(dtype = var_3059_to_fp16_dtype_0, x = var_3059)[name = tensor("cast_377")]; tensor x_687_cast_fp16 = mul(x = x_681_cast_fp16, y = var_3059_to_fp16)[name = tensor("x_687_cast_fp16")]; tensor layers_17_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_17_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(374033152)))]; tensor x_689_cast_fp16 = mul(x = layers_17_input_layernorm_weight_to_fp16, y = x_687_cast_fp16)[name = tensor("x_689_cast_fp16")]; tensor layers_17_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_17_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(374034240)))]; tensor linear_119_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_17_self_attn_q_proj_weight_to_fp16, x = x_689_cast_fp16)[name = tensor("linear_119_cast_fp16")]; tensor var_3073 = const()[name = tensor("op_3073"), val = tensor([1, 1, 12, 128])]; tensor x_691_cast_fp16 = reshape(shape = var_3073, x = linear_119_cast_fp16)[name = tensor("x_691_cast_fp16")]; tensor x_691_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_691_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3043_promoted_1 = const()[name = tensor("op_3043_promoted_1"), val = tensor(0x1p+1)]; tensor x_691_cast_fp16_to_fp32 = cast(dtype = x_691_cast_fp16_to_fp32_dtype_0, x = x_691_cast_fp16)[name = tensor("cast_376")]; tensor var_3077 = pow(x = x_691_cast_fp16_to_fp32, y = var_3043_promoted_1)[name = tensor("op_3077")]; tensor var_139_axes_0 = const()[name = tensor("var_139_axes_0"), val = tensor([-1])]; tensor var_139_keep_dims_0 = const()[name = tensor("var_139_keep_dims_0"), val = tensor(true)]; tensor var_139 = reduce_mean(axes = var_139_axes_0, keep_dims = var_139_keep_dims_0, x = var_3077)[name = tensor("var_139")]; tensor var_139_to_fp16_dtype_0 = const()[name = tensor("var_139_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3081_to_fp16 = const()[name = tensor("op_3081_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_139_to_fp16 = cast(dtype = var_139_to_fp16_dtype_0, x = var_139)[name = tensor("cast_375")]; tensor var_3082_cast_fp16 = add(x = var_139_to_fp16, y = var_3081_to_fp16)[name = tensor("op_3082_cast_fp16")]; tensor var_3082_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3082_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3083_epsilon_0 = const()[name = tensor("op_3083_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3082_cast_fp16_to_fp32 = cast(dtype = var_3082_cast_fp16_to_fp32_dtype_0, x = var_3082_cast_fp16)[name = tensor("cast_374")]; tensor var_3083 = rsqrt(epsilon = var_3083_epsilon_0, x = var_3082_cast_fp16_to_fp32)[name = tensor("op_3083")]; tensor var_3083_to_fp16_dtype_0 = const()[name = tensor("op_3083_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3083_to_fp16 = cast(dtype = var_3083_to_fp16_dtype_0, x = var_3083)[name = tensor("cast_373")]; tensor x_697_cast_fp16 = mul(x = x_691_cast_fp16, y = var_3083_to_fp16)[name = tensor("x_697_cast_fp16")]; tensor layers_17_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_17_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(375607168)))]; tensor var_3085_cast_fp16 = mul(x = layers_17_self_attn_q_norm_weight_to_fp16, y = x_697_cast_fp16)[name = tensor("op_3085_cast_fp16")]; tensor q_69_perm_0 = const()[name = tensor("q_69_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_17_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_17_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(375607488)))]; tensor linear_120_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_17_self_attn_k_proj_weight_to_fp16, x = x_689_cast_fp16)[name = tensor("linear_120_cast_fp16")]; tensor var_3089 = const()[name = tensor("op_3089"), val = tensor([1, 1, 4, 128])]; tensor x_699_cast_fp16 = reshape(shape = var_3089, x = linear_120_cast_fp16)[name = tensor("x_699_cast_fp16")]; tensor x_699_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_699_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3043_promoted_2 = const()[name = tensor("op_3043_promoted_2"), val = tensor(0x1p+1)]; tensor x_699_cast_fp16_to_fp32 = cast(dtype = x_699_cast_fp16_to_fp32_dtype_0, x = x_699_cast_fp16)[name = tensor("cast_372")]; tensor var_3093 = pow(x = x_699_cast_fp16_to_fp32, y = var_3043_promoted_2)[name = tensor("op_3093")]; tensor var_141_axes_0 = const()[name = tensor("var_141_axes_0"), val = tensor([-1])]; tensor var_141_keep_dims_0 = const()[name = tensor("var_141_keep_dims_0"), val = tensor(true)]; tensor var_141 = reduce_mean(axes = var_141_axes_0, keep_dims = var_141_keep_dims_0, x = var_3093)[name = tensor("var_141")]; tensor var_141_to_fp16_dtype_0 = const()[name = tensor("var_141_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3097_to_fp16 = const()[name = tensor("op_3097_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_141_to_fp16 = cast(dtype = var_141_to_fp16_dtype_0, x = var_141)[name = tensor("cast_371")]; tensor var_3098_cast_fp16 = add(x = var_141_to_fp16, y = var_3097_to_fp16)[name = tensor("op_3098_cast_fp16")]; tensor var_3098_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3098_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3099_epsilon_0 = const()[name = tensor("op_3099_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3098_cast_fp16_to_fp32 = cast(dtype = var_3098_cast_fp16_to_fp32_dtype_0, x = var_3098_cast_fp16)[name = tensor("cast_370")]; tensor var_3099 = rsqrt(epsilon = var_3099_epsilon_0, x = var_3098_cast_fp16_to_fp32)[name = tensor("op_3099")]; tensor var_3099_to_fp16_dtype_0 = const()[name = tensor("op_3099_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3099_to_fp16 = cast(dtype = var_3099_to_fp16_dtype_0, x = var_3099)[name = tensor("cast_369")]; tensor x_705_cast_fp16 = mul(x = x_699_cast_fp16, y = var_3099_to_fp16)[name = tensor("x_705_cast_fp16")]; tensor layers_17_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_17_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(376131840)))]; tensor var_3101_cast_fp16 = mul(x = layers_17_self_attn_k_norm_weight_to_fp16, y = x_705_cast_fp16)[name = tensor("op_3101_cast_fp16")]; tensor k_69_perm_0 = const()[name = tensor("k_69_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_17_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_17_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(376132160)))]; tensor linear_121_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_17_self_attn_v_proj_weight_to_fp16, x = x_689_cast_fp16)[name = tensor("linear_121_cast_fp16")]; tensor var_3105 = const()[name = tensor("op_3105"), val = tensor([1, 1, 4, 128])]; tensor var_3106_cast_fp16 = reshape(shape = var_3105, x = linear_121_cast_fp16)[name = tensor("op_3106_cast_fp16")]; tensor v_35_perm_0 = const()[name = tensor("v_35_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_69_cast_fp16 = transpose(perm = q_69_perm_0, x = var_3085_cast_fp16)[name = tensor("transpose_43")]; tensor var_3110_cast_fp16 = mul(x = q_69_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3110_cast_fp16")]; tensor x1_69_begin_0 = const()[name = tensor("x1_69_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_69_end_0 = const()[name = tensor("x1_69_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_69_end_mask_0 = const()[name = tensor("x1_69_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_69_cast_fp16 = slice_by_index(begin = x1_69_begin_0, end = x1_69_end_0, end_mask = x1_69_end_mask_0, x = q_69_cast_fp16)[name = tensor("x1_69_cast_fp16")]; tensor x2_69_begin_0 = const()[name = tensor("x2_69_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_69_end_0 = const()[name = tensor("x2_69_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_69_end_mask_0 = const()[name = tensor("x2_69_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_69_cast_fp16 = slice_by_index(begin = x2_69_begin_0, end = x2_69_end_0, end_mask = x2_69_end_mask_0, x = q_69_cast_fp16)[name = tensor("x2_69_cast_fp16")]; tensor const_35_promoted_to_fp16 = const()[name = tensor("const_35_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3113_cast_fp16 = mul(x = x2_69_cast_fp16, y = const_35_promoted_to_fp16)[name = tensor("op_3113_cast_fp16")]; tensor var_3115_interleave_0 = const()[name = tensor("op_3115_interleave_0"), val = tensor(false)]; tensor var_3115_cast_fp16 = concat(axis = var_3044, interleave = var_3115_interleave_0, values = (var_3113_cast_fp16, x1_69_cast_fp16))[name = tensor("op_3115_cast_fp16")]; tensor var_3116_cast_fp16 = mul(x = var_3115_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3116_cast_fp16")]; tensor q_71_cast_fp16 = add(x = var_3110_cast_fp16, y = var_3116_cast_fp16)[name = tensor("q_71_cast_fp16")]; tensor k_69_cast_fp16 = transpose(perm = k_69_perm_0, x = var_3101_cast_fp16)[name = tensor("transpose_42")]; tensor var_3118_cast_fp16 = mul(x = k_69_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3118_cast_fp16")]; tensor x1_71_begin_0 = const()[name = tensor("x1_71_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_71_end_0 = const()[name = tensor("x1_71_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_71_end_mask_0 = const()[name = tensor("x1_71_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_71_cast_fp16 = slice_by_index(begin = x1_71_begin_0, end = x1_71_end_0, end_mask = x1_71_end_mask_0, x = k_69_cast_fp16)[name = tensor("x1_71_cast_fp16")]; tensor x2_71_begin_0 = const()[name = tensor("x2_71_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_71_end_0 = const()[name = tensor("x2_71_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_71_end_mask_0 = const()[name = tensor("x2_71_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_71_cast_fp16 = slice_by_index(begin = x2_71_begin_0, end = x2_71_end_0, end_mask = x2_71_end_mask_0, x = k_69_cast_fp16)[name = tensor("x2_71_cast_fp16")]; tensor const_36_promoted_to_fp16 = const()[name = tensor("const_36_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3121_cast_fp16 = mul(x = x2_71_cast_fp16, y = const_36_promoted_to_fp16)[name = tensor("op_3121_cast_fp16")]; tensor var_3123_interleave_0 = const()[name = tensor("op_3123_interleave_0"), val = tensor(false)]; tensor var_3123_cast_fp16 = concat(axis = var_3044, interleave = var_3123_interleave_0, values = (var_3121_cast_fp16, x1_71_cast_fp16))[name = tensor("op_3123_cast_fp16")]; tensor var_3124_cast_fp16 = mul(x = var_3123_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3124_cast_fp16")]; tensor k_71_cast_fp16 = add(x = var_3118_cast_fp16, y = var_3124_cast_fp16)[name = tensor("k_71_cast_fp16")]; tensor var_3127_cast_fp16 = mul(x = k_cache_35_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3127_cast_fp16")]; tensor var_3128_cast_fp16 = mul(x = k_71_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3128_cast_fp16")]; tensor k_full_35_cast_fp16 = add(x = var_3127_cast_fp16, y = var_3128_cast_fp16)[name = tensor("k_full_35_cast_fp16")]; tensor var_3131_cast_fp16 = mul(x = v_cache_35_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3131_cast_fp16")]; tensor v_35_cast_fp16 = transpose(perm = v_35_perm_0, x = var_3106_cast_fp16)[name = tensor("transpose_41")]; tensor var_3132_cast_fp16 = mul(x = v_35_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3132_cast_fp16")]; tensor v_full_35_cast_fp16 = add(x = var_3131_cast_fp16, y = var_3132_cast_fp16)[name = tensor("v_full_35_cast_fp16")]; tensor var_3134_axes_0 = const()[name = tensor("op_3134_axes_0"), val = tensor([2])]; tensor var_3134_cast_fp16 = expand_dims(axes = var_3134_axes_0, x = k_full_35_cast_fp16)[name = tensor("op_3134_cast_fp16")]; tensor var_3136_reps_0 = const()[name = tensor("op_3136_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3136_cast_fp16 = tile(reps = var_3136_reps_0, x = var_3134_cast_fp16)[name = tensor("op_3136_cast_fp16")]; tensor var_3137 = const()[name = tensor("op_3137"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_35_cast_fp16 = reshape(shape = var_3137, x = var_3136_cast_fp16)[name = tensor("k_rep_35_cast_fp16")]; tensor var_3139_axes_0 = const()[name = tensor("op_3139_axes_0"), val = tensor([2])]; tensor var_3139_cast_fp16 = expand_dims(axes = var_3139_axes_0, x = v_full_35_cast_fp16)[name = tensor("op_3139_cast_fp16")]; tensor var_3141_reps_0 = const()[name = tensor("op_3141_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3141_cast_fp16 = tile(reps = var_3141_reps_0, x = var_3139_cast_fp16)[name = tensor("op_3141_cast_fp16")]; tensor var_3142 = const()[name = tensor("op_3142"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_35_cast_fp16 = reshape(shape = var_3142, x = var_3141_cast_fp16)[name = tensor("v_rep_35_cast_fp16")]; tensor var_3145_transpose_x_1 = const()[name = tensor("op_3145_transpose_x_1"), val = tensor(false)]; tensor var_3145_transpose_y_1 = const()[name = tensor("op_3145_transpose_y_1"), val = tensor(true)]; tensor var_3145_cast_fp16 = matmul(transpose_x = var_3145_transpose_x_1, transpose_y = var_3145_transpose_y_1, x = q_71_cast_fp16, y = k_rep_35_cast_fp16)[name = tensor("op_3145_cast_fp16")]; tensor var_3146_to_fp16 = const()[name = tensor("op_3146_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_69_cast_fp16 = mul(x = var_3145_cast_fp16, y = var_3146_to_fp16)[name = tensor("attn_69_cast_fp16")]; tensor input_71_cast_fp16 = add(x = attn_69_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_71_cast_fp16")]; tensor input_71_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_71_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_71_cast_fp16_to_fp32 = cast(dtype = input_71_cast_fp16_to_fp32_dtype_0, x = input_71_cast_fp16)[name = tensor("cast_368")]; tensor attn_71 = softmax(axis = var_3044, x = input_71_cast_fp16_to_fp32)[name = tensor("attn_71")]; tensor out_35_transpose_x_0 = const()[name = tensor("out_35_transpose_x_0"), val = tensor(false)]; tensor out_35_transpose_y_0 = const()[name = tensor("out_35_transpose_y_0"), val = tensor(false)]; tensor attn_71_to_fp16_dtype_0 = const()[name = tensor("attn_71_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_71_to_fp16 = cast(dtype = attn_71_to_fp16_dtype_0, x = attn_71)[name = tensor("cast_367")]; tensor out_35_cast_fp16 = matmul(transpose_x = out_35_transpose_x_0, transpose_y = out_35_transpose_y_0, x = attn_71_to_fp16, y = v_rep_35_cast_fp16)[name = tensor("out_35_cast_fp16")]; tensor var_3151_perm_0 = const()[name = tensor("op_3151_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3153 = const()[name = tensor("op_3153"), val = tensor([1, 1, 1536])]; tensor var_3151_cast_fp16 = transpose(perm = var_3151_perm_0, x = out_35_cast_fp16)[name = tensor("transpose_40")]; tensor x_707_cast_fp16 = reshape(shape = var_3153, x = var_3151_cast_fp16)[name = tensor("x_707_cast_fp16")]; tensor layers_17_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_17_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(376656512)))]; tensor linear_122_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_17_self_attn_o_proj_weight_to_fp16, x = x_707_cast_fp16)[name = tensor("linear_122_cast_fp16")]; tensor x_709_cast_fp16 = add(x = x_681_cast_fp16, y = linear_122_cast_fp16)[name = tensor("x_709_cast_fp16")]; tensor x_709_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_709_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3043_promoted_3 = const()[name = tensor("op_3043_promoted_3"), val = tensor(0x1p+1)]; tensor x_709_cast_fp16_to_fp32 = cast(dtype = x_709_cast_fp16_to_fp32_dtype_0, x = x_709_cast_fp16)[name = tensor("cast_366")]; tensor var_3164 = pow(x = x_709_cast_fp16_to_fp32, y = var_3043_promoted_3)[name = tensor("op_3164")]; tensor var_143_axes_0 = const()[name = tensor("var_143_axes_0"), val = tensor([-1])]; tensor var_143_keep_dims_0 = const()[name = tensor("var_143_keep_dims_0"), val = tensor(true)]; tensor var_143 = reduce_mean(axes = var_143_axes_0, keep_dims = var_143_keep_dims_0, x = var_3164)[name = tensor("var_143")]; tensor var_143_to_fp16_dtype_0 = const()[name = tensor("var_143_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3168_to_fp16 = const()[name = tensor("op_3168_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_143_to_fp16 = cast(dtype = var_143_to_fp16_dtype_0, x = var_143)[name = tensor("cast_365")]; tensor var_3169_cast_fp16 = add(x = var_143_to_fp16, y = var_3168_to_fp16)[name = tensor("op_3169_cast_fp16")]; tensor var_3169_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3169_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3170_epsilon_0 = const()[name = tensor("op_3170_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3169_cast_fp16_to_fp32 = cast(dtype = var_3169_cast_fp16_to_fp32_dtype_0, x = var_3169_cast_fp16)[name = tensor("cast_364")]; tensor var_3170 = rsqrt(epsilon = var_3170_epsilon_0, x = var_3169_cast_fp16_to_fp32)[name = tensor("op_3170")]; tensor var_3170_to_fp16_dtype_0 = const()[name = tensor("op_3170_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3170_to_fp16 = cast(dtype = var_3170_to_fp16_dtype_0, x = var_3170)[name = tensor("cast_363")]; tensor x_715_cast_fp16 = mul(x = x_709_cast_fp16, y = var_3170_to_fp16)[name = tensor("x_715_cast_fp16")]; tensor layers_17_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_17_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(378229440)))]; tensor x_717_cast_fp16 = mul(x = layers_17_post_attention_layernorm_weight_to_fp16, y = x_715_cast_fp16)[name = tensor("x_717_cast_fp16")]; tensor layers_17_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_17_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(378230528)))]; tensor linear_123_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_17_mlp_gate_proj_weight_to_fp16, x = x_717_cast_fp16)[name = tensor("linear_123_cast_fp16")]; tensor var_3181_cast_fp16 = silu(x = linear_123_cast_fp16)[name = tensor("op_3181_cast_fp16")]; tensor layers_17_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_17_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(379803456)))]; tensor linear_124_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_17_mlp_up_proj_weight_to_fp16, x = x_717_cast_fp16)[name = tensor("linear_124_cast_fp16")]; tensor x_719_cast_fp16 = mul(x = var_3181_cast_fp16, y = linear_124_cast_fp16)[name = tensor("x_719_cast_fp16")]; tensor layers_17_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_17_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(381376384)))]; tensor linear_125_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_17_mlp_down_proj_weight_to_fp16, x = x_719_cast_fp16)[name = tensor("linear_125_cast_fp16")]; tensor x_721_cast_fp16 = add(x = x_709_cast_fp16, y = linear_125_cast_fp16)[name = tensor("x_721_cast_fp16")]; tensor x_721_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_721_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_37_begin_0 = const()[name = tensor("k_cache_37_begin_0"), val = tensor([18, 0, 0, 0, 0])]; tensor k_cache_37_end_0 = const()[name = tensor("k_cache_37_end_0"), val = tensor([19, 1, 4, 2048, 128])]; tensor k_cache_37_end_mask_0 = const()[name = tensor("k_cache_37_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_37_squeeze_mask_0 = const()[name = tensor("k_cache_37_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_37_cast_fp16 = slice_by_index(begin = k_cache_37_begin_0, end = k_cache_37_end_0, end_mask = k_cache_37_end_mask_0, squeeze_mask = k_cache_37_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_37_cast_fp16")]; tensor v_cache_37_begin_0 = const()[name = tensor("v_cache_37_begin_0"), val = tensor([18, 0, 0, 0, 0])]; tensor v_cache_37_end_0 = const()[name = tensor("v_cache_37_end_0"), val = tensor([19, 1, 4, 2048, 128])]; tensor v_cache_37_end_mask_0 = const()[name = tensor("v_cache_37_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_37_squeeze_mask_0 = const()[name = tensor("v_cache_37_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_37_cast_fp16 = slice_by_index(begin = v_cache_37_begin_0, end = v_cache_37_end_0, end_mask = v_cache_37_end_mask_0, squeeze_mask = v_cache_37_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_37_cast_fp16")]; tensor var_3213 = const()[name = tensor("op_3213"), val = tensor(-1)]; tensor var_3212_promoted = const()[name = tensor("op_3212_promoted"), val = tensor(0x1p+1)]; tensor x_721_cast_fp16_to_fp32 = cast(dtype = x_721_cast_fp16_to_fp32_dtype_0, x = x_721_cast_fp16)[name = tensor("cast_362")]; tensor var_3222 = pow(x = x_721_cast_fp16_to_fp32, y = var_3212_promoted)[name = tensor("op_3222")]; tensor var_145_axes_0 = const()[name = tensor("var_145_axes_0"), val = tensor([-1])]; tensor var_145_keep_dims_0 = const()[name = tensor("var_145_keep_dims_0"), val = tensor(true)]; tensor var_145 = reduce_mean(axes = var_145_axes_0, keep_dims = var_145_keep_dims_0, x = var_3222)[name = tensor("var_145")]; tensor var_145_to_fp16_dtype_0 = const()[name = tensor("var_145_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3226_to_fp16 = const()[name = tensor("op_3226_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_145_to_fp16 = cast(dtype = var_145_to_fp16_dtype_0, x = var_145)[name = tensor("cast_361")]; tensor var_3227_cast_fp16 = add(x = var_145_to_fp16, y = var_3226_to_fp16)[name = tensor("op_3227_cast_fp16")]; tensor var_3227_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3227_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3228_epsilon_0 = const()[name = tensor("op_3228_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3227_cast_fp16_to_fp32 = cast(dtype = var_3227_cast_fp16_to_fp32_dtype_0, x = var_3227_cast_fp16)[name = tensor("cast_360")]; tensor var_3228 = rsqrt(epsilon = var_3228_epsilon_0, x = var_3227_cast_fp16_to_fp32)[name = tensor("op_3228")]; tensor var_3228_to_fp16_dtype_0 = const()[name = tensor("op_3228_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3228_to_fp16 = cast(dtype = var_3228_to_fp16_dtype_0, x = var_3228)[name = tensor("cast_359")]; tensor x_727_cast_fp16 = mul(x = x_721_cast_fp16, y = var_3228_to_fp16)[name = tensor("x_727_cast_fp16")]; tensor layers_18_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_18_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(382949312)))]; tensor x_729_cast_fp16 = mul(x = layers_18_input_layernorm_weight_to_fp16, y = x_727_cast_fp16)[name = tensor("x_729_cast_fp16")]; tensor layers_18_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_18_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(382950400)))]; tensor linear_126_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_18_self_attn_q_proj_weight_to_fp16, x = x_729_cast_fp16)[name = tensor("linear_126_cast_fp16")]; tensor var_3242 = const()[name = tensor("op_3242"), val = tensor([1, 1, 12, 128])]; tensor x_731_cast_fp16 = reshape(shape = var_3242, x = linear_126_cast_fp16)[name = tensor("x_731_cast_fp16")]; tensor x_731_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_731_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3212_promoted_1 = const()[name = tensor("op_3212_promoted_1"), val = tensor(0x1p+1)]; tensor x_731_cast_fp16_to_fp32 = cast(dtype = x_731_cast_fp16_to_fp32_dtype_0, x = x_731_cast_fp16)[name = tensor("cast_358")]; tensor var_3246 = pow(x = x_731_cast_fp16_to_fp32, y = var_3212_promoted_1)[name = tensor("op_3246")]; tensor var_147_axes_0 = const()[name = tensor("var_147_axes_0"), val = tensor([-1])]; tensor var_147_keep_dims_0 = const()[name = tensor("var_147_keep_dims_0"), val = tensor(true)]; tensor var_147 = reduce_mean(axes = var_147_axes_0, keep_dims = var_147_keep_dims_0, x = var_3246)[name = tensor("var_147")]; tensor var_147_to_fp16_dtype_0 = const()[name = tensor("var_147_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3250_to_fp16 = const()[name = tensor("op_3250_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_147_to_fp16 = cast(dtype = var_147_to_fp16_dtype_0, x = var_147)[name = tensor("cast_357")]; tensor var_3251_cast_fp16 = add(x = var_147_to_fp16, y = var_3250_to_fp16)[name = tensor("op_3251_cast_fp16")]; tensor var_3251_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3251_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3252_epsilon_0 = const()[name = tensor("op_3252_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3251_cast_fp16_to_fp32 = cast(dtype = var_3251_cast_fp16_to_fp32_dtype_0, x = var_3251_cast_fp16)[name = tensor("cast_356")]; tensor var_3252 = rsqrt(epsilon = var_3252_epsilon_0, x = var_3251_cast_fp16_to_fp32)[name = tensor("op_3252")]; tensor var_3252_to_fp16_dtype_0 = const()[name = tensor("op_3252_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3252_to_fp16 = cast(dtype = var_3252_to_fp16_dtype_0, x = var_3252)[name = tensor("cast_355")]; tensor x_737_cast_fp16 = mul(x = x_731_cast_fp16, y = var_3252_to_fp16)[name = tensor("x_737_cast_fp16")]; tensor layers_18_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_18_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(384523328)))]; tensor var_3254_cast_fp16 = mul(x = layers_18_self_attn_q_norm_weight_to_fp16, y = x_737_cast_fp16)[name = tensor("op_3254_cast_fp16")]; tensor q_73_perm_0 = const()[name = tensor("q_73_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_18_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_18_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(384523648)))]; tensor linear_127_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_18_self_attn_k_proj_weight_to_fp16, x = x_729_cast_fp16)[name = tensor("linear_127_cast_fp16")]; tensor var_3258 = const()[name = tensor("op_3258"), val = tensor([1, 1, 4, 128])]; tensor x_739_cast_fp16 = reshape(shape = var_3258, x = linear_127_cast_fp16)[name = tensor("x_739_cast_fp16")]; tensor x_739_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_739_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3212_promoted_2 = const()[name = tensor("op_3212_promoted_2"), val = tensor(0x1p+1)]; tensor x_739_cast_fp16_to_fp32 = cast(dtype = x_739_cast_fp16_to_fp32_dtype_0, x = x_739_cast_fp16)[name = tensor("cast_354")]; tensor var_3262 = pow(x = x_739_cast_fp16_to_fp32, y = var_3212_promoted_2)[name = tensor("op_3262")]; tensor var_149_axes_0 = const()[name = tensor("var_149_axes_0"), val = tensor([-1])]; tensor var_149_keep_dims_0 = const()[name = tensor("var_149_keep_dims_0"), val = tensor(true)]; tensor var_149 = reduce_mean(axes = var_149_axes_0, keep_dims = var_149_keep_dims_0, x = var_3262)[name = tensor("var_149")]; tensor var_149_to_fp16_dtype_0 = const()[name = tensor("var_149_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3266_to_fp16 = const()[name = tensor("op_3266_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_149_to_fp16 = cast(dtype = var_149_to_fp16_dtype_0, x = var_149)[name = tensor("cast_353")]; tensor var_3267_cast_fp16 = add(x = var_149_to_fp16, y = var_3266_to_fp16)[name = tensor("op_3267_cast_fp16")]; tensor var_3267_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3267_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3268_epsilon_0 = const()[name = tensor("op_3268_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3267_cast_fp16_to_fp32 = cast(dtype = var_3267_cast_fp16_to_fp32_dtype_0, x = var_3267_cast_fp16)[name = tensor("cast_352")]; tensor var_3268 = rsqrt(epsilon = var_3268_epsilon_0, x = var_3267_cast_fp16_to_fp32)[name = tensor("op_3268")]; tensor var_3268_to_fp16_dtype_0 = const()[name = tensor("op_3268_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3268_to_fp16 = cast(dtype = var_3268_to_fp16_dtype_0, x = var_3268)[name = tensor("cast_351")]; tensor x_745_cast_fp16 = mul(x = x_739_cast_fp16, y = var_3268_to_fp16)[name = tensor("x_745_cast_fp16")]; tensor layers_18_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_18_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385048000)))]; tensor var_3270_cast_fp16 = mul(x = layers_18_self_attn_k_norm_weight_to_fp16, y = x_745_cast_fp16)[name = tensor("op_3270_cast_fp16")]; tensor k_73_perm_0 = const()[name = tensor("k_73_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_18_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_18_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385048320)))]; tensor linear_128_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_18_self_attn_v_proj_weight_to_fp16, x = x_729_cast_fp16)[name = tensor("linear_128_cast_fp16")]; tensor var_3274 = const()[name = tensor("op_3274"), val = tensor([1, 1, 4, 128])]; tensor var_3275_cast_fp16 = reshape(shape = var_3274, x = linear_128_cast_fp16)[name = tensor("op_3275_cast_fp16")]; tensor v_37_perm_0 = const()[name = tensor("v_37_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_73_cast_fp16 = transpose(perm = q_73_perm_0, x = var_3254_cast_fp16)[name = tensor("transpose_39")]; tensor var_3279_cast_fp16 = mul(x = q_73_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3279_cast_fp16")]; tensor x1_73_begin_0 = const()[name = tensor("x1_73_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_73_end_0 = const()[name = tensor("x1_73_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_73_end_mask_0 = const()[name = tensor("x1_73_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_73_cast_fp16 = slice_by_index(begin = x1_73_begin_0, end = x1_73_end_0, end_mask = x1_73_end_mask_0, x = q_73_cast_fp16)[name = tensor("x1_73_cast_fp16")]; tensor x2_73_begin_0 = const()[name = tensor("x2_73_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_73_end_0 = const()[name = tensor("x2_73_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_73_end_mask_0 = const()[name = tensor("x2_73_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_73_cast_fp16 = slice_by_index(begin = x2_73_begin_0, end = x2_73_end_0, end_mask = x2_73_end_mask_0, x = q_73_cast_fp16)[name = tensor("x2_73_cast_fp16")]; tensor const_37_promoted_to_fp16 = const()[name = tensor("const_37_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3282_cast_fp16 = mul(x = x2_73_cast_fp16, y = const_37_promoted_to_fp16)[name = tensor("op_3282_cast_fp16")]; tensor var_3284_interleave_0 = const()[name = tensor("op_3284_interleave_0"), val = tensor(false)]; tensor var_3284_cast_fp16 = concat(axis = var_3213, interleave = var_3284_interleave_0, values = (var_3282_cast_fp16, x1_73_cast_fp16))[name = tensor("op_3284_cast_fp16")]; tensor var_3285_cast_fp16 = mul(x = var_3284_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3285_cast_fp16")]; tensor q_75_cast_fp16 = add(x = var_3279_cast_fp16, y = var_3285_cast_fp16)[name = tensor("q_75_cast_fp16")]; tensor k_73_cast_fp16 = transpose(perm = k_73_perm_0, x = var_3270_cast_fp16)[name = tensor("transpose_38")]; tensor var_3287_cast_fp16 = mul(x = k_73_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3287_cast_fp16")]; tensor x1_75_begin_0 = const()[name = tensor("x1_75_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_75_end_0 = const()[name = tensor("x1_75_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_75_end_mask_0 = const()[name = tensor("x1_75_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_75_cast_fp16 = slice_by_index(begin = x1_75_begin_0, end = x1_75_end_0, end_mask = x1_75_end_mask_0, x = k_73_cast_fp16)[name = tensor("x1_75_cast_fp16")]; tensor x2_75_begin_0 = const()[name = tensor("x2_75_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_75_end_0 = const()[name = tensor("x2_75_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_75_end_mask_0 = const()[name = tensor("x2_75_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_75_cast_fp16 = slice_by_index(begin = x2_75_begin_0, end = x2_75_end_0, end_mask = x2_75_end_mask_0, x = k_73_cast_fp16)[name = tensor("x2_75_cast_fp16")]; tensor const_38_promoted_to_fp16 = const()[name = tensor("const_38_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3290_cast_fp16 = mul(x = x2_75_cast_fp16, y = const_38_promoted_to_fp16)[name = tensor("op_3290_cast_fp16")]; tensor var_3292_interleave_0 = const()[name = tensor("op_3292_interleave_0"), val = tensor(false)]; tensor var_3292_cast_fp16 = concat(axis = var_3213, interleave = var_3292_interleave_0, values = (var_3290_cast_fp16, x1_75_cast_fp16))[name = tensor("op_3292_cast_fp16")]; tensor var_3293_cast_fp16 = mul(x = var_3292_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3293_cast_fp16")]; tensor k_75_cast_fp16 = add(x = var_3287_cast_fp16, y = var_3293_cast_fp16)[name = tensor("k_75_cast_fp16")]; tensor var_3296_cast_fp16 = mul(x = k_cache_37_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3296_cast_fp16")]; tensor var_3297_cast_fp16 = mul(x = k_75_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3297_cast_fp16")]; tensor k_full_37_cast_fp16 = add(x = var_3296_cast_fp16, y = var_3297_cast_fp16)[name = tensor("k_full_37_cast_fp16")]; tensor var_3300_cast_fp16 = mul(x = v_cache_37_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3300_cast_fp16")]; tensor v_37_cast_fp16 = transpose(perm = v_37_perm_0, x = var_3275_cast_fp16)[name = tensor("transpose_37")]; tensor var_3301_cast_fp16 = mul(x = v_37_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3301_cast_fp16")]; tensor v_full_37_cast_fp16 = add(x = var_3300_cast_fp16, y = var_3301_cast_fp16)[name = tensor("v_full_37_cast_fp16")]; tensor var_3303_axes_0 = const()[name = tensor("op_3303_axes_0"), val = tensor([2])]; tensor var_3303_cast_fp16 = expand_dims(axes = var_3303_axes_0, x = k_full_37_cast_fp16)[name = tensor("op_3303_cast_fp16")]; tensor var_3305_reps_0 = const()[name = tensor("op_3305_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3305_cast_fp16 = tile(reps = var_3305_reps_0, x = var_3303_cast_fp16)[name = tensor("op_3305_cast_fp16")]; tensor var_3306 = const()[name = tensor("op_3306"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_37_cast_fp16 = reshape(shape = var_3306, x = var_3305_cast_fp16)[name = tensor("k_rep_37_cast_fp16")]; tensor var_3308_axes_0 = const()[name = tensor("op_3308_axes_0"), val = tensor([2])]; tensor var_3308_cast_fp16 = expand_dims(axes = var_3308_axes_0, x = v_full_37_cast_fp16)[name = tensor("op_3308_cast_fp16")]; tensor var_3310_reps_0 = const()[name = tensor("op_3310_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3310_cast_fp16 = tile(reps = var_3310_reps_0, x = var_3308_cast_fp16)[name = tensor("op_3310_cast_fp16")]; tensor var_3311 = const()[name = tensor("op_3311"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_37_cast_fp16 = reshape(shape = var_3311, x = var_3310_cast_fp16)[name = tensor("v_rep_37_cast_fp16")]; tensor var_3314_transpose_x_1 = const()[name = tensor("op_3314_transpose_x_1"), val = tensor(false)]; tensor var_3314_transpose_y_1 = const()[name = tensor("op_3314_transpose_y_1"), val = tensor(true)]; tensor var_3314_cast_fp16 = matmul(transpose_x = var_3314_transpose_x_1, transpose_y = var_3314_transpose_y_1, x = q_75_cast_fp16, y = k_rep_37_cast_fp16)[name = tensor("op_3314_cast_fp16")]; tensor var_3315_to_fp16 = const()[name = tensor("op_3315_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_73_cast_fp16 = mul(x = var_3314_cast_fp16, y = var_3315_to_fp16)[name = tensor("attn_73_cast_fp16")]; tensor input_75_cast_fp16 = add(x = attn_73_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_75_cast_fp16")]; tensor input_75_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_75_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_75_cast_fp16_to_fp32 = cast(dtype = input_75_cast_fp16_to_fp32_dtype_0, x = input_75_cast_fp16)[name = tensor("cast_350")]; tensor attn_75 = softmax(axis = var_3213, x = input_75_cast_fp16_to_fp32)[name = tensor("attn_75")]; tensor out_37_transpose_x_0 = const()[name = tensor("out_37_transpose_x_0"), val = tensor(false)]; tensor out_37_transpose_y_0 = const()[name = tensor("out_37_transpose_y_0"), val = tensor(false)]; tensor attn_75_to_fp16_dtype_0 = const()[name = tensor("attn_75_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_75_to_fp16 = cast(dtype = attn_75_to_fp16_dtype_0, x = attn_75)[name = tensor("cast_349")]; tensor out_37_cast_fp16 = matmul(transpose_x = out_37_transpose_x_0, transpose_y = out_37_transpose_y_0, x = attn_75_to_fp16, y = v_rep_37_cast_fp16)[name = tensor("out_37_cast_fp16")]; tensor var_3320_perm_0 = const()[name = tensor("op_3320_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3322 = const()[name = tensor("op_3322"), val = tensor([1, 1, 1536])]; tensor var_3320_cast_fp16 = transpose(perm = var_3320_perm_0, x = out_37_cast_fp16)[name = tensor("transpose_36")]; tensor x_747_cast_fp16 = reshape(shape = var_3322, x = var_3320_cast_fp16)[name = tensor("x_747_cast_fp16")]; tensor layers_18_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_18_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385572672)))]; tensor linear_129_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_18_self_attn_o_proj_weight_to_fp16, x = x_747_cast_fp16)[name = tensor("linear_129_cast_fp16")]; tensor x_749_cast_fp16 = add(x = x_721_cast_fp16, y = linear_129_cast_fp16)[name = tensor("x_749_cast_fp16")]; tensor x_749_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_749_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3212_promoted_3 = const()[name = tensor("op_3212_promoted_3"), val = tensor(0x1p+1)]; tensor x_749_cast_fp16_to_fp32 = cast(dtype = x_749_cast_fp16_to_fp32_dtype_0, x = x_749_cast_fp16)[name = tensor("cast_348")]; tensor var_3333 = pow(x = x_749_cast_fp16_to_fp32, y = var_3212_promoted_3)[name = tensor("op_3333")]; tensor var_151_axes_0 = const()[name = tensor("var_151_axes_0"), val = tensor([-1])]; tensor var_151_keep_dims_0 = const()[name = tensor("var_151_keep_dims_0"), val = tensor(true)]; tensor var_151 = reduce_mean(axes = var_151_axes_0, keep_dims = var_151_keep_dims_0, x = var_3333)[name = tensor("var_151")]; tensor var_151_to_fp16_dtype_0 = const()[name = tensor("var_151_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3337_to_fp16 = const()[name = tensor("op_3337_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_151_to_fp16 = cast(dtype = var_151_to_fp16_dtype_0, x = var_151)[name = tensor("cast_347")]; tensor var_3338_cast_fp16 = add(x = var_151_to_fp16, y = var_3337_to_fp16)[name = tensor("op_3338_cast_fp16")]; tensor var_3338_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3338_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3339_epsilon_0 = const()[name = tensor("op_3339_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3338_cast_fp16_to_fp32 = cast(dtype = var_3338_cast_fp16_to_fp32_dtype_0, x = var_3338_cast_fp16)[name = tensor("cast_346")]; tensor var_3339 = rsqrt(epsilon = var_3339_epsilon_0, x = var_3338_cast_fp16_to_fp32)[name = tensor("op_3339")]; tensor var_3339_to_fp16_dtype_0 = const()[name = tensor("op_3339_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3339_to_fp16 = cast(dtype = var_3339_to_fp16_dtype_0, x = var_3339)[name = tensor("cast_345")]; tensor x_755_cast_fp16 = mul(x = x_749_cast_fp16, y = var_3339_to_fp16)[name = tensor("x_755_cast_fp16")]; tensor layers_18_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_18_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(387145600)))]; tensor x_757_cast_fp16 = mul(x = layers_18_post_attention_layernorm_weight_to_fp16, y = x_755_cast_fp16)[name = tensor("x_757_cast_fp16")]; tensor layers_18_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_18_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(387146688)))]; tensor linear_130_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_18_mlp_gate_proj_weight_to_fp16, x = x_757_cast_fp16)[name = tensor("linear_130_cast_fp16")]; tensor var_3350_cast_fp16 = silu(x = linear_130_cast_fp16)[name = tensor("op_3350_cast_fp16")]; tensor layers_18_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_18_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(388719616)))]; tensor linear_131_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_18_mlp_up_proj_weight_to_fp16, x = x_757_cast_fp16)[name = tensor("linear_131_cast_fp16")]; tensor x_759_cast_fp16 = mul(x = var_3350_cast_fp16, y = linear_131_cast_fp16)[name = tensor("x_759_cast_fp16")]; tensor layers_18_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_18_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(390292544)))]; tensor linear_132_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_18_mlp_down_proj_weight_to_fp16, x = x_759_cast_fp16)[name = tensor("linear_132_cast_fp16")]; tensor x_761_cast_fp16 = add(x = x_749_cast_fp16, y = linear_132_cast_fp16)[name = tensor("x_761_cast_fp16")]; tensor x_761_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_761_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_39_begin_0 = const()[name = tensor("k_cache_39_begin_0"), val = tensor([19, 0, 0, 0, 0])]; tensor k_cache_39_end_0 = const()[name = tensor("k_cache_39_end_0"), val = tensor([20, 1, 4, 2048, 128])]; tensor k_cache_39_end_mask_0 = const()[name = tensor("k_cache_39_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_39_squeeze_mask_0 = const()[name = tensor("k_cache_39_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_39_cast_fp16 = slice_by_index(begin = k_cache_39_begin_0, end = k_cache_39_end_0, end_mask = k_cache_39_end_mask_0, squeeze_mask = k_cache_39_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_39_cast_fp16")]; tensor v_cache_39_begin_0 = const()[name = tensor("v_cache_39_begin_0"), val = tensor([19, 0, 0, 0, 0])]; tensor v_cache_39_end_0 = const()[name = tensor("v_cache_39_end_0"), val = tensor([20, 1, 4, 2048, 128])]; tensor v_cache_39_end_mask_0 = const()[name = tensor("v_cache_39_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_39_squeeze_mask_0 = const()[name = tensor("v_cache_39_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_39_cast_fp16 = slice_by_index(begin = v_cache_39_begin_0, end = v_cache_39_end_0, end_mask = v_cache_39_end_mask_0, squeeze_mask = v_cache_39_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_39_cast_fp16")]; tensor var_3382 = const()[name = tensor("op_3382"), val = tensor(-1)]; tensor var_3381_promoted = const()[name = tensor("op_3381_promoted"), val = tensor(0x1p+1)]; tensor x_761_cast_fp16_to_fp32 = cast(dtype = x_761_cast_fp16_to_fp32_dtype_0, x = x_761_cast_fp16)[name = tensor("cast_344")]; tensor var_3391 = pow(x = x_761_cast_fp16_to_fp32, y = var_3381_promoted)[name = tensor("op_3391")]; tensor var_153_axes_0 = const()[name = tensor("var_153_axes_0"), val = tensor([-1])]; tensor var_153_keep_dims_0 = const()[name = tensor("var_153_keep_dims_0"), val = tensor(true)]; tensor var_153 = reduce_mean(axes = var_153_axes_0, keep_dims = var_153_keep_dims_0, x = var_3391)[name = tensor("var_153")]; tensor var_153_to_fp16_dtype_0 = const()[name = tensor("var_153_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3395_to_fp16 = const()[name = tensor("op_3395_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_153_to_fp16 = cast(dtype = var_153_to_fp16_dtype_0, x = var_153)[name = tensor("cast_343")]; tensor var_3396_cast_fp16 = add(x = var_153_to_fp16, y = var_3395_to_fp16)[name = tensor("op_3396_cast_fp16")]; tensor var_3396_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3396_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3397_epsilon_0 = const()[name = tensor("op_3397_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3396_cast_fp16_to_fp32 = cast(dtype = var_3396_cast_fp16_to_fp32_dtype_0, x = var_3396_cast_fp16)[name = tensor("cast_342")]; tensor var_3397 = rsqrt(epsilon = var_3397_epsilon_0, x = var_3396_cast_fp16_to_fp32)[name = tensor("op_3397")]; tensor var_3397_to_fp16_dtype_0 = const()[name = tensor("op_3397_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3397_to_fp16 = cast(dtype = var_3397_to_fp16_dtype_0, x = var_3397)[name = tensor("cast_341")]; tensor x_767_cast_fp16 = mul(x = x_761_cast_fp16, y = var_3397_to_fp16)[name = tensor("x_767_cast_fp16")]; tensor layers_19_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_19_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(391865472)))]; tensor x_769_cast_fp16 = mul(x = layers_19_input_layernorm_weight_to_fp16, y = x_767_cast_fp16)[name = tensor("x_769_cast_fp16")]; tensor layers_19_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_19_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(391866560)))]; tensor linear_133_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_19_self_attn_q_proj_weight_to_fp16, x = x_769_cast_fp16)[name = tensor("linear_133_cast_fp16")]; tensor var_3411 = const()[name = tensor("op_3411"), val = tensor([1, 1, 12, 128])]; tensor x_771_cast_fp16 = reshape(shape = var_3411, x = linear_133_cast_fp16)[name = tensor("x_771_cast_fp16")]; tensor x_771_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_771_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3381_promoted_1 = const()[name = tensor("op_3381_promoted_1"), val = tensor(0x1p+1)]; tensor x_771_cast_fp16_to_fp32 = cast(dtype = x_771_cast_fp16_to_fp32_dtype_0, x = x_771_cast_fp16)[name = tensor("cast_340")]; tensor var_3415 = pow(x = x_771_cast_fp16_to_fp32, y = var_3381_promoted_1)[name = tensor("op_3415")]; tensor var_155_axes_0 = const()[name = tensor("var_155_axes_0"), val = tensor([-1])]; tensor var_155_keep_dims_0 = const()[name = tensor("var_155_keep_dims_0"), val = tensor(true)]; tensor var_155 = reduce_mean(axes = var_155_axes_0, keep_dims = var_155_keep_dims_0, x = var_3415)[name = tensor("var_155")]; tensor var_155_to_fp16_dtype_0 = const()[name = tensor("var_155_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3419_to_fp16 = const()[name = tensor("op_3419_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_155_to_fp16 = cast(dtype = var_155_to_fp16_dtype_0, x = var_155)[name = tensor("cast_339")]; tensor var_3420_cast_fp16 = add(x = var_155_to_fp16, y = var_3419_to_fp16)[name = tensor("op_3420_cast_fp16")]; tensor var_3420_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3420_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3421_epsilon_0 = const()[name = tensor("op_3421_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3420_cast_fp16_to_fp32 = cast(dtype = var_3420_cast_fp16_to_fp32_dtype_0, x = var_3420_cast_fp16)[name = tensor("cast_338")]; tensor var_3421 = rsqrt(epsilon = var_3421_epsilon_0, x = var_3420_cast_fp16_to_fp32)[name = tensor("op_3421")]; tensor var_3421_to_fp16_dtype_0 = const()[name = tensor("op_3421_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3421_to_fp16 = cast(dtype = var_3421_to_fp16_dtype_0, x = var_3421)[name = tensor("cast_337")]; tensor x_777_cast_fp16 = mul(x = x_771_cast_fp16, y = var_3421_to_fp16)[name = tensor("x_777_cast_fp16")]; tensor layers_19_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_19_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(393439488)))]; tensor var_3423_cast_fp16 = mul(x = layers_19_self_attn_q_norm_weight_to_fp16, y = x_777_cast_fp16)[name = tensor("op_3423_cast_fp16")]; tensor q_77_perm_0 = const()[name = tensor("q_77_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_19_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_19_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(393439808)))]; tensor linear_134_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_19_self_attn_k_proj_weight_to_fp16, x = x_769_cast_fp16)[name = tensor("linear_134_cast_fp16")]; tensor var_3427 = const()[name = tensor("op_3427"), val = tensor([1, 1, 4, 128])]; tensor x_779_cast_fp16 = reshape(shape = var_3427, x = linear_134_cast_fp16)[name = tensor("x_779_cast_fp16")]; tensor x_779_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_779_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3381_promoted_2 = const()[name = tensor("op_3381_promoted_2"), val = tensor(0x1p+1)]; tensor x_779_cast_fp16_to_fp32 = cast(dtype = x_779_cast_fp16_to_fp32_dtype_0, x = x_779_cast_fp16)[name = tensor("cast_336")]; tensor var_3431 = pow(x = x_779_cast_fp16_to_fp32, y = var_3381_promoted_2)[name = tensor("op_3431")]; tensor var_157_axes_0 = const()[name = tensor("var_157_axes_0"), val = tensor([-1])]; tensor var_157_keep_dims_0 = const()[name = tensor("var_157_keep_dims_0"), val = tensor(true)]; tensor var_157 = reduce_mean(axes = var_157_axes_0, keep_dims = var_157_keep_dims_0, x = var_3431)[name = tensor("var_157")]; tensor var_157_to_fp16_dtype_0 = const()[name = tensor("var_157_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3435_to_fp16 = const()[name = tensor("op_3435_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_157_to_fp16 = cast(dtype = var_157_to_fp16_dtype_0, x = var_157)[name = tensor("cast_335")]; tensor var_3436_cast_fp16 = add(x = var_157_to_fp16, y = var_3435_to_fp16)[name = tensor("op_3436_cast_fp16")]; tensor var_3436_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3436_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3437_epsilon_0 = const()[name = tensor("op_3437_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3436_cast_fp16_to_fp32 = cast(dtype = var_3436_cast_fp16_to_fp32_dtype_0, x = var_3436_cast_fp16)[name = tensor("cast_334")]; tensor var_3437 = rsqrt(epsilon = var_3437_epsilon_0, x = var_3436_cast_fp16_to_fp32)[name = tensor("op_3437")]; tensor var_3437_to_fp16_dtype_0 = const()[name = tensor("op_3437_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3437_to_fp16 = cast(dtype = var_3437_to_fp16_dtype_0, x = var_3437)[name = tensor("cast_333")]; tensor x_785_cast_fp16 = mul(x = x_779_cast_fp16, y = var_3437_to_fp16)[name = tensor("x_785_cast_fp16")]; tensor layers_19_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_19_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(393964160)))]; tensor var_3439_cast_fp16 = mul(x = layers_19_self_attn_k_norm_weight_to_fp16, y = x_785_cast_fp16)[name = tensor("op_3439_cast_fp16")]; tensor k_77_perm_0 = const()[name = tensor("k_77_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_19_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_19_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(393964480)))]; tensor linear_135_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_19_self_attn_v_proj_weight_to_fp16, x = x_769_cast_fp16)[name = tensor("linear_135_cast_fp16")]; tensor var_3443 = const()[name = tensor("op_3443"), val = tensor([1, 1, 4, 128])]; tensor var_3444_cast_fp16 = reshape(shape = var_3443, x = linear_135_cast_fp16)[name = tensor("op_3444_cast_fp16")]; tensor v_39_perm_0 = const()[name = tensor("v_39_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_77_cast_fp16 = transpose(perm = q_77_perm_0, x = var_3423_cast_fp16)[name = tensor("transpose_35")]; tensor var_3448_cast_fp16 = mul(x = q_77_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3448_cast_fp16")]; tensor x1_77_begin_0 = const()[name = tensor("x1_77_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_77_end_0 = const()[name = tensor("x1_77_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_77_end_mask_0 = const()[name = tensor("x1_77_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_77_cast_fp16 = slice_by_index(begin = x1_77_begin_0, end = x1_77_end_0, end_mask = x1_77_end_mask_0, x = q_77_cast_fp16)[name = tensor("x1_77_cast_fp16")]; tensor x2_77_begin_0 = const()[name = tensor("x2_77_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_77_end_0 = const()[name = tensor("x2_77_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_77_end_mask_0 = const()[name = tensor("x2_77_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_77_cast_fp16 = slice_by_index(begin = x2_77_begin_0, end = x2_77_end_0, end_mask = x2_77_end_mask_0, x = q_77_cast_fp16)[name = tensor("x2_77_cast_fp16")]; tensor const_39_promoted_to_fp16 = const()[name = tensor("const_39_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3451_cast_fp16 = mul(x = x2_77_cast_fp16, y = const_39_promoted_to_fp16)[name = tensor("op_3451_cast_fp16")]; tensor var_3453_interleave_0 = const()[name = tensor("op_3453_interleave_0"), val = tensor(false)]; tensor var_3453_cast_fp16 = concat(axis = var_3382, interleave = var_3453_interleave_0, values = (var_3451_cast_fp16, x1_77_cast_fp16))[name = tensor("op_3453_cast_fp16")]; tensor var_3454_cast_fp16 = mul(x = var_3453_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3454_cast_fp16")]; tensor q_79_cast_fp16 = add(x = var_3448_cast_fp16, y = var_3454_cast_fp16)[name = tensor("q_79_cast_fp16")]; tensor k_77_cast_fp16 = transpose(perm = k_77_perm_0, x = var_3439_cast_fp16)[name = tensor("transpose_34")]; tensor var_3456_cast_fp16 = mul(x = k_77_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3456_cast_fp16")]; tensor x1_79_begin_0 = const()[name = tensor("x1_79_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_79_end_0 = const()[name = tensor("x1_79_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_79_end_mask_0 = const()[name = tensor("x1_79_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_79_cast_fp16 = slice_by_index(begin = x1_79_begin_0, end = x1_79_end_0, end_mask = x1_79_end_mask_0, x = k_77_cast_fp16)[name = tensor("x1_79_cast_fp16")]; tensor x2_79_begin_0 = const()[name = tensor("x2_79_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_79_end_0 = const()[name = tensor("x2_79_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_79_end_mask_0 = const()[name = tensor("x2_79_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_79_cast_fp16 = slice_by_index(begin = x2_79_begin_0, end = x2_79_end_0, end_mask = x2_79_end_mask_0, x = k_77_cast_fp16)[name = tensor("x2_79_cast_fp16")]; tensor const_40_promoted_to_fp16 = const()[name = tensor("const_40_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3459_cast_fp16 = mul(x = x2_79_cast_fp16, y = const_40_promoted_to_fp16)[name = tensor("op_3459_cast_fp16")]; tensor var_3461_interleave_0 = const()[name = tensor("op_3461_interleave_0"), val = tensor(false)]; tensor var_3461_cast_fp16 = concat(axis = var_3382, interleave = var_3461_interleave_0, values = (var_3459_cast_fp16, x1_79_cast_fp16))[name = tensor("op_3461_cast_fp16")]; tensor var_3462_cast_fp16 = mul(x = var_3461_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3462_cast_fp16")]; tensor k_79_cast_fp16 = add(x = var_3456_cast_fp16, y = var_3462_cast_fp16)[name = tensor("k_79_cast_fp16")]; tensor var_3465_cast_fp16 = mul(x = k_cache_39_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3465_cast_fp16")]; tensor var_3466_cast_fp16 = mul(x = k_79_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3466_cast_fp16")]; tensor k_full_39_cast_fp16 = add(x = var_3465_cast_fp16, y = var_3466_cast_fp16)[name = tensor("k_full_39_cast_fp16")]; tensor var_3469_cast_fp16 = mul(x = v_cache_39_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3469_cast_fp16")]; tensor v_39_cast_fp16 = transpose(perm = v_39_perm_0, x = var_3444_cast_fp16)[name = tensor("transpose_33")]; tensor var_3470_cast_fp16 = mul(x = v_39_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3470_cast_fp16")]; tensor v_full_39_cast_fp16 = add(x = var_3469_cast_fp16, y = var_3470_cast_fp16)[name = tensor("v_full_39_cast_fp16")]; tensor var_3472_axes_0 = const()[name = tensor("op_3472_axes_0"), val = tensor([2])]; tensor var_3472_cast_fp16 = expand_dims(axes = var_3472_axes_0, x = k_full_39_cast_fp16)[name = tensor("op_3472_cast_fp16")]; tensor var_3474_reps_0 = const()[name = tensor("op_3474_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3474_cast_fp16 = tile(reps = var_3474_reps_0, x = var_3472_cast_fp16)[name = tensor("op_3474_cast_fp16")]; tensor var_3475 = const()[name = tensor("op_3475"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_39_cast_fp16 = reshape(shape = var_3475, x = var_3474_cast_fp16)[name = tensor("k_rep_39_cast_fp16")]; tensor var_3477_axes_0 = const()[name = tensor("op_3477_axes_0"), val = tensor([2])]; tensor var_3477_cast_fp16 = expand_dims(axes = var_3477_axes_0, x = v_full_39_cast_fp16)[name = tensor("op_3477_cast_fp16")]; tensor var_3479_reps_0 = const()[name = tensor("op_3479_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3479_cast_fp16 = tile(reps = var_3479_reps_0, x = var_3477_cast_fp16)[name = tensor("op_3479_cast_fp16")]; tensor var_3480 = const()[name = tensor("op_3480"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_39_cast_fp16 = reshape(shape = var_3480, x = var_3479_cast_fp16)[name = tensor("v_rep_39_cast_fp16")]; tensor var_3483_transpose_x_1 = const()[name = tensor("op_3483_transpose_x_1"), val = tensor(false)]; tensor var_3483_transpose_y_1 = const()[name = tensor("op_3483_transpose_y_1"), val = tensor(true)]; tensor var_3483_cast_fp16 = matmul(transpose_x = var_3483_transpose_x_1, transpose_y = var_3483_transpose_y_1, x = q_79_cast_fp16, y = k_rep_39_cast_fp16)[name = tensor("op_3483_cast_fp16")]; tensor var_3484_to_fp16 = const()[name = tensor("op_3484_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_77_cast_fp16 = mul(x = var_3483_cast_fp16, y = var_3484_to_fp16)[name = tensor("attn_77_cast_fp16")]; tensor input_79_cast_fp16 = add(x = attn_77_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_79_cast_fp16")]; tensor input_79_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_79_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_79_cast_fp16_to_fp32 = cast(dtype = input_79_cast_fp16_to_fp32_dtype_0, x = input_79_cast_fp16)[name = tensor("cast_332")]; tensor attn_79 = softmax(axis = var_3382, x = input_79_cast_fp16_to_fp32)[name = tensor("attn_79")]; tensor out_39_transpose_x_0 = const()[name = tensor("out_39_transpose_x_0"), val = tensor(false)]; tensor out_39_transpose_y_0 = const()[name = tensor("out_39_transpose_y_0"), val = tensor(false)]; tensor attn_79_to_fp16_dtype_0 = const()[name = tensor("attn_79_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_79_to_fp16 = cast(dtype = attn_79_to_fp16_dtype_0, x = attn_79)[name = tensor("cast_331")]; tensor out_39_cast_fp16 = matmul(transpose_x = out_39_transpose_x_0, transpose_y = out_39_transpose_y_0, x = attn_79_to_fp16, y = v_rep_39_cast_fp16)[name = tensor("out_39_cast_fp16")]; tensor var_3489_perm_0 = const()[name = tensor("op_3489_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3491 = const()[name = tensor("op_3491"), val = tensor([1, 1, 1536])]; tensor var_3489_cast_fp16 = transpose(perm = var_3489_perm_0, x = out_39_cast_fp16)[name = tensor("transpose_32")]; tensor x_787_cast_fp16 = reshape(shape = var_3491, x = var_3489_cast_fp16)[name = tensor("x_787_cast_fp16")]; tensor layers_19_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_19_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(394488832)))]; tensor linear_136_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_19_self_attn_o_proj_weight_to_fp16, x = x_787_cast_fp16)[name = tensor("linear_136_cast_fp16")]; tensor x_789_cast_fp16 = add(x = x_761_cast_fp16, y = linear_136_cast_fp16)[name = tensor("x_789_cast_fp16")]; tensor x_789_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_789_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3381_promoted_3 = const()[name = tensor("op_3381_promoted_3"), val = tensor(0x1p+1)]; tensor x_789_cast_fp16_to_fp32 = cast(dtype = x_789_cast_fp16_to_fp32_dtype_0, x = x_789_cast_fp16)[name = tensor("cast_330")]; tensor var_3502 = pow(x = x_789_cast_fp16_to_fp32, y = var_3381_promoted_3)[name = tensor("op_3502")]; tensor var_159_axes_0 = const()[name = tensor("var_159_axes_0"), val = tensor([-1])]; tensor var_159_keep_dims_0 = const()[name = tensor("var_159_keep_dims_0"), val = tensor(true)]; tensor var_159 = reduce_mean(axes = var_159_axes_0, keep_dims = var_159_keep_dims_0, x = var_3502)[name = tensor("var_159")]; tensor var_159_to_fp16_dtype_0 = const()[name = tensor("var_159_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3506_to_fp16 = const()[name = tensor("op_3506_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_159_to_fp16 = cast(dtype = var_159_to_fp16_dtype_0, x = var_159)[name = tensor("cast_329")]; tensor var_3507_cast_fp16 = add(x = var_159_to_fp16, y = var_3506_to_fp16)[name = tensor("op_3507_cast_fp16")]; tensor var_3507_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3507_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3508_epsilon_0 = const()[name = tensor("op_3508_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3507_cast_fp16_to_fp32 = cast(dtype = var_3507_cast_fp16_to_fp32_dtype_0, x = var_3507_cast_fp16)[name = tensor("cast_328")]; tensor var_3508 = rsqrt(epsilon = var_3508_epsilon_0, x = var_3507_cast_fp16_to_fp32)[name = tensor("op_3508")]; tensor var_3508_to_fp16_dtype_0 = const()[name = tensor("op_3508_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3508_to_fp16 = cast(dtype = var_3508_to_fp16_dtype_0, x = var_3508)[name = tensor("cast_327")]; tensor x_795_cast_fp16 = mul(x = x_789_cast_fp16, y = var_3508_to_fp16)[name = tensor("x_795_cast_fp16")]; tensor layers_19_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_19_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(396061760)))]; tensor x_797_cast_fp16 = mul(x = layers_19_post_attention_layernorm_weight_to_fp16, y = x_795_cast_fp16)[name = tensor("x_797_cast_fp16")]; tensor layers_19_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_19_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(396062848)))]; tensor linear_137_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_19_mlp_gate_proj_weight_to_fp16, x = x_797_cast_fp16)[name = tensor("linear_137_cast_fp16")]; tensor var_3519_cast_fp16 = silu(x = linear_137_cast_fp16)[name = tensor("op_3519_cast_fp16")]; tensor layers_19_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_19_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(397635776)))]; tensor linear_138_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_19_mlp_up_proj_weight_to_fp16, x = x_797_cast_fp16)[name = tensor("linear_138_cast_fp16")]; tensor x_799_cast_fp16 = mul(x = var_3519_cast_fp16, y = linear_138_cast_fp16)[name = tensor("x_799_cast_fp16")]; tensor layers_19_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_19_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(399208704)))]; tensor linear_139_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_19_mlp_down_proj_weight_to_fp16, x = x_799_cast_fp16)[name = tensor("linear_139_cast_fp16")]; tensor x_801_cast_fp16 = add(x = x_789_cast_fp16, y = linear_139_cast_fp16)[name = tensor("x_801_cast_fp16")]; tensor x_801_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_801_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_41_begin_0 = const()[name = tensor("k_cache_41_begin_0"), val = tensor([20, 0, 0, 0, 0])]; tensor k_cache_41_end_0 = const()[name = tensor("k_cache_41_end_0"), val = tensor([21, 1, 4, 2048, 128])]; tensor k_cache_41_end_mask_0 = const()[name = tensor("k_cache_41_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_41_squeeze_mask_0 = const()[name = tensor("k_cache_41_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_41_cast_fp16 = slice_by_index(begin = k_cache_41_begin_0, end = k_cache_41_end_0, end_mask = k_cache_41_end_mask_0, squeeze_mask = k_cache_41_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_41_cast_fp16")]; tensor v_cache_41_begin_0 = const()[name = tensor("v_cache_41_begin_0"), val = tensor([20, 0, 0, 0, 0])]; tensor v_cache_41_end_0 = const()[name = tensor("v_cache_41_end_0"), val = tensor([21, 1, 4, 2048, 128])]; tensor v_cache_41_end_mask_0 = const()[name = tensor("v_cache_41_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_41_squeeze_mask_0 = const()[name = tensor("v_cache_41_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_41_cast_fp16 = slice_by_index(begin = v_cache_41_begin_0, end = v_cache_41_end_0, end_mask = v_cache_41_end_mask_0, squeeze_mask = v_cache_41_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_41_cast_fp16")]; tensor var_3551 = const()[name = tensor("op_3551"), val = tensor(-1)]; tensor var_3550_promoted = const()[name = tensor("op_3550_promoted"), val = tensor(0x1p+1)]; tensor x_801_cast_fp16_to_fp32 = cast(dtype = x_801_cast_fp16_to_fp32_dtype_0, x = x_801_cast_fp16)[name = tensor("cast_326")]; tensor var_3560 = pow(x = x_801_cast_fp16_to_fp32, y = var_3550_promoted)[name = tensor("op_3560")]; tensor var_161_axes_0 = const()[name = tensor("var_161_axes_0"), val = tensor([-1])]; tensor var_161_keep_dims_0 = const()[name = tensor("var_161_keep_dims_0"), val = tensor(true)]; tensor var_161 = reduce_mean(axes = var_161_axes_0, keep_dims = var_161_keep_dims_0, x = var_3560)[name = tensor("var_161")]; tensor var_161_to_fp16_dtype_0 = const()[name = tensor("var_161_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3564_to_fp16 = const()[name = tensor("op_3564_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_161_to_fp16 = cast(dtype = var_161_to_fp16_dtype_0, x = var_161)[name = tensor("cast_325")]; tensor var_3565_cast_fp16 = add(x = var_161_to_fp16, y = var_3564_to_fp16)[name = tensor("op_3565_cast_fp16")]; tensor var_3565_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3565_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3566_epsilon_0 = const()[name = tensor("op_3566_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3565_cast_fp16_to_fp32 = cast(dtype = var_3565_cast_fp16_to_fp32_dtype_0, x = var_3565_cast_fp16)[name = tensor("cast_324")]; tensor var_3566 = rsqrt(epsilon = var_3566_epsilon_0, x = var_3565_cast_fp16_to_fp32)[name = tensor("op_3566")]; tensor var_3566_to_fp16_dtype_0 = const()[name = tensor("op_3566_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3566_to_fp16 = cast(dtype = var_3566_to_fp16_dtype_0, x = var_3566)[name = tensor("cast_323")]; tensor x_807_cast_fp16 = mul(x = x_801_cast_fp16, y = var_3566_to_fp16)[name = tensor("x_807_cast_fp16")]; tensor layers_20_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_20_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(400781632)))]; tensor x_809_cast_fp16 = mul(x = layers_20_input_layernorm_weight_to_fp16, y = x_807_cast_fp16)[name = tensor("x_809_cast_fp16")]; tensor layers_20_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_20_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(400782720)))]; tensor linear_140_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_20_self_attn_q_proj_weight_to_fp16, x = x_809_cast_fp16)[name = tensor("linear_140_cast_fp16")]; tensor var_3580 = const()[name = tensor("op_3580"), val = tensor([1, 1, 12, 128])]; tensor x_811_cast_fp16 = reshape(shape = var_3580, x = linear_140_cast_fp16)[name = tensor("x_811_cast_fp16")]; tensor x_811_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_811_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3550_promoted_1 = const()[name = tensor("op_3550_promoted_1"), val = tensor(0x1p+1)]; tensor x_811_cast_fp16_to_fp32 = cast(dtype = x_811_cast_fp16_to_fp32_dtype_0, x = x_811_cast_fp16)[name = tensor("cast_322")]; tensor var_3584 = pow(x = x_811_cast_fp16_to_fp32, y = var_3550_promoted_1)[name = tensor("op_3584")]; tensor var_163_axes_0 = const()[name = tensor("var_163_axes_0"), val = tensor([-1])]; tensor var_163_keep_dims_0 = const()[name = tensor("var_163_keep_dims_0"), val = tensor(true)]; tensor var_163 = reduce_mean(axes = var_163_axes_0, keep_dims = var_163_keep_dims_0, x = var_3584)[name = tensor("var_163")]; tensor var_163_to_fp16_dtype_0 = const()[name = tensor("var_163_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3588_to_fp16 = const()[name = tensor("op_3588_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_163_to_fp16_0 = cast(dtype = var_163_to_fp16_dtype_0, x = var_163)[name = tensor("cast_321")]; tensor var_3589_cast_fp16 = add(x = var_163_to_fp16_0, y = var_3588_to_fp16)[name = tensor("op_3589_cast_fp16")]; tensor var_3589_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3589_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3590_epsilon_0 = const()[name = tensor("op_3590_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3589_cast_fp16_to_fp32 = cast(dtype = var_3589_cast_fp16_to_fp32_dtype_0, x = var_3589_cast_fp16)[name = tensor("cast_320")]; tensor var_3590 = rsqrt(epsilon = var_3590_epsilon_0, x = var_3589_cast_fp16_to_fp32)[name = tensor("op_3590")]; tensor var_3590_to_fp16_dtype_0 = const()[name = tensor("op_3590_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3590_to_fp16 = cast(dtype = var_3590_to_fp16_dtype_0, x = var_3590)[name = tensor("cast_319")]; tensor x_817_cast_fp16 = mul(x = x_811_cast_fp16, y = var_3590_to_fp16)[name = tensor("x_817_cast_fp16")]; tensor layers_20_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_20_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402355648)))]; tensor var_3592_cast_fp16 = mul(x = layers_20_self_attn_q_norm_weight_to_fp16, y = x_817_cast_fp16)[name = tensor("op_3592_cast_fp16")]; tensor q_81_perm_0 = const()[name = tensor("q_81_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_20_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_20_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402355968)))]; tensor linear_141_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_20_self_attn_k_proj_weight_to_fp16, x = x_809_cast_fp16)[name = tensor("linear_141_cast_fp16")]; tensor var_3596 = const()[name = tensor("op_3596"), val = tensor([1, 1, 4, 128])]; tensor x_819_cast_fp16 = reshape(shape = var_3596, x = linear_141_cast_fp16)[name = tensor("x_819_cast_fp16")]; tensor x_819_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_819_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3550_promoted_2 = const()[name = tensor("op_3550_promoted_2"), val = tensor(0x1p+1)]; tensor x_819_cast_fp16_to_fp32 = cast(dtype = x_819_cast_fp16_to_fp32_dtype_0, x = x_819_cast_fp16)[name = tensor("cast_318")]; tensor var_3600 = pow(x = x_819_cast_fp16_to_fp32, y = var_3550_promoted_2)[name = tensor("op_3600")]; tensor var_165_axes_0 = const()[name = tensor("var_165_axes_0"), val = tensor([-1])]; tensor var_165_keep_dims_0 = const()[name = tensor("var_165_keep_dims_0"), val = tensor(true)]; tensor var_165 = reduce_mean(axes = var_165_axes_0, keep_dims = var_165_keep_dims_0, x = var_3600)[name = tensor("var_165")]; tensor var_165_to_fp16_dtype_0 = const()[name = tensor("var_165_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3604_to_fp16 = const()[name = tensor("op_3604_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_165_to_fp16 = cast(dtype = var_165_to_fp16_dtype_0, x = var_165)[name = tensor("cast_317")]; tensor var_3605_cast_fp16 = add(x = var_165_to_fp16, y = var_3604_to_fp16)[name = tensor("op_3605_cast_fp16")]; tensor var_3605_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3605_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3606_epsilon_0 = const()[name = tensor("op_3606_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3605_cast_fp16_to_fp32 = cast(dtype = var_3605_cast_fp16_to_fp32_dtype_0, x = var_3605_cast_fp16)[name = tensor("cast_316")]; tensor var_3606 = rsqrt(epsilon = var_3606_epsilon_0, x = var_3605_cast_fp16_to_fp32)[name = tensor("op_3606")]; tensor var_3606_to_fp16_dtype_0 = const()[name = tensor("op_3606_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3606_to_fp16 = cast(dtype = var_3606_to_fp16_dtype_0, x = var_3606)[name = tensor("cast_315")]; tensor x_825_cast_fp16 = mul(x = x_819_cast_fp16, y = var_3606_to_fp16)[name = tensor("x_825_cast_fp16")]; tensor layers_20_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_20_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402880320)))]; tensor var_3608_cast_fp16 = mul(x = layers_20_self_attn_k_norm_weight_to_fp16, y = x_825_cast_fp16)[name = tensor("op_3608_cast_fp16")]; tensor k_81_perm_0 = const()[name = tensor("k_81_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_20_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_20_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402880640)))]; tensor linear_142_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_20_self_attn_v_proj_weight_to_fp16, x = x_809_cast_fp16)[name = tensor("linear_142_cast_fp16")]; tensor var_3612 = const()[name = tensor("op_3612"), val = tensor([1, 1, 4, 128])]; tensor var_3613_cast_fp16 = reshape(shape = var_3612, x = linear_142_cast_fp16)[name = tensor("op_3613_cast_fp16")]; tensor v_41_perm_0 = const()[name = tensor("v_41_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_81_cast_fp16 = transpose(perm = q_81_perm_0, x = var_3592_cast_fp16)[name = tensor("transpose_31")]; tensor var_3617_cast_fp16 = mul(x = q_81_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3617_cast_fp16")]; tensor x1_81_begin_0 = const()[name = tensor("x1_81_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_81_end_0 = const()[name = tensor("x1_81_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_81_end_mask_0 = const()[name = tensor("x1_81_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_81_cast_fp16 = slice_by_index(begin = x1_81_begin_0, end = x1_81_end_0, end_mask = x1_81_end_mask_0, x = q_81_cast_fp16)[name = tensor("x1_81_cast_fp16")]; tensor x2_81_begin_0 = const()[name = tensor("x2_81_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_81_end_0 = const()[name = tensor("x2_81_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_81_end_mask_0 = const()[name = tensor("x2_81_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_81_cast_fp16 = slice_by_index(begin = x2_81_begin_0, end = x2_81_end_0, end_mask = x2_81_end_mask_0, x = q_81_cast_fp16)[name = tensor("x2_81_cast_fp16")]; tensor const_41_promoted_to_fp16 = const()[name = tensor("const_41_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3620_cast_fp16 = mul(x = x2_81_cast_fp16, y = const_41_promoted_to_fp16)[name = tensor("op_3620_cast_fp16")]; tensor var_3622_interleave_0 = const()[name = tensor("op_3622_interleave_0"), val = tensor(false)]; tensor var_3622_cast_fp16 = concat(axis = var_3551, interleave = var_3622_interleave_0, values = (var_3620_cast_fp16, x1_81_cast_fp16))[name = tensor("op_3622_cast_fp16")]; tensor var_3623_cast_fp16 = mul(x = var_3622_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3623_cast_fp16")]; tensor q_83_cast_fp16 = add(x = var_3617_cast_fp16, y = var_3623_cast_fp16)[name = tensor("q_83_cast_fp16")]; tensor k_81_cast_fp16 = transpose(perm = k_81_perm_0, x = var_3608_cast_fp16)[name = tensor("transpose_30")]; tensor var_3625_cast_fp16 = mul(x = k_81_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3625_cast_fp16")]; tensor x1_83_begin_0 = const()[name = tensor("x1_83_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_83_end_0 = const()[name = tensor("x1_83_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_83_end_mask_0 = const()[name = tensor("x1_83_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_83_cast_fp16 = slice_by_index(begin = x1_83_begin_0, end = x1_83_end_0, end_mask = x1_83_end_mask_0, x = k_81_cast_fp16)[name = tensor("x1_83_cast_fp16")]; tensor x2_83_begin_0 = const()[name = tensor("x2_83_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_83_end_0 = const()[name = tensor("x2_83_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_83_end_mask_0 = const()[name = tensor("x2_83_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_83_cast_fp16 = slice_by_index(begin = x2_83_begin_0, end = x2_83_end_0, end_mask = x2_83_end_mask_0, x = k_81_cast_fp16)[name = tensor("x2_83_cast_fp16")]; tensor const_42_promoted_to_fp16 = const()[name = tensor("const_42_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3628_cast_fp16 = mul(x = x2_83_cast_fp16, y = const_42_promoted_to_fp16)[name = tensor("op_3628_cast_fp16")]; tensor var_3630_interleave_0 = const()[name = tensor("op_3630_interleave_0"), val = tensor(false)]; tensor var_3630_cast_fp16 = concat(axis = var_3551, interleave = var_3630_interleave_0, values = (var_3628_cast_fp16, x1_83_cast_fp16))[name = tensor("op_3630_cast_fp16")]; tensor var_3631_cast_fp16 = mul(x = var_3630_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3631_cast_fp16")]; tensor k_83_cast_fp16 = add(x = var_3625_cast_fp16, y = var_3631_cast_fp16)[name = tensor("k_83_cast_fp16")]; tensor var_3634_cast_fp16 = mul(x = k_cache_41_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3634_cast_fp16")]; tensor var_3635_cast_fp16 = mul(x = k_83_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3635_cast_fp16")]; tensor k_full_41_cast_fp16 = add(x = var_3634_cast_fp16, y = var_3635_cast_fp16)[name = tensor("k_full_41_cast_fp16")]; tensor var_3638_cast_fp16 = mul(x = v_cache_41_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3638_cast_fp16")]; tensor v_41_cast_fp16 = transpose(perm = v_41_perm_0, x = var_3613_cast_fp16)[name = tensor("transpose_29")]; tensor var_3639_cast_fp16 = mul(x = v_41_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3639_cast_fp16")]; tensor v_full_41_cast_fp16 = add(x = var_3638_cast_fp16, y = var_3639_cast_fp16)[name = tensor("v_full_41_cast_fp16")]; tensor var_3641_axes_0 = const()[name = tensor("op_3641_axes_0"), val = tensor([2])]; tensor var_3641_cast_fp16 = expand_dims(axes = var_3641_axes_0, x = k_full_41_cast_fp16)[name = tensor("op_3641_cast_fp16")]; tensor var_3643_reps_0 = const()[name = tensor("op_3643_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3643_cast_fp16 = tile(reps = var_3643_reps_0, x = var_3641_cast_fp16)[name = tensor("op_3643_cast_fp16")]; tensor var_3644 = const()[name = tensor("op_3644"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_41_cast_fp16 = reshape(shape = var_3644, x = var_3643_cast_fp16)[name = tensor("k_rep_41_cast_fp16")]; tensor var_3646_axes_0 = const()[name = tensor("op_3646_axes_0"), val = tensor([2])]; tensor var_3646_cast_fp16 = expand_dims(axes = var_3646_axes_0, x = v_full_41_cast_fp16)[name = tensor("op_3646_cast_fp16")]; tensor var_3648_reps_0 = const()[name = tensor("op_3648_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3648_cast_fp16 = tile(reps = var_3648_reps_0, x = var_3646_cast_fp16)[name = tensor("op_3648_cast_fp16")]; tensor var_3649 = const()[name = tensor("op_3649"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_41_cast_fp16 = reshape(shape = var_3649, x = var_3648_cast_fp16)[name = tensor("v_rep_41_cast_fp16")]; tensor var_3652_transpose_x_1 = const()[name = tensor("op_3652_transpose_x_1"), val = tensor(false)]; tensor var_3652_transpose_y_1 = const()[name = tensor("op_3652_transpose_y_1"), val = tensor(true)]; tensor var_3652_cast_fp16 = matmul(transpose_x = var_3652_transpose_x_1, transpose_y = var_3652_transpose_y_1, x = q_83_cast_fp16, y = k_rep_41_cast_fp16)[name = tensor("op_3652_cast_fp16")]; tensor var_3653_to_fp16 = const()[name = tensor("op_3653_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_81_cast_fp16 = mul(x = var_3652_cast_fp16, y = var_3653_to_fp16)[name = tensor("attn_81_cast_fp16")]; tensor input_83_cast_fp16 = add(x = attn_81_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_83_cast_fp16")]; tensor input_83_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_83_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_83_cast_fp16_to_fp32 = cast(dtype = input_83_cast_fp16_to_fp32_dtype_0, x = input_83_cast_fp16)[name = tensor("cast_314")]; tensor attn_83 = softmax(axis = var_3551, x = input_83_cast_fp16_to_fp32)[name = tensor("attn_83")]; tensor out_41_transpose_x_0 = const()[name = tensor("out_41_transpose_x_0"), val = tensor(false)]; tensor out_41_transpose_y_0 = const()[name = tensor("out_41_transpose_y_0"), val = tensor(false)]; tensor attn_83_to_fp16_dtype_0 = const()[name = tensor("attn_83_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_83_to_fp16 = cast(dtype = attn_83_to_fp16_dtype_0, x = attn_83)[name = tensor("cast_313")]; tensor out_41_cast_fp16 = matmul(transpose_x = out_41_transpose_x_0, transpose_y = out_41_transpose_y_0, x = attn_83_to_fp16, y = v_rep_41_cast_fp16)[name = tensor("out_41_cast_fp16")]; tensor var_3658_perm_0 = const()[name = tensor("op_3658_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3660 = const()[name = tensor("op_3660"), val = tensor([1, 1, 1536])]; tensor var_3658_cast_fp16 = transpose(perm = var_3658_perm_0, x = out_41_cast_fp16)[name = tensor("transpose_28")]; tensor x_827_cast_fp16 = reshape(shape = var_3660, x = var_3658_cast_fp16)[name = tensor("x_827_cast_fp16")]; tensor layers_20_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_20_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(403404992)))]; tensor linear_143_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_20_self_attn_o_proj_weight_to_fp16, x = x_827_cast_fp16)[name = tensor("linear_143_cast_fp16")]; tensor x_829_cast_fp16 = add(x = x_801_cast_fp16, y = linear_143_cast_fp16)[name = tensor("x_829_cast_fp16")]; tensor x_829_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_829_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3550_promoted_3 = const()[name = tensor("op_3550_promoted_3"), val = tensor(0x1p+1)]; tensor x_829_cast_fp16_to_fp32 = cast(dtype = x_829_cast_fp16_to_fp32_dtype_0, x = x_829_cast_fp16)[name = tensor("cast_312")]; tensor var_3671 = pow(x = x_829_cast_fp16_to_fp32, y = var_3550_promoted_3)[name = tensor("op_3671")]; tensor var_167_axes_0 = const()[name = tensor("var_167_axes_0"), val = tensor([-1])]; tensor var_167_keep_dims_0 = const()[name = tensor("var_167_keep_dims_0"), val = tensor(true)]; tensor var_167 = reduce_mean(axes = var_167_axes_0, keep_dims = var_167_keep_dims_0, x = var_3671)[name = tensor("var_167")]; tensor var_167_to_fp16_dtype_0 = const()[name = tensor("var_167_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3675_to_fp16 = const()[name = tensor("op_3675_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_167_to_fp16 = cast(dtype = var_167_to_fp16_dtype_0, x = var_167)[name = tensor("cast_311")]; tensor var_3676_cast_fp16 = add(x = var_167_to_fp16, y = var_3675_to_fp16)[name = tensor("op_3676_cast_fp16")]; tensor var_3676_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3676_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3677_epsilon_0 = const()[name = tensor("op_3677_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3676_cast_fp16_to_fp32 = cast(dtype = var_3676_cast_fp16_to_fp32_dtype_0, x = var_3676_cast_fp16)[name = tensor("cast_310")]; tensor var_3677 = rsqrt(epsilon = var_3677_epsilon_0, x = var_3676_cast_fp16_to_fp32)[name = tensor("op_3677")]; tensor var_3677_to_fp16_dtype_0 = const()[name = tensor("op_3677_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3677_to_fp16 = cast(dtype = var_3677_to_fp16_dtype_0, x = var_3677)[name = tensor("cast_309")]; tensor x_835_cast_fp16 = mul(x = x_829_cast_fp16, y = var_3677_to_fp16)[name = tensor("x_835_cast_fp16")]; tensor layers_20_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_20_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(404977920)))]; tensor x_837_cast_fp16 = mul(x = layers_20_post_attention_layernorm_weight_to_fp16, y = x_835_cast_fp16)[name = tensor("x_837_cast_fp16")]; tensor layers_20_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_20_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(404979008)))]; tensor linear_144_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_20_mlp_gate_proj_weight_to_fp16, x = x_837_cast_fp16)[name = tensor("linear_144_cast_fp16")]; tensor var_3688_cast_fp16 = silu(x = linear_144_cast_fp16)[name = tensor("op_3688_cast_fp16")]; tensor layers_20_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_20_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(406551936)))]; tensor linear_145_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_20_mlp_up_proj_weight_to_fp16, x = x_837_cast_fp16)[name = tensor("linear_145_cast_fp16")]; tensor x_839_cast_fp16 = mul(x = var_3688_cast_fp16, y = linear_145_cast_fp16)[name = tensor("x_839_cast_fp16")]; tensor layers_20_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_20_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(408124864)))]; tensor linear_146_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_20_mlp_down_proj_weight_to_fp16, x = x_839_cast_fp16)[name = tensor("linear_146_cast_fp16")]; tensor x_841_cast_fp16 = add(x = x_829_cast_fp16, y = linear_146_cast_fp16)[name = tensor("x_841_cast_fp16")]; tensor x_841_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_841_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_43_begin_0 = const()[name = tensor("k_cache_43_begin_0"), val = tensor([21, 0, 0, 0, 0])]; tensor k_cache_43_end_0 = const()[name = tensor("k_cache_43_end_0"), val = tensor([22, 1, 4, 2048, 128])]; tensor k_cache_43_end_mask_0 = const()[name = tensor("k_cache_43_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_43_squeeze_mask_0 = const()[name = tensor("k_cache_43_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_43_cast_fp16 = slice_by_index(begin = k_cache_43_begin_0, end = k_cache_43_end_0, end_mask = k_cache_43_end_mask_0, squeeze_mask = k_cache_43_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_43_cast_fp16")]; tensor v_cache_43_begin_0 = const()[name = tensor("v_cache_43_begin_0"), val = tensor([21, 0, 0, 0, 0])]; tensor v_cache_43_end_0 = const()[name = tensor("v_cache_43_end_0"), val = tensor([22, 1, 4, 2048, 128])]; tensor v_cache_43_end_mask_0 = const()[name = tensor("v_cache_43_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_43_squeeze_mask_0 = const()[name = tensor("v_cache_43_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_43_cast_fp16 = slice_by_index(begin = v_cache_43_begin_0, end = v_cache_43_end_0, end_mask = v_cache_43_end_mask_0, squeeze_mask = v_cache_43_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_43_cast_fp16")]; tensor var_3720 = const()[name = tensor("op_3720"), val = tensor(-1)]; tensor var_3719_promoted = const()[name = tensor("op_3719_promoted"), val = tensor(0x1p+1)]; tensor x_841_cast_fp16_to_fp32 = cast(dtype = x_841_cast_fp16_to_fp32_dtype_0, x = x_841_cast_fp16)[name = tensor("cast_308")]; tensor var_3729 = pow(x = x_841_cast_fp16_to_fp32, y = var_3719_promoted)[name = tensor("op_3729")]; tensor var_169_axes_0 = const()[name = tensor("var_169_axes_0"), val = tensor([-1])]; tensor var_169_keep_dims_0 = const()[name = tensor("var_169_keep_dims_0"), val = tensor(true)]; tensor var_169 = reduce_mean(axes = var_169_axes_0, keep_dims = var_169_keep_dims_0, x = var_3729)[name = tensor("var_169")]; tensor var_169_to_fp16_dtype_0 = const()[name = tensor("var_169_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3733_to_fp16 = const()[name = tensor("op_3733_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_169_to_fp16 = cast(dtype = var_169_to_fp16_dtype_0, x = var_169)[name = tensor("cast_307")]; tensor var_3734_cast_fp16 = add(x = var_169_to_fp16, y = var_3733_to_fp16)[name = tensor("op_3734_cast_fp16")]; tensor var_3734_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3734_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3735_epsilon_0 = const()[name = tensor("op_3735_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3734_cast_fp16_to_fp32 = cast(dtype = var_3734_cast_fp16_to_fp32_dtype_0, x = var_3734_cast_fp16)[name = tensor("cast_306")]; tensor var_3735 = rsqrt(epsilon = var_3735_epsilon_0, x = var_3734_cast_fp16_to_fp32)[name = tensor("op_3735")]; tensor var_3735_to_fp16_dtype_0 = const()[name = tensor("op_3735_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3735_to_fp16 = cast(dtype = var_3735_to_fp16_dtype_0, x = var_3735)[name = tensor("cast_305")]; tensor x_847_cast_fp16 = mul(x = x_841_cast_fp16, y = var_3735_to_fp16)[name = tensor("x_847_cast_fp16")]; tensor layers_21_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_21_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(409697792)))]; tensor x_849_cast_fp16 = mul(x = layers_21_input_layernorm_weight_to_fp16, y = x_847_cast_fp16)[name = tensor("x_849_cast_fp16")]; tensor layers_21_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_21_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(409698880)))]; tensor linear_147_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_21_self_attn_q_proj_weight_to_fp16, x = x_849_cast_fp16)[name = tensor("linear_147_cast_fp16")]; tensor var_3749 = const()[name = tensor("op_3749"), val = tensor([1, 1, 12, 128])]; tensor x_851_cast_fp16 = reshape(shape = var_3749, x = linear_147_cast_fp16)[name = tensor("x_851_cast_fp16")]; tensor x_851_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_851_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3719_promoted_1 = const()[name = tensor("op_3719_promoted_1"), val = tensor(0x1p+1)]; tensor x_851_cast_fp16_to_fp32 = cast(dtype = x_851_cast_fp16_to_fp32_dtype_0, x = x_851_cast_fp16)[name = tensor("cast_304")]; tensor var_3753 = pow(x = x_851_cast_fp16_to_fp32, y = var_3719_promoted_1)[name = tensor("op_3753")]; tensor var_171_axes_0 = const()[name = tensor("var_171_axes_0"), val = tensor([-1])]; tensor var_171_keep_dims_0 = const()[name = tensor("var_171_keep_dims_0"), val = tensor(true)]; tensor var_171_0 = reduce_mean(axes = var_171_axes_0, keep_dims = var_171_keep_dims_0, x = var_3753)[name = tensor("var_171")]; tensor var_171_to_fp16_dtype_0 = const()[name = tensor("var_171_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3757_to_fp16 = const()[name = tensor("op_3757_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_171_to_fp16 = cast(dtype = var_171_to_fp16_dtype_0, x = var_171_0)[name = tensor("cast_303")]; tensor var_3758_cast_fp16 = add(x = var_171_to_fp16, y = var_3757_to_fp16)[name = tensor("op_3758_cast_fp16")]; tensor var_3758_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3758_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3759_epsilon_0 = const()[name = tensor("op_3759_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3758_cast_fp16_to_fp32 = cast(dtype = var_3758_cast_fp16_to_fp32_dtype_0, x = var_3758_cast_fp16)[name = tensor("cast_302")]; tensor var_3759 = rsqrt(epsilon = var_3759_epsilon_0, x = var_3758_cast_fp16_to_fp32)[name = tensor("op_3759")]; tensor var_3759_to_fp16_dtype_0 = const()[name = tensor("op_3759_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3759_to_fp16 = cast(dtype = var_3759_to_fp16_dtype_0, x = var_3759)[name = tensor("cast_301")]; tensor x_857_cast_fp16 = mul(x = x_851_cast_fp16, y = var_3759_to_fp16)[name = tensor("x_857_cast_fp16")]; tensor layers_21_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_21_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(411271808)))]; tensor var_3761_cast_fp16 = mul(x = layers_21_self_attn_q_norm_weight_to_fp16, y = x_857_cast_fp16)[name = tensor("op_3761_cast_fp16")]; tensor q_85_perm_0 = const()[name = tensor("q_85_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_21_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_21_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(411272128)))]; tensor linear_148_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_21_self_attn_k_proj_weight_to_fp16, x = x_849_cast_fp16)[name = tensor("linear_148_cast_fp16")]; tensor var_3765 = const()[name = tensor("op_3765"), val = tensor([1, 1, 4, 128])]; tensor x_859_cast_fp16 = reshape(shape = var_3765, x = linear_148_cast_fp16)[name = tensor("x_859_cast_fp16")]; tensor x_859_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_859_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3719_promoted_2 = const()[name = tensor("op_3719_promoted_2"), val = tensor(0x1p+1)]; tensor x_859_cast_fp16_to_fp32 = cast(dtype = x_859_cast_fp16_to_fp32_dtype_0, x = x_859_cast_fp16)[name = tensor("cast_300")]; tensor var_3769 = pow(x = x_859_cast_fp16_to_fp32, y = var_3719_promoted_2)[name = tensor("op_3769")]; tensor var_173_axes_0 = const()[name = tensor("var_173_axes_0"), val = tensor([-1])]; tensor var_173_keep_dims_0 = const()[name = tensor("var_173_keep_dims_0"), val = tensor(true)]; tensor var_173 = reduce_mean(axes = var_173_axes_0, keep_dims = var_173_keep_dims_0, x = var_3769)[name = tensor("var_173")]; tensor var_173_to_fp16_dtype_0 = const()[name = tensor("var_173_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3773_to_fp16 = const()[name = tensor("op_3773_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_173_to_fp16 = cast(dtype = var_173_to_fp16_dtype_0, x = var_173)[name = tensor("cast_299")]; tensor var_3774_cast_fp16 = add(x = var_173_to_fp16, y = var_3773_to_fp16)[name = tensor("op_3774_cast_fp16")]; tensor var_3774_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3774_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3775_epsilon_0 = const()[name = tensor("op_3775_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3774_cast_fp16_to_fp32 = cast(dtype = var_3774_cast_fp16_to_fp32_dtype_0, x = var_3774_cast_fp16)[name = tensor("cast_298")]; tensor var_3775 = rsqrt(epsilon = var_3775_epsilon_0, x = var_3774_cast_fp16_to_fp32)[name = tensor("op_3775")]; tensor var_3775_to_fp16_dtype_0 = const()[name = tensor("op_3775_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3775_to_fp16 = cast(dtype = var_3775_to_fp16_dtype_0, x = var_3775)[name = tensor("cast_297")]; tensor x_865_cast_fp16 = mul(x = x_859_cast_fp16, y = var_3775_to_fp16)[name = tensor("x_865_cast_fp16")]; tensor layers_21_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_21_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(411796480)))]; tensor var_3777_cast_fp16 = mul(x = layers_21_self_attn_k_norm_weight_to_fp16, y = x_865_cast_fp16)[name = tensor("op_3777_cast_fp16")]; tensor k_85_perm_0 = const()[name = tensor("k_85_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_21_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_21_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(411796800)))]; tensor linear_149_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_21_self_attn_v_proj_weight_to_fp16, x = x_849_cast_fp16)[name = tensor("linear_149_cast_fp16")]; tensor var_3781 = const()[name = tensor("op_3781"), val = tensor([1, 1, 4, 128])]; tensor var_3782_cast_fp16 = reshape(shape = var_3781, x = linear_149_cast_fp16)[name = tensor("op_3782_cast_fp16")]; tensor v_43_perm_0 = const()[name = tensor("v_43_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_85_cast_fp16 = transpose(perm = q_85_perm_0, x = var_3761_cast_fp16)[name = tensor("transpose_27")]; tensor var_3786_cast_fp16 = mul(x = q_85_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3786_cast_fp16")]; tensor x1_85_begin_0 = const()[name = tensor("x1_85_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_85_end_0 = const()[name = tensor("x1_85_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_85_end_mask_0 = const()[name = tensor("x1_85_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_85_cast_fp16 = slice_by_index(begin = x1_85_begin_0, end = x1_85_end_0, end_mask = x1_85_end_mask_0, x = q_85_cast_fp16)[name = tensor("x1_85_cast_fp16")]; tensor x2_85_begin_0 = const()[name = tensor("x2_85_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_85_end_0 = const()[name = tensor("x2_85_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_85_end_mask_0 = const()[name = tensor("x2_85_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_85_cast_fp16 = slice_by_index(begin = x2_85_begin_0, end = x2_85_end_0, end_mask = x2_85_end_mask_0, x = q_85_cast_fp16)[name = tensor("x2_85_cast_fp16")]; tensor const_43_promoted_to_fp16 = const()[name = tensor("const_43_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3789_cast_fp16 = mul(x = x2_85_cast_fp16, y = const_43_promoted_to_fp16)[name = tensor("op_3789_cast_fp16")]; tensor var_3791_interleave_0 = const()[name = tensor("op_3791_interleave_0"), val = tensor(false)]; tensor var_3791_cast_fp16 = concat(axis = var_3720, interleave = var_3791_interleave_0, values = (var_3789_cast_fp16, x1_85_cast_fp16))[name = tensor("op_3791_cast_fp16")]; tensor var_3792_cast_fp16 = mul(x = var_3791_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3792_cast_fp16")]; tensor q_87_cast_fp16 = add(x = var_3786_cast_fp16, y = var_3792_cast_fp16)[name = tensor("q_87_cast_fp16")]; tensor k_85_cast_fp16 = transpose(perm = k_85_perm_0, x = var_3777_cast_fp16)[name = tensor("transpose_26")]; tensor var_3794_cast_fp16 = mul(x = k_85_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3794_cast_fp16")]; tensor x1_87_begin_0 = const()[name = tensor("x1_87_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_87_end_0 = const()[name = tensor("x1_87_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_87_end_mask_0 = const()[name = tensor("x1_87_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_87_cast_fp16 = slice_by_index(begin = x1_87_begin_0, end = x1_87_end_0, end_mask = x1_87_end_mask_0, x = k_85_cast_fp16)[name = tensor("x1_87_cast_fp16")]; tensor x2_87_begin_0 = const()[name = tensor("x2_87_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_87_end_0 = const()[name = tensor("x2_87_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_87_end_mask_0 = const()[name = tensor("x2_87_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_87_cast_fp16 = slice_by_index(begin = x2_87_begin_0, end = x2_87_end_0, end_mask = x2_87_end_mask_0, x = k_85_cast_fp16)[name = tensor("x2_87_cast_fp16")]; tensor const_44_promoted_to_fp16 = const()[name = tensor("const_44_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3797_cast_fp16 = mul(x = x2_87_cast_fp16, y = const_44_promoted_to_fp16)[name = tensor("op_3797_cast_fp16")]; tensor var_3799_interleave_0 = const()[name = tensor("op_3799_interleave_0"), val = tensor(false)]; tensor var_3799_cast_fp16 = concat(axis = var_3720, interleave = var_3799_interleave_0, values = (var_3797_cast_fp16, x1_87_cast_fp16))[name = tensor("op_3799_cast_fp16")]; tensor var_3800_cast_fp16 = mul(x = var_3799_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3800_cast_fp16")]; tensor k_87_cast_fp16 = add(x = var_3794_cast_fp16, y = var_3800_cast_fp16)[name = tensor("k_87_cast_fp16")]; tensor var_3803_cast_fp16 = mul(x = k_cache_43_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3803_cast_fp16")]; tensor var_3804_cast_fp16 = mul(x = k_87_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3804_cast_fp16")]; tensor k_full_43_cast_fp16 = add(x = var_3803_cast_fp16, y = var_3804_cast_fp16)[name = tensor("k_full_43_cast_fp16")]; tensor var_3807_cast_fp16 = mul(x = v_cache_43_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3807_cast_fp16")]; tensor v_43_cast_fp16 = transpose(perm = v_43_perm_0, x = var_3782_cast_fp16)[name = tensor("transpose_25")]; tensor var_3808_cast_fp16 = mul(x = v_43_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3808_cast_fp16")]; tensor v_full_43_cast_fp16 = add(x = var_3807_cast_fp16, y = var_3808_cast_fp16)[name = tensor("v_full_43_cast_fp16")]; tensor var_3810_axes_0 = const()[name = tensor("op_3810_axes_0"), val = tensor([2])]; tensor var_3810_cast_fp16 = expand_dims(axes = var_3810_axes_0, x = k_full_43_cast_fp16)[name = tensor("op_3810_cast_fp16")]; tensor var_3812_reps_0 = const()[name = tensor("op_3812_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3812_cast_fp16 = tile(reps = var_3812_reps_0, x = var_3810_cast_fp16)[name = tensor("op_3812_cast_fp16")]; tensor var_3813 = const()[name = tensor("op_3813"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_43_cast_fp16 = reshape(shape = var_3813, x = var_3812_cast_fp16)[name = tensor("k_rep_43_cast_fp16")]; tensor var_3815_axes_0 = const()[name = tensor("op_3815_axes_0"), val = tensor([2])]; tensor var_3815_cast_fp16 = expand_dims(axes = var_3815_axes_0, x = v_full_43_cast_fp16)[name = tensor("op_3815_cast_fp16")]; tensor var_3817_reps_0 = const()[name = tensor("op_3817_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3817_cast_fp16 = tile(reps = var_3817_reps_0, x = var_3815_cast_fp16)[name = tensor("op_3817_cast_fp16")]; tensor var_3818 = const()[name = tensor("op_3818"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_43_cast_fp16 = reshape(shape = var_3818, x = var_3817_cast_fp16)[name = tensor("v_rep_43_cast_fp16")]; tensor var_3821_transpose_x_1 = const()[name = tensor("op_3821_transpose_x_1"), val = tensor(false)]; tensor var_3821_transpose_y_1 = const()[name = tensor("op_3821_transpose_y_1"), val = tensor(true)]; tensor var_3821_cast_fp16 = matmul(transpose_x = var_3821_transpose_x_1, transpose_y = var_3821_transpose_y_1, x = q_87_cast_fp16, y = k_rep_43_cast_fp16)[name = tensor("op_3821_cast_fp16")]; tensor var_3822_to_fp16 = const()[name = tensor("op_3822_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_85_cast_fp16 = mul(x = var_3821_cast_fp16, y = var_3822_to_fp16)[name = tensor("attn_85_cast_fp16")]; tensor input_87_cast_fp16 = add(x = attn_85_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_87_cast_fp16")]; tensor input_87_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_87_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_87_cast_fp16_to_fp32 = cast(dtype = input_87_cast_fp16_to_fp32_dtype_0, x = input_87_cast_fp16)[name = tensor("cast_296")]; tensor attn_87 = softmax(axis = var_3720, x = input_87_cast_fp16_to_fp32)[name = tensor("attn_87")]; tensor out_43_transpose_x_0 = const()[name = tensor("out_43_transpose_x_0"), val = tensor(false)]; tensor out_43_transpose_y_0 = const()[name = tensor("out_43_transpose_y_0"), val = tensor(false)]; tensor attn_87_to_fp16_dtype_0 = const()[name = tensor("attn_87_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_87_to_fp16 = cast(dtype = attn_87_to_fp16_dtype_0, x = attn_87)[name = tensor("cast_295")]; tensor out_43_cast_fp16 = matmul(transpose_x = out_43_transpose_x_0, transpose_y = out_43_transpose_y_0, x = attn_87_to_fp16, y = v_rep_43_cast_fp16)[name = tensor("out_43_cast_fp16")]; tensor var_3827_perm_0 = const()[name = tensor("op_3827_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3829 = const()[name = tensor("op_3829"), val = tensor([1, 1, 1536])]; tensor var_3827_cast_fp16 = transpose(perm = var_3827_perm_0, x = out_43_cast_fp16)[name = tensor("transpose_24")]; tensor x_867_cast_fp16 = reshape(shape = var_3829, x = var_3827_cast_fp16)[name = tensor("x_867_cast_fp16")]; tensor layers_21_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_21_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(412321152)))]; tensor linear_150_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_21_self_attn_o_proj_weight_to_fp16, x = x_867_cast_fp16)[name = tensor("linear_150_cast_fp16")]; tensor x_869_cast_fp16 = add(x = x_841_cast_fp16, y = linear_150_cast_fp16)[name = tensor("x_869_cast_fp16")]; tensor x_869_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_869_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3719_promoted_3 = const()[name = tensor("op_3719_promoted_3"), val = tensor(0x1p+1)]; tensor x_869_cast_fp16_to_fp32 = cast(dtype = x_869_cast_fp16_to_fp32_dtype_0, x = x_869_cast_fp16)[name = tensor("cast_294")]; tensor var_3840 = pow(x = x_869_cast_fp16_to_fp32, y = var_3719_promoted_3)[name = tensor("op_3840")]; tensor var_175_axes_0 = const()[name = tensor("var_175_axes_0"), val = tensor([-1])]; tensor var_175_keep_dims_0 = const()[name = tensor("var_175_keep_dims_0"), val = tensor(true)]; tensor var_175 = reduce_mean(axes = var_175_axes_0, keep_dims = var_175_keep_dims_0, x = var_3840)[name = tensor("var_175")]; tensor var_175_to_fp16_dtype_0 = const()[name = tensor("var_175_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3844_to_fp16 = const()[name = tensor("op_3844_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_175_to_fp16 = cast(dtype = var_175_to_fp16_dtype_0, x = var_175)[name = tensor("cast_293")]; tensor var_3845_cast_fp16 = add(x = var_175_to_fp16, y = var_3844_to_fp16)[name = tensor("op_3845_cast_fp16")]; tensor var_3845_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3845_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3846_epsilon_0 = const()[name = tensor("op_3846_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3845_cast_fp16_to_fp32 = cast(dtype = var_3845_cast_fp16_to_fp32_dtype_0, x = var_3845_cast_fp16)[name = tensor("cast_292")]; tensor var_3846 = rsqrt(epsilon = var_3846_epsilon_0, x = var_3845_cast_fp16_to_fp32)[name = tensor("op_3846")]; tensor var_3846_to_fp16_dtype_0 = const()[name = tensor("op_3846_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3846_to_fp16 = cast(dtype = var_3846_to_fp16_dtype_0, x = var_3846)[name = tensor("cast_291")]; tensor x_875_cast_fp16 = mul(x = x_869_cast_fp16, y = var_3846_to_fp16)[name = tensor("x_875_cast_fp16")]; tensor layers_21_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_21_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(413894080)))]; tensor x_877_cast_fp16 = mul(x = layers_21_post_attention_layernorm_weight_to_fp16, y = x_875_cast_fp16)[name = tensor("x_877_cast_fp16")]; tensor layers_21_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_21_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(413895168)))]; tensor linear_151_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_21_mlp_gate_proj_weight_to_fp16, x = x_877_cast_fp16)[name = tensor("linear_151_cast_fp16")]; tensor var_3857_cast_fp16 = silu(x = linear_151_cast_fp16)[name = tensor("op_3857_cast_fp16")]; tensor layers_21_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_21_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(415468096)))]; tensor linear_152_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_21_mlp_up_proj_weight_to_fp16, x = x_877_cast_fp16)[name = tensor("linear_152_cast_fp16")]; tensor x_879_cast_fp16 = mul(x = var_3857_cast_fp16, y = linear_152_cast_fp16)[name = tensor("x_879_cast_fp16")]; tensor layers_21_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_21_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(417041024)))]; tensor linear_153_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_21_mlp_down_proj_weight_to_fp16, x = x_879_cast_fp16)[name = tensor("linear_153_cast_fp16")]; tensor x_881_cast_fp16 = add(x = x_869_cast_fp16, y = linear_153_cast_fp16)[name = tensor("x_881_cast_fp16")]; tensor x_881_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_881_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_45_begin_0 = const()[name = tensor("k_cache_45_begin_0"), val = tensor([22, 0, 0, 0, 0])]; tensor k_cache_45_end_0 = const()[name = tensor("k_cache_45_end_0"), val = tensor([23, 1, 4, 2048, 128])]; tensor k_cache_45_end_mask_0 = const()[name = tensor("k_cache_45_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_45_squeeze_mask_0 = const()[name = tensor("k_cache_45_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_45_cast_fp16 = slice_by_index(begin = k_cache_45_begin_0, end = k_cache_45_end_0, end_mask = k_cache_45_end_mask_0, squeeze_mask = k_cache_45_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_45_cast_fp16")]; tensor v_cache_45_begin_0 = const()[name = tensor("v_cache_45_begin_0"), val = tensor([22, 0, 0, 0, 0])]; tensor v_cache_45_end_0 = const()[name = tensor("v_cache_45_end_0"), val = tensor([23, 1, 4, 2048, 128])]; tensor v_cache_45_end_mask_0 = const()[name = tensor("v_cache_45_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_45_squeeze_mask_0 = const()[name = tensor("v_cache_45_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_45_cast_fp16 = slice_by_index(begin = v_cache_45_begin_0, end = v_cache_45_end_0, end_mask = v_cache_45_end_mask_0, squeeze_mask = v_cache_45_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_45_cast_fp16")]; tensor var_3889 = const()[name = tensor("op_3889"), val = tensor(-1)]; tensor var_3888_promoted = const()[name = tensor("op_3888_promoted"), val = tensor(0x1p+1)]; tensor x_881_cast_fp16_to_fp32 = cast(dtype = x_881_cast_fp16_to_fp32_dtype_0, x = x_881_cast_fp16)[name = tensor("cast_290")]; tensor var_3898 = pow(x = x_881_cast_fp16_to_fp32, y = var_3888_promoted)[name = tensor("op_3898")]; tensor var_177_axes_0 = const()[name = tensor("var_177_axes_0"), val = tensor([-1])]; tensor var_177_keep_dims_0 = const()[name = tensor("var_177_keep_dims_0"), val = tensor(true)]; tensor var_177 = reduce_mean(axes = var_177_axes_0, keep_dims = var_177_keep_dims_0, x = var_3898)[name = tensor("var_177")]; tensor var_177_to_fp16_dtype_0 = const()[name = tensor("var_177_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3902_to_fp16 = const()[name = tensor("op_3902_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_177_to_fp16 = cast(dtype = var_177_to_fp16_dtype_0, x = var_177)[name = tensor("cast_289")]; tensor var_3903_cast_fp16 = add(x = var_177_to_fp16, y = var_3902_to_fp16)[name = tensor("op_3903_cast_fp16")]; tensor var_3903_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3903_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3904_epsilon_0 = const()[name = tensor("op_3904_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3903_cast_fp16_to_fp32 = cast(dtype = var_3903_cast_fp16_to_fp32_dtype_0, x = var_3903_cast_fp16)[name = tensor("cast_288")]; tensor var_3904 = rsqrt(epsilon = var_3904_epsilon_0, x = var_3903_cast_fp16_to_fp32)[name = tensor("op_3904")]; tensor var_3904_to_fp16_dtype_0 = const()[name = tensor("op_3904_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3904_to_fp16 = cast(dtype = var_3904_to_fp16_dtype_0, x = var_3904)[name = tensor("cast_287")]; tensor x_887_cast_fp16 = mul(x = x_881_cast_fp16, y = var_3904_to_fp16)[name = tensor("x_887_cast_fp16")]; tensor layers_22_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_22_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(418613952)))]; tensor x_889_cast_fp16 = mul(x = layers_22_input_layernorm_weight_to_fp16, y = x_887_cast_fp16)[name = tensor("x_889_cast_fp16")]; tensor layers_22_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_22_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(418615040)))]; tensor linear_154_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_22_self_attn_q_proj_weight_to_fp16, x = x_889_cast_fp16)[name = tensor("linear_154_cast_fp16")]; tensor var_3918 = const()[name = tensor("op_3918"), val = tensor([1, 1, 12, 128])]; tensor x_891_cast_fp16 = reshape(shape = var_3918, x = linear_154_cast_fp16)[name = tensor("x_891_cast_fp16")]; tensor x_891_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_891_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3888_promoted_1 = const()[name = tensor("op_3888_promoted_1"), val = tensor(0x1p+1)]; tensor x_891_cast_fp16_to_fp32 = cast(dtype = x_891_cast_fp16_to_fp32_dtype_0, x = x_891_cast_fp16)[name = tensor("cast_286")]; tensor var_3922 = pow(x = x_891_cast_fp16_to_fp32, y = var_3888_promoted_1)[name = tensor("op_3922")]; tensor var_179_axes_0 = const()[name = tensor("var_179_axes_0"), val = tensor([-1])]; tensor var_179_keep_dims_0 = const()[name = tensor("var_179_keep_dims_0"), val = tensor(true)]; tensor var_179 = reduce_mean(axes = var_179_axes_0, keep_dims = var_179_keep_dims_0, x = var_3922)[name = tensor("var_179")]; tensor var_179_to_fp16_dtype_0 = const()[name = tensor("var_179_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3926_to_fp16 = const()[name = tensor("op_3926_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_179_to_fp16 = cast(dtype = var_179_to_fp16_dtype_0, x = var_179)[name = tensor("cast_285")]; tensor var_3927_cast_fp16 = add(x = var_179_to_fp16, y = var_3926_to_fp16)[name = tensor("op_3927_cast_fp16")]; tensor var_3927_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3927_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3928_epsilon_0 = const()[name = tensor("op_3928_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3927_cast_fp16_to_fp32 = cast(dtype = var_3927_cast_fp16_to_fp32_dtype_0, x = var_3927_cast_fp16)[name = tensor("cast_284")]; tensor var_3928 = rsqrt(epsilon = var_3928_epsilon_0, x = var_3927_cast_fp16_to_fp32)[name = tensor("op_3928")]; tensor var_3928_to_fp16_dtype_0 = const()[name = tensor("op_3928_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3928_to_fp16 = cast(dtype = var_3928_to_fp16_dtype_0, x = var_3928)[name = tensor("cast_283")]; tensor x_897_cast_fp16 = mul(x = x_891_cast_fp16, y = var_3928_to_fp16)[name = tensor("x_897_cast_fp16")]; tensor layers_22_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_22_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(420187968)))]; tensor var_3930_cast_fp16 = mul(x = layers_22_self_attn_q_norm_weight_to_fp16, y = x_897_cast_fp16)[name = tensor("op_3930_cast_fp16")]; tensor q_89_perm_0 = const()[name = tensor("q_89_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_22_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_22_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(420188288)))]; tensor linear_155_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_22_self_attn_k_proj_weight_to_fp16, x = x_889_cast_fp16)[name = tensor("linear_155_cast_fp16")]; tensor var_3934 = const()[name = tensor("op_3934"), val = tensor([1, 1, 4, 128])]; tensor x_899_cast_fp16 = reshape(shape = var_3934, x = linear_155_cast_fp16)[name = tensor("x_899_cast_fp16")]; tensor x_899_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_899_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3888_promoted_2 = const()[name = tensor("op_3888_promoted_2"), val = tensor(0x1p+1)]; tensor x_899_cast_fp16_to_fp32 = cast(dtype = x_899_cast_fp16_to_fp32_dtype_0, x = x_899_cast_fp16)[name = tensor("cast_282")]; tensor var_3938 = pow(x = x_899_cast_fp16_to_fp32, y = var_3888_promoted_2)[name = tensor("op_3938")]; tensor var_181_axes_0 = const()[name = tensor("var_181_axes_0"), val = tensor([-1])]; tensor var_181_keep_dims_0 = const()[name = tensor("var_181_keep_dims_0"), val = tensor(true)]; tensor var_181 = reduce_mean(axes = var_181_axes_0, keep_dims = var_181_keep_dims_0, x = var_3938)[name = tensor("var_181")]; tensor var_181_to_fp16_dtype_0 = const()[name = tensor("var_181_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3942_to_fp16 = const()[name = tensor("op_3942_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_181_to_fp16 = cast(dtype = var_181_to_fp16_dtype_0, x = var_181)[name = tensor("cast_281")]; tensor var_3943_cast_fp16 = add(x = var_181_to_fp16, y = var_3942_to_fp16)[name = tensor("op_3943_cast_fp16")]; tensor var_3943_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_3943_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3944_epsilon_0 = const()[name = tensor("op_3944_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_3943_cast_fp16_to_fp32 = cast(dtype = var_3943_cast_fp16_to_fp32_dtype_0, x = var_3943_cast_fp16)[name = tensor("cast_280")]; tensor var_3944 = rsqrt(epsilon = var_3944_epsilon_0, x = var_3943_cast_fp16_to_fp32)[name = tensor("op_3944")]; tensor var_3944_to_fp16_dtype_0 = const()[name = tensor("op_3944_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_3944_to_fp16 = cast(dtype = var_3944_to_fp16_dtype_0, x = var_3944)[name = tensor("cast_279")]; tensor x_905_cast_fp16 = mul(x = x_899_cast_fp16, y = var_3944_to_fp16)[name = tensor("x_905_cast_fp16")]; tensor layers_22_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_22_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(420712640)))]; tensor var_3946_cast_fp16 = mul(x = layers_22_self_attn_k_norm_weight_to_fp16, y = x_905_cast_fp16)[name = tensor("op_3946_cast_fp16")]; tensor k_89_perm_0 = const()[name = tensor("k_89_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_22_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_22_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(420712960)))]; tensor linear_156_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_22_self_attn_v_proj_weight_to_fp16, x = x_889_cast_fp16)[name = tensor("linear_156_cast_fp16")]; tensor var_3950 = const()[name = tensor("op_3950"), val = tensor([1, 1, 4, 128])]; tensor var_3951_cast_fp16 = reshape(shape = var_3950, x = linear_156_cast_fp16)[name = tensor("op_3951_cast_fp16")]; tensor v_45_perm_0 = const()[name = tensor("v_45_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_89_cast_fp16 = transpose(perm = q_89_perm_0, x = var_3930_cast_fp16)[name = tensor("transpose_23")]; tensor var_3955_cast_fp16 = mul(x = q_89_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3955_cast_fp16")]; tensor x1_89_begin_0 = const()[name = tensor("x1_89_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_89_end_0 = const()[name = tensor("x1_89_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_89_end_mask_0 = const()[name = tensor("x1_89_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_89_cast_fp16 = slice_by_index(begin = x1_89_begin_0, end = x1_89_end_0, end_mask = x1_89_end_mask_0, x = q_89_cast_fp16)[name = tensor("x1_89_cast_fp16")]; tensor x2_89_begin_0 = const()[name = tensor("x2_89_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_89_end_0 = const()[name = tensor("x2_89_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_89_end_mask_0 = const()[name = tensor("x2_89_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_89_cast_fp16 = slice_by_index(begin = x2_89_begin_0, end = x2_89_end_0, end_mask = x2_89_end_mask_0, x = q_89_cast_fp16)[name = tensor("x2_89_cast_fp16")]; tensor const_45_promoted_to_fp16 = const()[name = tensor("const_45_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3958_cast_fp16 = mul(x = x2_89_cast_fp16, y = const_45_promoted_to_fp16)[name = tensor("op_3958_cast_fp16")]; tensor var_3960_interleave_0 = const()[name = tensor("op_3960_interleave_0"), val = tensor(false)]; tensor var_3960_cast_fp16 = concat(axis = var_3889, interleave = var_3960_interleave_0, values = (var_3958_cast_fp16, x1_89_cast_fp16))[name = tensor("op_3960_cast_fp16")]; tensor var_3961_cast_fp16 = mul(x = var_3960_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3961_cast_fp16")]; tensor q_91_cast_fp16 = add(x = var_3955_cast_fp16, y = var_3961_cast_fp16)[name = tensor("q_91_cast_fp16")]; tensor k_89_cast_fp16 = transpose(perm = k_89_perm_0, x = var_3946_cast_fp16)[name = tensor("transpose_22")]; tensor var_3963_cast_fp16 = mul(x = k_89_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_3963_cast_fp16")]; tensor x1_91_begin_0 = const()[name = tensor("x1_91_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_91_end_0 = const()[name = tensor("x1_91_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_91_end_mask_0 = const()[name = tensor("x1_91_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_91_cast_fp16 = slice_by_index(begin = x1_91_begin_0, end = x1_91_end_0, end_mask = x1_91_end_mask_0, x = k_89_cast_fp16)[name = tensor("x1_91_cast_fp16")]; tensor x2_91_begin_0 = const()[name = tensor("x2_91_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_91_end_0 = const()[name = tensor("x2_91_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_91_end_mask_0 = const()[name = tensor("x2_91_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_91_cast_fp16 = slice_by_index(begin = x2_91_begin_0, end = x2_91_end_0, end_mask = x2_91_end_mask_0, x = k_89_cast_fp16)[name = tensor("x2_91_cast_fp16")]; tensor const_46_promoted_to_fp16 = const()[name = tensor("const_46_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_3966_cast_fp16 = mul(x = x2_91_cast_fp16, y = const_46_promoted_to_fp16)[name = tensor("op_3966_cast_fp16")]; tensor var_3968_interleave_0 = const()[name = tensor("op_3968_interleave_0"), val = tensor(false)]; tensor var_3968_cast_fp16 = concat(axis = var_3889, interleave = var_3968_interleave_0, values = (var_3966_cast_fp16, x1_91_cast_fp16))[name = tensor("op_3968_cast_fp16")]; tensor var_3969_cast_fp16 = mul(x = var_3968_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_3969_cast_fp16")]; tensor k_91_cast_fp16 = add(x = var_3963_cast_fp16, y = var_3969_cast_fp16)[name = tensor("k_91_cast_fp16")]; tensor var_3972_cast_fp16 = mul(x = k_cache_45_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3972_cast_fp16")]; tensor var_3973_cast_fp16 = mul(x = k_91_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3973_cast_fp16")]; tensor k_full_45_cast_fp16 = add(x = var_3972_cast_fp16, y = var_3973_cast_fp16)[name = tensor("k_full_45_cast_fp16")]; tensor var_3976_cast_fp16 = mul(x = v_cache_45_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_3976_cast_fp16")]; tensor v_45_cast_fp16 = transpose(perm = v_45_perm_0, x = var_3951_cast_fp16)[name = tensor("transpose_21")]; tensor var_3977_cast_fp16 = mul(x = v_45_cast_fp16, y = var_107_to_fp16)[name = tensor("op_3977_cast_fp16")]; tensor v_full_45_cast_fp16 = add(x = var_3976_cast_fp16, y = var_3977_cast_fp16)[name = tensor("v_full_45_cast_fp16")]; tensor var_3979_axes_0 = const()[name = tensor("op_3979_axes_0"), val = tensor([2])]; tensor var_3979_cast_fp16 = expand_dims(axes = var_3979_axes_0, x = k_full_45_cast_fp16)[name = tensor("op_3979_cast_fp16")]; tensor var_3981_reps_0 = const()[name = tensor("op_3981_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3981_cast_fp16 = tile(reps = var_3981_reps_0, x = var_3979_cast_fp16)[name = tensor("op_3981_cast_fp16")]; tensor var_3982 = const()[name = tensor("op_3982"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_45_cast_fp16 = reshape(shape = var_3982, x = var_3981_cast_fp16)[name = tensor("k_rep_45_cast_fp16")]; tensor var_3984_axes_0 = const()[name = tensor("op_3984_axes_0"), val = tensor([2])]; tensor var_3984_cast_fp16 = expand_dims(axes = var_3984_axes_0, x = v_full_45_cast_fp16)[name = tensor("op_3984_cast_fp16")]; tensor var_3986_reps_0 = const()[name = tensor("op_3986_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_3986_cast_fp16 = tile(reps = var_3986_reps_0, x = var_3984_cast_fp16)[name = tensor("op_3986_cast_fp16")]; tensor var_3987 = const()[name = tensor("op_3987"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_45_cast_fp16 = reshape(shape = var_3987, x = var_3986_cast_fp16)[name = tensor("v_rep_45_cast_fp16")]; tensor var_3990_transpose_x_1 = const()[name = tensor("op_3990_transpose_x_1"), val = tensor(false)]; tensor var_3990_transpose_y_1 = const()[name = tensor("op_3990_transpose_y_1"), val = tensor(true)]; tensor var_3990_cast_fp16 = matmul(transpose_x = var_3990_transpose_x_1, transpose_y = var_3990_transpose_y_1, x = q_91_cast_fp16, y = k_rep_45_cast_fp16)[name = tensor("op_3990_cast_fp16")]; tensor var_3991_to_fp16 = const()[name = tensor("op_3991_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_89_cast_fp16 = mul(x = var_3990_cast_fp16, y = var_3991_to_fp16)[name = tensor("attn_89_cast_fp16")]; tensor input_91_cast_fp16 = add(x = attn_89_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_91_cast_fp16")]; tensor input_91_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_91_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_91_cast_fp16_to_fp32 = cast(dtype = input_91_cast_fp16_to_fp32_dtype_0, x = input_91_cast_fp16)[name = tensor("cast_278")]; tensor attn_91 = softmax(axis = var_3889, x = input_91_cast_fp16_to_fp32)[name = tensor("attn_91")]; tensor out_45_transpose_x_0 = const()[name = tensor("out_45_transpose_x_0"), val = tensor(false)]; tensor out_45_transpose_y_0 = const()[name = tensor("out_45_transpose_y_0"), val = tensor(false)]; tensor attn_91_to_fp16_dtype_0 = const()[name = tensor("attn_91_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_91_to_fp16 = cast(dtype = attn_91_to_fp16_dtype_0, x = attn_91)[name = tensor("cast_277")]; tensor out_45_cast_fp16 = matmul(transpose_x = out_45_transpose_x_0, transpose_y = out_45_transpose_y_0, x = attn_91_to_fp16, y = v_rep_45_cast_fp16)[name = tensor("out_45_cast_fp16")]; tensor var_3996_perm_0 = const()[name = tensor("op_3996_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3998 = const()[name = tensor("op_3998"), val = tensor([1, 1, 1536])]; tensor var_3996_cast_fp16 = transpose(perm = var_3996_perm_0, x = out_45_cast_fp16)[name = tensor("transpose_20")]; tensor x_907_cast_fp16 = reshape(shape = var_3998, x = var_3996_cast_fp16)[name = tensor("x_907_cast_fp16")]; tensor layers_22_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_22_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(421237312)))]; tensor linear_157_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_22_self_attn_o_proj_weight_to_fp16, x = x_907_cast_fp16)[name = tensor("linear_157_cast_fp16")]; tensor x_909_cast_fp16 = add(x = x_881_cast_fp16, y = linear_157_cast_fp16)[name = tensor("x_909_cast_fp16")]; tensor x_909_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_909_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_3888_promoted_3 = const()[name = tensor("op_3888_promoted_3"), val = tensor(0x1p+1)]; tensor x_909_cast_fp16_to_fp32 = cast(dtype = x_909_cast_fp16_to_fp32_dtype_0, x = x_909_cast_fp16)[name = tensor("cast_276")]; tensor var_4009 = pow(x = x_909_cast_fp16_to_fp32, y = var_3888_promoted_3)[name = tensor("op_4009")]; tensor var_183_axes_0 = const()[name = tensor("var_183_axes_0"), val = tensor([-1])]; tensor var_183_keep_dims_0 = const()[name = tensor("var_183_keep_dims_0"), val = tensor(true)]; tensor var_183 = reduce_mean(axes = var_183_axes_0, keep_dims = var_183_keep_dims_0, x = var_4009)[name = tensor("var_183")]; tensor var_183_to_fp16_dtype_0 = const()[name = tensor("var_183_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4013_to_fp16 = const()[name = tensor("op_4013_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_183_to_fp16 = cast(dtype = var_183_to_fp16_dtype_0, x = var_183)[name = tensor("cast_275")]; tensor var_4014_cast_fp16 = add(x = var_183_to_fp16, y = var_4013_to_fp16)[name = tensor("op_4014_cast_fp16")]; tensor var_4014_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4014_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4015_epsilon_0 = const()[name = tensor("op_4015_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4014_cast_fp16_to_fp32 = cast(dtype = var_4014_cast_fp16_to_fp32_dtype_0, x = var_4014_cast_fp16)[name = tensor("cast_274")]; tensor var_4015 = rsqrt(epsilon = var_4015_epsilon_0, x = var_4014_cast_fp16_to_fp32)[name = tensor("op_4015")]; tensor var_4015_to_fp16_dtype_0 = const()[name = tensor("op_4015_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4015_to_fp16 = cast(dtype = var_4015_to_fp16_dtype_0, x = var_4015)[name = tensor("cast_273")]; tensor x_915_cast_fp16 = mul(x = x_909_cast_fp16, y = var_4015_to_fp16)[name = tensor("x_915_cast_fp16")]; tensor layers_22_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_22_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(422810240)))]; tensor x_917_cast_fp16 = mul(x = layers_22_post_attention_layernorm_weight_to_fp16, y = x_915_cast_fp16)[name = tensor("x_917_cast_fp16")]; tensor layers_22_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_22_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(422811328)))]; tensor linear_158_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_22_mlp_gate_proj_weight_to_fp16, x = x_917_cast_fp16)[name = tensor("linear_158_cast_fp16")]; tensor var_4026_cast_fp16 = silu(x = linear_158_cast_fp16)[name = tensor("op_4026_cast_fp16")]; tensor layers_22_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_22_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(424384256)))]; tensor linear_159_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_22_mlp_up_proj_weight_to_fp16, x = x_917_cast_fp16)[name = tensor("linear_159_cast_fp16")]; tensor x_919_cast_fp16 = mul(x = var_4026_cast_fp16, y = linear_159_cast_fp16)[name = tensor("x_919_cast_fp16")]; tensor layers_22_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_22_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(425957184)))]; tensor linear_160_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_22_mlp_down_proj_weight_to_fp16, x = x_919_cast_fp16)[name = tensor("linear_160_cast_fp16")]; tensor x_921_cast_fp16 = add(x = x_909_cast_fp16, y = linear_160_cast_fp16)[name = tensor("x_921_cast_fp16")]; tensor x_921_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_921_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_47_begin_0 = const()[name = tensor("k_cache_47_begin_0"), val = tensor([23, 0, 0, 0, 0])]; tensor k_cache_47_end_0 = const()[name = tensor("k_cache_47_end_0"), val = tensor([24, 1, 4, 2048, 128])]; tensor k_cache_47_end_mask_0 = const()[name = tensor("k_cache_47_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_47_squeeze_mask_0 = const()[name = tensor("k_cache_47_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_47_cast_fp16 = slice_by_index(begin = k_cache_47_begin_0, end = k_cache_47_end_0, end_mask = k_cache_47_end_mask_0, squeeze_mask = k_cache_47_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_47_cast_fp16")]; tensor v_cache_47_begin_0 = const()[name = tensor("v_cache_47_begin_0"), val = tensor([23, 0, 0, 0, 0])]; tensor v_cache_47_end_0 = const()[name = tensor("v_cache_47_end_0"), val = tensor([24, 1, 4, 2048, 128])]; tensor v_cache_47_end_mask_0 = const()[name = tensor("v_cache_47_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_47_squeeze_mask_0 = const()[name = tensor("v_cache_47_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_47_cast_fp16 = slice_by_index(begin = v_cache_47_begin_0, end = v_cache_47_end_0, end_mask = v_cache_47_end_mask_0, squeeze_mask = v_cache_47_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_47_cast_fp16")]; tensor var_4058 = const()[name = tensor("op_4058"), val = tensor(-1)]; tensor var_4057_promoted = const()[name = tensor("op_4057_promoted"), val = tensor(0x1p+1)]; tensor x_921_cast_fp16_to_fp32 = cast(dtype = x_921_cast_fp16_to_fp32_dtype_0, x = x_921_cast_fp16)[name = tensor("cast_272")]; tensor var_4067 = pow(x = x_921_cast_fp16_to_fp32, y = var_4057_promoted)[name = tensor("op_4067")]; tensor var_185_axes_0 = const()[name = tensor("var_185_axes_0"), val = tensor([-1])]; tensor var_185_keep_dims_0 = const()[name = tensor("var_185_keep_dims_0"), val = tensor(true)]; tensor var_185 = reduce_mean(axes = var_185_axes_0, keep_dims = var_185_keep_dims_0, x = var_4067)[name = tensor("var_185")]; tensor var_185_to_fp16_dtype_0 = const()[name = tensor("var_185_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4071_to_fp16 = const()[name = tensor("op_4071_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_185_to_fp16 = cast(dtype = var_185_to_fp16_dtype_0, x = var_185)[name = tensor("cast_271")]; tensor var_4072_cast_fp16 = add(x = var_185_to_fp16, y = var_4071_to_fp16)[name = tensor("op_4072_cast_fp16")]; tensor var_4072_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4072_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4073_epsilon_0 = const()[name = tensor("op_4073_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4072_cast_fp16_to_fp32 = cast(dtype = var_4072_cast_fp16_to_fp32_dtype_0, x = var_4072_cast_fp16)[name = tensor("cast_270")]; tensor var_4073 = rsqrt(epsilon = var_4073_epsilon_0, x = var_4072_cast_fp16_to_fp32)[name = tensor("op_4073")]; tensor var_4073_to_fp16_dtype_0 = const()[name = tensor("op_4073_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4073_to_fp16 = cast(dtype = var_4073_to_fp16_dtype_0, x = var_4073)[name = tensor("cast_269")]; tensor x_927_cast_fp16 = mul(x = x_921_cast_fp16, y = var_4073_to_fp16)[name = tensor("x_927_cast_fp16")]; tensor layers_23_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_23_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(427530112)))]; tensor x_929_cast_fp16 = mul(x = layers_23_input_layernorm_weight_to_fp16, y = x_927_cast_fp16)[name = tensor("x_929_cast_fp16")]; tensor layers_23_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_23_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(427531200)))]; tensor linear_161_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_23_self_attn_q_proj_weight_to_fp16, x = x_929_cast_fp16)[name = tensor("linear_161_cast_fp16")]; tensor var_4087 = const()[name = tensor("op_4087"), val = tensor([1, 1, 12, 128])]; tensor x_931_cast_fp16 = reshape(shape = var_4087, x = linear_161_cast_fp16)[name = tensor("x_931_cast_fp16")]; tensor x_931_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_931_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4057_promoted_1 = const()[name = tensor("op_4057_promoted_1"), val = tensor(0x1p+1)]; tensor x_931_cast_fp16_to_fp32 = cast(dtype = x_931_cast_fp16_to_fp32_dtype_0, x = x_931_cast_fp16)[name = tensor("cast_268")]; tensor var_4091 = pow(x = x_931_cast_fp16_to_fp32, y = var_4057_promoted_1)[name = tensor("op_4091")]; tensor var_187_axes_0 = const()[name = tensor("var_187_axes_0"), val = tensor([-1])]; tensor var_187_keep_dims_0 = const()[name = tensor("var_187_keep_dims_0"), val = tensor(true)]; tensor var_187 = reduce_mean(axes = var_187_axes_0, keep_dims = var_187_keep_dims_0, x = var_4091)[name = tensor("var_187")]; tensor var_187_to_fp16_dtype_0 = const()[name = tensor("var_187_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4095_to_fp16 = const()[name = tensor("op_4095_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_187_to_fp16 = cast(dtype = var_187_to_fp16_dtype_0, x = var_187)[name = tensor("cast_267")]; tensor var_4096_cast_fp16 = add(x = var_187_to_fp16, y = var_4095_to_fp16)[name = tensor("op_4096_cast_fp16")]; tensor var_4096_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4096_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4097_epsilon_0 = const()[name = tensor("op_4097_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4096_cast_fp16_to_fp32 = cast(dtype = var_4096_cast_fp16_to_fp32_dtype_0, x = var_4096_cast_fp16)[name = tensor("cast_266")]; tensor var_4097 = rsqrt(epsilon = var_4097_epsilon_0, x = var_4096_cast_fp16_to_fp32)[name = tensor("op_4097")]; tensor var_4097_to_fp16_dtype_0 = const()[name = tensor("op_4097_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4097_to_fp16 = cast(dtype = var_4097_to_fp16_dtype_0, x = var_4097)[name = tensor("cast_265")]; tensor x_937_cast_fp16 = mul(x = x_931_cast_fp16, y = var_4097_to_fp16)[name = tensor("x_937_cast_fp16")]; tensor layers_23_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_23_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(429104128)))]; tensor var_4099_cast_fp16 = mul(x = layers_23_self_attn_q_norm_weight_to_fp16, y = x_937_cast_fp16)[name = tensor("op_4099_cast_fp16")]; tensor q_93_perm_0 = const()[name = tensor("q_93_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_23_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_23_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(429104448)))]; tensor linear_162_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_23_self_attn_k_proj_weight_to_fp16, x = x_929_cast_fp16)[name = tensor("linear_162_cast_fp16")]; tensor var_4103 = const()[name = tensor("op_4103"), val = tensor([1, 1, 4, 128])]; tensor x_939_cast_fp16 = reshape(shape = var_4103, x = linear_162_cast_fp16)[name = tensor("x_939_cast_fp16")]; tensor x_939_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_939_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4057_promoted_2 = const()[name = tensor("op_4057_promoted_2"), val = tensor(0x1p+1)]; tensor x_939_cast_fp16_to_fp32 = cast(dtype = x_939_cast_fp16_to_fp32_dtype_0, x = x_939_cast_fp16)[name = tensor("cast_264")]; tensor var_4107 = pow(x = x_939_cast_fp16_to_fp32, y = var_4057_promoted_2)[name = tensor("op_4107")]; tensor var_189_axes_0 = const()[name = tensor("var_189_axes_0"), val = tensor([-1])]; tensor var_189_keep_dims_0 = const()[name = tensor("var_189_keep_dims_0"), val = tensor(true)]; tensor var_189 = reduce_mean(axes = var_189_axes_0, keep_dims = var_189_keep_dims_0, x = var_4107)[name = tensor("var_189")]; tensor var_189_to_fp16_dtype_0 = const()[name = tensor("var_189_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4111_to_fp16 = const()[name = tensor("op_4111_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_189_to_fp16 = cast(dtype = var_189_to_fp16_dtype_0, x = var_189)[name = tensor("cast_263")]; tensor var_4112_cast_fp16 = add(x = var_189_to_fp16, y = var_4111_to_fp16)[name = tensor("op_4112_cast_fp16")]; tensor var_4112_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4112_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4113_epsilon_0 = const()[name = tensor("op_4113_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4112_cast_fp16_to_fp32 = cast(dtype = var_4112_cast_fp16_to_fp32_dtype_0, x = var_4112_cast_fp16)[name = tensor("cast_262")]; tensor var_4113 = rsqrt(epsilon = var_4113_epsilon_0, x = var_4112_cast_fp16_to_fp32)[name = tensor("op_4113")]; tensor var_4113_to_fp16_dtype_0 = const()[name = tensor("op_4113_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4113_to_fp16 = cast(dtype = var_4113_to_fp16_dtype_0, x = var_4113)[name = tensor("cast_261")]; tensor x_945_cast_fp16 = mul(x = x_939_cast_fp16, y = var_4113_to_fp16)[name = tensor("x_945_cast_fp16")]; tensor layers_23_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_23_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(429628800)))]; tensor var_4115_cast_fp16 = mul(x = layers_23_self_attn_k_norm_weight_to_fp16, y = x_945_cast_fp16)[name = tensor("op_4115_cast_fp16")]; tensor k_93_perm_0 = const()[name = tensor("k_93_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_23_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_23_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(429629120)))]; tensor linear_163_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_23_self_attn_v_proj_weight_to_fp16, x = x_929_cast_fp16)[name = tensor("linear_163_cast_fp16")]; tensor var_4119 = const()[name = tensor("op_4119"), val = tensor([1, 1, 4, 128])]; tensor var_4120_cast_fp16 = reshape(shape = var_4119, x = linear_163_cast_fp16)[name = tensor("op_4120_cast_fp16")]; tensor v_47_perm_0 = const()[name = tensor("v_47_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_93_cast_fp16 = transpose(perm = q_93_perm_0, x = var_4099_cast_fp16)[name = tensor("transpose_19")]; tensor var_4124_cast_fp16 = mul(x = q_93_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_4124_cast_fp16")]; tensor x1_93_begin_0 = const()[name = tensor("x1_93_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_93_end_0 = const()[name = tensor("x1_93_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_93_end_mask_0 = const()[name = tensor("x1_93_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_93_cast_fp16 = slice_by_index(begin = x1_93_begin_0, end = x1_93_end_0, end_mask = x1_93_end_mask_0, x = q_93_cast_fp16)[name = tensor("x1_93_cast_fp16")]; tensor x2_93_begin_0 = const()[name = tensor("x2_93_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_93_end_0 = const()[name = tensor("x2_93_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_93_end_mask_0 = const()[name = tensor("x2_93_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_93_cast_fp16 = slice_by_index(begin = x2_93_begin_0, end = x2_93_end_0, end_mask = x2_93_end_mask_0, x = q_93_cast_fp16)[name = tensor("x2_93_cast_fp16")]; tensor const_47_promoted_to_fp16 = const()[name = tensor("const_47_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_4127_cast_fp16 = mul(x = x2_93_cast_fp16, y = const_47_promoted_to_fp16)[name = tensor("op_4127_cast_fp16")]; tensor var_4129_interleave_0 = const()[name = tensor("op_4129_interleave_0"), val = tensor(false)]; tensor var_4129_cast_fp16 = concat(axis = var_4058, interleave = var_4129_interleave_0, values = (var_4127_cast_fp16, x1_93_cast_fp16))[name = tensor("op_4129_cast_fp16")]; tensor var_4130_cast_fp16 = mul(x = var_4129_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_4130_cast_fp16")]; tensor q_95_cast_fp16 = add(x = var_4124_cast_fp16, y = var_4130_cast_fp16)[name = tensor("q_95_cast_fp16")]; tensor k_93_cast_fp16 = transpose(perm = k_93_perm_0, x = var_4115_cast_fp16)[name = tensor("transpose_18")]; tensor var_4132_cast_fp16 = mul(x = k_93_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_4132_cast_fp16")]; tensor x1_95_begin_0 = const()[name = tensor("x1_95_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_95_end_0 = const()[name = tensor("x1_95_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_95_end_mask_0 = const()[name = tensor("x1_95_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_95_cast_fp16 = slice_by_index(begin = x1_95_begin_0, end = x1_95_end_0, end_mask = x1_95_end_mask_0, x = k_93_cast_fp16)[name = tensor("x1_95_cast_fp16")]; tensor x2_95_begin_0 = const()[name = tensor("x2_95_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_95_end_0 = const()[name = tensor("x2_95_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_95_end_mask_0 = const()[name = tensor("x2_95_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_95_cast_fp16 = slice_by_index(begin = x2_95_begin_0, end = x2_95_end_0, end_mask = x2_95_end_mask_0, x = k_93_cast_fp16)[name = tensor("x2_95_cast_fp16")]; tensor const_48_promoted_to_fp16 = const()[name = tensor("const_48_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_4135_cast_fp16 = mul(x = x2_95_cast_fp16, y = const_48_promoted_to_fp16)[name = tensor("op_4135_cast_fp16")]; tensor var_4137_interleave_0 = const()[name = tensor("op_4137_interleave_0"), val = tensor(false)]; tensor var_4137_cast_fp16 = concat(axis = var_4058, interleave = var_4137_interleave_0, values = (var_4135_cast_fp16, x1_95_cast_fp16))[name = tensor("op_4137_cast_fp16")]; tensor var_4138_cast_fp16 = mul(x = var_4137_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_4138_cast_fp16")]; tensor k_95_cast_fp16 = add(x = var_4132_cast_fp16, y = var_4138_cast_fp16)[name = tensor("k_95_cast_fp16")]; tensor var_4141_cast_fp16 = mul(x = k_cache_47_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_4141_cast_fp16")]; tensor var_4142_cast_fp16 = mul(x = k_95_cast_fp16, y = var_107_to_fp16)[name = tensor("op_4142_cast_fp16")]; tensor k_full_47_cast_fp16 = add(x = var_4141_cast_fp16, y = var_4142_cast_fp16)[name = tensor("k_full_47_cast_fp16")]; tensor var_4145_cast_fp16 = mul(x = v_cache_47_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_4145_cast_fp16")]; tensor v_47_cast_fp16 = transpose(perm = v_47_perm_0, x = var_4120_cast_fp16)[name = tensor("transpose_17")]; tensor var_4146_cast_fp16 = mul(x = v_47_cast_fp16, y = var_107_to_fp16)[name = tensor("op_4146_cast_fp16")]; tensor v_full_47_cast_fp16 = add(x = var_4145_cast_fp16, y = var_4146_cast_fp16)[name = tensor("v_full_47_cast_fp16")]; tensor var_4148_axes_0 = const()[name = tensor("op_4148_axes_0"), val = tensor([2])]; tensor var_4148_cast_fp16 = expand_dims(axes = var_4148_axes_0, x = k_full_47_cast_fp16)[name = tensor("op_4148_cast_fp16")]; tensor var_4150_reps_0 = const()[name = tensor("op_4150_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_4150_cast_fp16 = tile(reps = var_4150_reps_0, x = var_4148_cast_fp16)[name = tensor("op_4150_cast_fp16")]; tensor var_4151 = const()[name = tensor("op_4151"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_47_cast_fp16 = reshape(shape = var_4151, x = var_4150_cast_fp16)[name = tensor("k_rep_47_cast_fp16")]; tensor var_4153_axes_0 = const()[name = tensor("op_4153_axes_0"), val = tensor([2])]; tensor var_4153_cast_fp16 = expand_dims(axes = var_4153_axes_0, x = v_full_47_cast_fp16)[name = tensor("op_4153_cast_fp16")]; tensor var_4155_reps_0 = const()[name = tensor("op_4155_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_4155_cast_fp16 = tile(reps = var_4155_reps_0, x = var_4153_cast_fp16)[name = tensor("op_4155_cast_fp16")]; tensor var_4156 = const()[name = tensor("op_4156"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_47_cast_fp16 = reshape(shape = var_4156, x = var_4155_cast_fp16)[name = tensor("v_rep_47_cast_fp16")]; tensor var_4159_transpose_x_1 = const()[name = tensor("op_4159_transpose_x_1"), val = tensor(false)]; tensor var_4159_transpose_y_1 = const()[name = tensor("op_4159_transpose_y_1"), val = tensor(true)]; tensor var_4159_cast_fp16 = matmul(transpose_x = var_4159_transpose_x_1, transpose_y = var_4159_transpose_y_1, x = q_95_cast_fp16, y = k_rep_47_cast_fp16)[name = tensor("op_4159_cast_fp16")]; tensor var_4160_to_fp16 = const()[name = tensor("op_4160_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_93_cast_fp16 = mul(x = var_4159_cast_fp16, y = var_4160_to_fp16)[name = tensor("attn_93_cast_fp16")]; tensor input_95_cast_fp16 = add(x = attn_93_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_95_cast_fp16")]; tensor input_95_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_95_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_95_cast_fp16_to_fp32 = cast(dtype = input_95_cast_fp16_to_fp32_dtype_0, x = input_95_cast_fp16)[name = tensor("cast_260")]; tensor attn_95 = softmax(axis = var_4058, x = input_95_cast_fp16_to_fp32)[name = tensor("attn_95")]; tensor out_47_transpose_x_0 = const()[name = tensor("out_47_transpose_x_0"), val = tensor(false)]; tensor out_47_transpose_y_0 = const()[name = tensor("out_47_transpose_y_0"), val = tensor(false)]; tensor attn_95_to_fp16_dtype_0 = const()[name = tensor("attn_95_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_95_to_fp16 = cast(dtype = attn_95_to_fp16_dtype_0, x = attn_95)[name = tensor("cast_259")]; tensor out_47_cast_fp16 = matmul(transpose_x = out_47_transpose_x_0, transpose_y = out_47_transpose_y_0, x = attn_95_to_fp16, y = v_rep_47_cast_fp16)[name = tensor("out_47_cast_fp16")]; tensor var_4165_perm_0 = const()[name = tensor("op_4165_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_4167 = const()[name = tensor("op_4167"), val = tensor([1, 1, 1536])]; tensor var_4165_cast_fp16 = transpose(perm = var_4165_perm_0, x = out_47_cast_fp16)[name = tensor("transpose_16")]; tensor x_947_cast_fp16 = reshape(shape = var_4167, x = var_4165_cast_fp16)[name = tensor("x_947_cast_fp16")]; tensor layers_23_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_23_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(430153472)))]; tensor linear_164_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_23_self_attn_o_proj_weight_to_fp16, x = x_947_cast_fp16)[name = tensor("linear_164_cast_fp16")]; tensor x_949_cast_fp16 = add(x = x_921_cast_fp16, y = linear_164_cast_fp16)[name = tensor("x_949_cast_fp16")]; tensor x_949_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_949_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4057_promoted_3 = const()[name = tensor("op_4057_promoted_3"), val = tensor(0x1p+1)]; tensor x_949_cast_fp16_to_fp32 = cast(dtype = x_949_cast_fp16_to_fp32_dtype_0, x = x_949_cast_fp16)[name = tensor("cast_258")]; tensor var_4178 = pow(x = x_949_cast_fp16_to_fp32, y = var_4057_promoted_3)[name = tensor("op_4178")]; tensor var_191_axes_0 = const()[name = tensor("var_191_axes_0"), val = tensor([-1])]; tensor var_191_keep_dims_0 = const()[name = tensor("var_191_keep_dims_0"), val = tensor(true)]; tensor var_191 = reduce_mean(axes = var_191_axes_0, keep_dims = var_191_keep_dims_0, x = var_4178)[name = tensor("var_191")]; tensor var_191_to_fp16_dtype_0 = const()[name = tensor("var_191_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4182_to_fp16 = const()[name = tensor("op_4182_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_191_to_fp16 = cast(dtype = var_191_to_fp16_dtype_0, x = var_191)[name = tensor("cast_257")]; tensor var_4183_cast_fp16 = add(x = var_191_to_fp16, y = var_4182_to_fp16)[name = tensor("op_4183_cast_fp16")]; tensor var_4183_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4183_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4184_epsilon_0 = const()[name = tensor("op_4184_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4183_cast_fp16_to_fp32 = cast(dtype = var_4183_cast_fp16_to_fp32_dtype_0, x = var_4183_cast_fp16)[name = tensor("cast_256")]; tensor var_4184 = rsqrt(epsilon = var_4184_epsilon_0, x = var_4183_cast_fp16_to_fp32)[name = tensor("op_4184")]; tensor var_4184_to_fp16_dtype_0 = const()[name = tensor("op_4184_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4184_to_fp16 = cast(dtype = var_4184_to_fp16_dtype_0, x = var_4184)[name = tensor("cast_255")]; tensor x_955_cast_fp16 = mul(x = x_949_cast_fp16, y = var_4184_to_fp16)[name = tensor("x_955_cast_fp16")]; tensor layers_23_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_23_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(431726400)))]; tensor x_957_cast_fp16 = mul(x = layers_23_post_attention_layernorm_weight_to_fp16, y = x_955_cast_fp16)[name = tensor("x_957_cast_fp16")]; tensor layers_23_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_23_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(431727488)))]; tensor linear_165_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_23_mlp_gate_proj_weight_to_fp16, x = x_957_cast_fp16)[name = tensor("linear_165_cast_fp16")]; tensor var_4195_cast_fp16 = silu(x = linear_165_cast_fp16)[name = tensor("op_4195_cast_fp16")]; tensor layers_23_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_23_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(433300416)))]; tensor linear_166_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_23_mlp_up_proj_weight_to_fp16, x = x_957_cast_fp16)[name = tensor("linear_166_cast_fp16")]; tensor x_959_cast_fp16 = mul(x = var_4195_cast_fp16, y = linear_166_cast_fp16)[name = tensor("x_959_cast_fp16")]; tensor layers_23_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_23_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(434873344)))]; tensor linear_167_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_23_mlp_down_proj_weight_to_fp16, x = x_959_cast_fp16)[name = tensor("linear_167_cast_fp16")]; tensor x_961_cast_fp16 = add(x = x_949_cast_fp16, y = linear_167_cast_fp16)[name = tensor("x_961_cast_fp16")]; tensor x_961_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_961_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_49_begin_0 = const()[name = tensor("k_cache_49_begin_0"), val = tensor([24, 0, 0, 0, 0])]; tensor k_cache_49_end_0 = const()[name = tensor("k_cache_49_end_0"), val = tensor([25, 1, 4, 2048, 128])]; tensor k_cache_49_end_mask_0 = const()[name = tensor("k_cache_49_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_49_squeeze_mask_0 = const()[name = tensor("k_cache_49_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_49_cast_fp16 = slice_by_index(begin = k_cache_49_begin_0, end = k_cache_49_end_0, end_mask = k_cache_49_end_mask_0, squeeze_mask = k_cache_49_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_49_cast_fp16")]; tensor v_cache_49_begin_0 = const()[name = tensor("v_cache_49_begin_0"), val = tensor([24, 0, 0, 0, 0])]; tensor v_cache_49_end_0 = const()[name = tensor("v_cache_49_end_0"), val = tensor([25, 1, 4, 2048, 128])]; tensor v_cache_49_end_mask_0 = const()[name = tensor("v_cache_49_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_49_squeeze_mask_0 = const()[name = tensor("v_cache_49_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_49_cast_fp16 = slice_by_index(begin = v_cache_49_begin_0, end = v_cache_49_end_0, end_mask = v_cache_49_end_mask_0, squeeze_mask = v_cache_49_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_49_cast_fp16")]; tensor var_4227 = const()[name = tensor("op_4227"), val = tensor(-1)]; tensor var_4226_promoted = const()[name = tensor("op_4226_promoted"), val = tensor(0x1p+1)]; tensor x_961_cast_fp16_to_fp32 = cast(dtype = x_961_cast_fp16_to_fp32_dtype_0, x = x_961_cast_fp16)[name = tensor("cast_254")]; tensor var_4236 = pow(x = x_961_cast_fp16_to_fp32, y = var_4226_promoted)[name = tensor("op_4236")]; tensor var_193_axes_0 = const()[name = tensor("var_193_axes_0"), val = tensor([-1])]; tensor var_193_keep_dims_0 = const()[name = tensor("var_193_keep_dims_0"), val = tensor(true)]; tensor var_193 = reduce_mean(axes = var_193_axes_0, keep_dims = var_193_keep_dims_0, x = var_4236)[name = tensor("var_193")]; tensor var_193_to_fp16_dtype_0 = const()[name = tensor("var_193_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4240_to_fp16 = const()[name = tensor("op_4240_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_193_to_fp16 = cast(dtype = var_193_to_fp16_dtype_0, x = var_193)[name = tensor("cast_253")]; tensor var_4241_cast_fp16 = add(x = var_193_to_fp16, y = var_4240_to_fp16)[name = tensor("op_4241_cast_fp16")]; tensor var_4241_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4241_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4242_epsilon_0 = const()[name = tensor("op_4242_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4241_cast_fp16_to_fp32 = cast(dtype = var_4241_cast_fp16_to_fp32_dtype_0, x = var_4241_cast_fp16)[name = tensor("cast_252")]; tensor var_4242 = rsqrt(epsilon = var_4242_epsilon_0, x = var_4241_cast_fp16_to_fp32)[name = tensor("op_4242")]; tensor var_4242_to_fp16_dtype_0 = const()[name = tensor("op_4242_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4242_to_fp16 = cast(dtype = var_4242_to_fp16_dtype_0, x = var_4242)[name = tensor("cast_251")]; tensor x_967_cast_fp16 = mul(x = x_961_cast_fp16, y = var_4242_to_fp16)[name = tensor("x_967_cast_fp16")]; tensor layers_24_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_24_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(436446272)))]; tensor x_969_cast_fp16 = mul(x = layers_24_input_layernorm_weight_to_fp16, y = x_967_cast_fp16)[name = tensor("x_969_cast_fp16")]; tensor layers_24_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_24_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(436447360)))]; tensor linear_168_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_24_self_attn_q_proj_weight_to_fp16, x = x_969_cast_fp16)[name = tensor("linear_168_cast_fp16")]; tensor var_4256 = const()[name = tensor("op_4256"), val = tensor([1, 1, 12, 128])]; tensor x_971_cast_fp16 = reshape(shape = var_4256, x = linear_168_cast_fp16)[name = tensor("x_971_cast_fp16")]; tensor x_971_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_971_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4226_promoted_1 = const()[name = tensor("op_4226_promoted_1"), val = tensor(0x1p+1)]; tensor x_971_cast_fp16_to_fp32 = cast(dtype = x_971_cast_fp16_to_fp32_dtype_0, x = x_971_cast_fp16)[name = tensor("cast_250")]; tensor var_4260 = pow(x = x_971_cast_fp16_to_fp32, y = var_4226_promoted_1)[name = tensor("op_4260")]; tensor var_195_axes_0 = const()[name = tensor("var_195_axes_0"), val = tensor([-1])]; tensor var_195_keep_dims_0 = const()[name = tensor("var_195_keep_dims_0"), val = tensor(true)]; tensor var_195 = reduce_mean(axes = var_195_axes_0, keep_dims = var_195_keep_dims_0, x = var_4260)[name = tensor("var_195")]; tensor var_195_to_fp16_dtype_0 = const()[name = tensor("var_195_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4264_to_fp16 = const()[name = tensor("op_4264_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_195_to_fp16 = cast(dtype = var_195_to_fp16_dtype_0, x = var_195)[name = tensor("cast_249")]; tensor var_4265_cast_fp16 = add(x = var_195_to_fp16, y = var_4264_to_fp16)[name = tensor("op_4265_cast_fp16")]; tensor var_4265_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4265_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4266_epsilon_0 = const()[name = tensor("op_4266_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4265_cast_fp16_to_fp32 = cast(dtype = var_4265_cast_fp16_to_fp32_dtype_0, x = var_4265_cast_fp16)[name = tensor("cast_248")]; tensor var_4266 = rsqrt(epsilon = var_4266_epsilon_0, x = var_4265_cast_fp16_to_fp32)[name = tensor("op_4266")]; tensor var_4266_to_fp16_dtype_0 = const()[name = tensor("op_4266_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4266_to_fp16 = cast(dtype = var_4266_to_fp16_dtype_0, x = var_4266)[name = tensor("cast_247")]; tensor x_977_cast_fp16 = mul(x = x_971_cast_fp16, y = var_4266_to_fp16)[name = tensor("x_977_cast_fp16")]; tensor layers_24_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_24_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(438020288)))]; tensor var_4268_cast_fp16 = mul(x = layers_24_self_attn_q_norm_weight_to_fp16, y = x_977_cast_fp16)[name = tensor("op_4268_cast_fp16")]; tensor q_97_perm_0 = const()[name = tensor("q_97_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_24_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_24_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(438020608)))]; tensor linear_169_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_24_self_attn_k_proj_weight_to_fp16, x = x_969_cast_fp16)[name = tensor("linear_169_cast_fp16")]; tensor var_4272 = const()[name = tensor("op_4272"), val = tensor([1, 1, 4, 128])]; tensor x_979_cast_fp16 = reshape(shape = var_4272, x = linear_169_cast_fp16)[name = tensor("x_979_cast_fp16")]; tensor x_979_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_979_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4226_promoted_2 = const()[name = tensor("op_4226_promoted_2"), val = tensor(0x1p+1)]; tensor x_979_cast_fp16_to_fp32 = cast(dtype = x_979_cast_fp16_to_fp32_dtype_0, x = x_979_cast_fp16)[name = tensor("cast_246")]; tensor var_4276 = pow(x = x_979_cast_fp16_to_fp32, y = var_4226_promoted_2)[name = tensor("op_4276")]; tensor var_197_axes_0 = const()[name = tensor("var_197_axes_0"), val = tensor([-1])]; tensor var_197_keep_dims_0 = const()[name = tensor("var_197_keep_dims_0"), val = tensor(true)]; tensor var_197 = reduce_mean(axes = var_197_axes_0, keep_dims = var_197_keep_dims_0, x = var_4276)[name = tensor("var_197")]; tensor var_197_to_fp16_dtype_0 = const()[name = tensor("var_197_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4280_to_fp16 = const()[name = tensor("op_4280_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_197_to_fp16 = cast(dtype = var_197_to_fp16_dtype_0, x = var_197)[name = tensor("cast_245")]; tensor var_4281_cast_fp16 = add(x = var_197_to_fp16, y = var_4280_to_fp16)[name = tensor("op_4281_cast_fp16")]; tensor var_4281_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4281_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4282_epsilon_0 = const()[name = tensor("op_4282_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4281_cast_fp16_to_fp32 = cast(dtype = var_4281_cast_fp16_to_fp32_dtype_0, x = var_4281_cast_fp16)[name = tensor("cast_244")]; tensor var_4282 = rsqrt(epsilon = var_4282_epsilon_0, x = var_4281_cast_fp16_to_fp32)[name = tensor("op_4282")]; tensor var_4282_to_fp16_dtype_0 = const()[name = tensor("op_4282_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4282_to_fp16 = cast(dtype = var_4282_to_fp16_dtype_0, x = var_4282)[name = tensor("cast_243")]; tensor x_985_cast_fp16 = mul(x = x_979_cast_fp16, y = var_4282_to_fp16)[name = tensor("x_985_cast_fp16")]; tensor layers_24_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_24_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(438544960)))]; tensor var_4284_cast_fp16 = mul(x = layers_24_self_attn_k_norm_weight_to_fp16, y = x_985_cast_fp16)[name = tensor("op_4284_cast_fp16")]; tensor k_97_perm_0 = const()[name = tensor("k_97_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_24_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_24_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(438545280)))]; tensor linear_170_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_24_self_attn_v_proj_weight_to_fp16, x = x_969_cast_fp16)[name = tensor("linear_170_cast_fp16")]; tensor var_4288 = const()[name = tensor("op_4288"), val = tensor([1, 1, 4, 128])]; tensor var_4289_cast_fp16 = reshape(shape = var_4288, x = linear_170_cast_fp16)[name = tensor("op_4289_cast_fp16")]; tensor v_49_perm_0 = const()[name = tensor("v_49_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_97_cast_fp16 = transpose(perm = q_97_perm_0, x = var_4268_cast_fp16)[name = tensor("transpose_15")]; tensor var_4293_cast_fp16 = mul(x = q_97_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_4293_cast_fp16")]; tensor x1_97_begin_0 = const()[name = tensor("x1_97_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_97_end_0 = const()[name = tensor("x1_97_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_97_end_mask_0 = const()[name = tensor("x1_97_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_97_cast_fp16 = slice_by_index(begin = x1_97_begin_0, end = x1_97_end_0, end_mask = x1_97_end_mask_0, x = q_97_cast_fp16)[name = tensor("x1_97_cast_fp16")]; tensor x2_97_begin_0 = const()[name = tensor("x2_97_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_97_end_0 = const()[name = tensor("x2_97_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_97_end_mask_0 = const()[name = tensor("x2_97_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_97_cast_fp16 = slice_by_index(begin = x2_97_begin_0, end = x2_97_end_0, end_mask = x2_97_end_mask_0, x = q_97_cast_fp16)[name = tensor("x2_97_cast_fp16")]; tensor const_49_promoted_to_fp16 = const()[name = tensor("const_49_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_4296_cast_fp16 = mul(x = x2_97_cast_fp16, y = const_49_promoted_to_fp16)[name = tensor("op_4296_cast_fp16")]; tensor var_4298_interleave_0 = const()[name = tensor("op_4298_interleave_0"), val = tensor(false)]; tensor var_4298_cast_fp16 = concat(axis = var_4227, interleave = var_4298_interleave_0, values = (var_4296_cast_fp16, x1_97_cast_fp16))[name = tensor("op_4298_cast_fp16")]; tensor var_4299_cast_fp16 = mul(x = var_4298_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_4299_cast_fp16")]; tensor q_99_cast_fp16 = add(x = var_4293_cast_fp16, y = var_4299_cast_fp16)[name = tensor("q_99_cast_fp16")]; tensor k_97_cast_fp16 = transpose(perm = k_97_perm_0, x = var_4284_cast_fp16)[name = tensor("transpose_14")]; tensor var_4301_cast_fp16 = mul(x = k_97_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_4301_cast_fp16")]; tensor x1_99_begin_0 = const()[name = tensor("x1_99_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_99_end_0 = const()[name = tensor("x1_99_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_99_end_mask_0 = const()[name = tensor("x1_99_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_99_cast_fp16 = slice_by_index(begin = x1_99_begin_0, end = x1_99_end_0, end_mask = x1_99_end_mask_0, x = k_97_cast_fp16)[name = tensor("x1_99_cast_fp16")]; tensor x2_99_begin_0 = const()[name = tensor("x2_99_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_99_end_0 = const()[name = tensor("x2_99_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_99_end_mask_0 = const()[name = tensor("x2_99_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_99_cast_fp16 = slice_by_index(begin = x2_99_begin_0, end = x2_99_end_0, end_mask = x2_99_end_mask_0, x = k_97_cast_fp16)[name = tensor("x2_99_cast_fp16")]; tensor const_50_promoted_to_fp16 = const()[name = tensor("const_50_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_4304_cast_fp16 = mul(x = x2_99_cast_fp16, y = const_50_promoted_to_fp16)[name = tensor("op_4304_cast_fp16")]; tensor var_4306_interleave_0 = const()[name = tensor("op_4306_interleave_0"), val = tensor(false)]; tensor var_4306_cast_fp16 = concat(axis = var_4227, interleave = var_4306_interleave_0, values = (var_4304_cast_fp16, x1_99_cast_fp16))[name = tensor("op_4306_cast_fp16")]; tensor var_4307_cast_fp16 = mul(x = var_4306_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_4307_cast_fp16")]; tensor k_99_cast_fp16 = add(x = var_4301_cast_fp16, y = var_4307_cast_fp16)[name = tensor("k_99_cast_fp16")]; tensor var_4310_cast_fp16 = mul(x = k_cache_49_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_4310_cast_fp16")]; tensor var_4311_cast_fp16 = mul(x = k_99_cast_fp16, y = var_107_to_fp16)[name = tensor("op_4311_cast_fp16")]; tensor k_full_49_cast_fp16 = add(x = var_4310_cast_fp16, y = var_4311_cast_fp16)[name = tensor("k_full_49_cast_fp16")]; tensor var_4314_cast_fp16 = mul(x = v_cache_49_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_4314_cast_fp16")]; tensor v_49_cast_fp16 = transpose(perm = v_49_perm_0, x = var_4289_cast_fp16)[name = tensor("transpose_13")]; tensor var_4315_cast_fp16 = mul(x = v_49_cast_fp16, y = var_107_to_fp16)[name = tensor("op_4315_cast_fp16")]; tensor v_full_49_cast_fp16 = add(x = var_4314_cast_fp16, y = var_4315_cast_fp16)[name = tensor("v_full_49_cast_fp16")]; tensor var_4317_axes_0 = const()[name = tensor("op_4317_axes_0"), val = tensor([2])]; tensor var_4317_cast_fp16 = expand_dims(axes = var_4317_axes_0, x = k_full_49_cast_fp16)[name = tensor("op_4317_cast_fp16")]; tensor var_4319_reps_0 = const()[name = tensor("op_4319_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_4319_cast_fp16 = tile(reps = var_4319_reps_0, x = var_4317_cast_fp16)[name = tensor("op_4319_cast_fp16")]; tensor var_4320 = const()[name = tensor("op_4320"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_49_cast_fp16 = reshape(shape = var_4320, x = var_4319_cast_fp16)[name = tensor("k_rep_49_cast_fp16")]; tensor var_4322_axes_0 = const()[name = tensor("op_4322_axes_0"), val = tensor([2])]; tensor var_4322_cast_fp16 = expand_dims(axes = var_4322_axes_0, x = v_full_49_cast_fp16)[name = tensor("op_4322_cast_fp16")]; tensor var_4324_reps_0 = const()[name = tensor("op_4324_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_4324_cast_fp16 = tile(reps = var_4324_reps_0, x = var_4322_cast_fp16)[name = tensor("op_4324_cast_fp16")]; tensor var_4325 = const()[name = tensor("op_4325"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_49_cast_fp16 = reshape(shape = var_4325, x = var_4324_cast_fp16)[name = tensor("v_rep_49_cast_fp16")]; tensor var_4328_transpose_x_1 = const()[name = tensor("op_4328_transpose_x_1"), val = tensor(false)]; tensor var_4328_transpose_y_1 = const()[name = tensor("op_4328_transpose_y_1"), val = tensor(true)]; tensor var_4328_cast_fp16 = matmul(transpose_x = var_4328_transpose_x_1, transpose_y = var_4328_transpose_y_1, x = q_99_cast_fp16, y = k_rep_49_cast_fp16)[name = tensor("op_4328_cast_fp16")]; tensor var_4329_to_fp16 = const()[name = tensor("op_4329_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_97_cast_fp16 = mul(x = var_4328_cast_fp16, y = var_4329_to_fp16)[name = tensor("attn_97_cast_fp16")]; tensor input_99_cast_fp16 = add(x = attn_97_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_99_cast_fp16")]; tensor input_99_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_99_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_99_cast_fp16_to_fp32 = cast(dtype = input_99_cast_fp16_to_fp32_dtype_0, x = input_99_cast_fp16)[name = tensor("cast_242")]; tensor attn_99 = softmax(axis = var_4227, x = input_99_cast_fp16_to_fp32)[name = tensor("attn_99")]; tensor out_49_transpose_x_0 = const()[name = tensor("out_49_transpose_x_0"), val = tensor(false)]; tensor out_49_transpose_y_0 = const()[name = tensor("out_49_transpose_y_0"), val = tensor(false)]; tensor attn_99_to_fp16_dtype_0 = const()[name = tensor("attn_99_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_99_to_fp16 = cast(dtype = attn_99_to_fp16_dtype_0, x = attn_99)[name = tensor("cast_241")]; tensor out_49_cast_fp16 = matmul(transpose_x = out_49_transpose_x_0, transpose_y = out_49_transpose_y_0, x = attn_99_to_fp16, y = v_rep_49_cast_fp16)[name = tensor("out_49_cast_fp16")]; tensor var_4334_perm_0 = const()[name = tensor("op_4334_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_4336 = const()[name = tensor("op_4336"), val = tensor([1, 1, 1536])]; tensor var_4334_cast_fp16 = transpose(perm = var_4334_perm_0, x = out_49_cast_fp16)[name = tensor("transpose_12")]; tensor x_987_cast_fp16 = reshape(shape = var_4336, x = var_4334_cast_fp16)[name = tensor("x_987_cast_fp16")]; tensor layers_24_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_24_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(439069632)))]; tensor linear_171_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_24_self_attn_o_proj_weight_to_fp16, x = x_987_cast_fp16)[name = tensor("linear_171_cast_fp16")]; tensor x_989_cast_fp16 = add(x = x_961_cast_fp16, y = linear_171_cast_fp16)[name = tensor("x_989_cast_fp16")]; tensor x_989_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_989_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4226_promoted_3 = const()[name = tensor("op_4226_promoted_3"), val = tensor(0x1p+1)]; tensor x_989_cast_fp16_to_fp32 = cast(dtype = x_989_cast_fp16_to_fp32_dtype_0, x = x_989_cast_fp16)[name = tensor("cast_240")]; tensor var_4347 = pow(x = x_989_cast_fp16_to_fp32, y = var_4226_promoted_3)[name = tensor("op_4347")]; tensor var_199_axes_0 = const()[name = tensor("var_199_axes_0"), val = tensor([-1])]; tensor var_199_keep_dims_0 = const()[name = tensor("var_199_keep_dims_0"), val = tensor(true)]; tensor var_199 = reduce_mean(axes = var_199_axes_0, keep_dims = var_199_keep_dims_0, x = var_4347)[name = tensor("var_199")]; tensor var_199_to_fp16_dtype_0 = const()[name = tensor("var_199_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4351_to_fp16 = const()[name = tensor("op_4351_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_199_to_fp16 = cast(dtype = var_199_to_fp16_dtype_0, x = var_199)[name = tensor("cast_239")]; tensor var_4352_cast_fp16 = add(x = var_199_to_fp16, y = var_4351_to_fp16)[name = tensor("op_4352_cast_fp16")]; tensor var_4352_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4352_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4353_epsilon_0 = const()[name = tensor("op_4353_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4352_cast_fp16_to_fp32 = cast(dtype = var_4352_cast_fp16_to_fp32_dtype_0, x = var_4352_cast_fp16)[name = tensor("cast_238")]; tensor var_4353 = rsqrt(epsilon = var_4353_epsilon_0, x = var_4352_cast_fp16_to_fp32)[name = tensor("op_4353")]; tensor var_4353_to_fp16_dtype_0 = const()[name = tensor("op_4353_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4353_to_fp16 = cast(dtype = var_4353_to_fp16_dtype_0, x = var_4353)[name = tensor("cast_237")]; tensor x_995_cast_fp16 = mul(x = x_989_cast_fp16, y = var_4353_to_fp16)[name = tensor("x_995_cast_fp16")]; tensor layers_24_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_24_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(440642560)))]; tensor x_997_cast_fp16 = mul(x = layers_24_post_attention_layernorm_weight_to_fp16, y = x_995_cast_fp16)[name = tensor("x_997_cast_fp16")]; tensor layers_24_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_24_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(440643648)))]; tensor linear_172_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_24_mlp_gate_proj_weight_to_fp16, x = x_997_cast_fp16)[name = tensor("linear_172_cast_fp16")]; tensor var_4364_cast_fp16 = silu(x = linear_172_cast_fp16)[name = tensor("op_4364_cast_fp16")]; tensor layers_24_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_24_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(442216576)))]; tensor linear_173_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_24_mlp_up_proj_weight_to_fp16, x = x_997_cast_fp16)[name = tensor("linear_173_cast_fp16")]; tensor x_999_cast_fp16 = mul(x = var_4364_cast_fp16, y = linear_173_cast_fp16)[name = tensor("x_999_cast_fp16")]; tensor layers_24_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_24_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(443789504)))]; tensor linear_174_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_24_mlp_down_proj_weight_to_fp16, x = x_999_cast_fp16)[name = tensor("linear_174_cast_fp16")]; tensor x_1001_cast_fp16 = add(x = x_989_cast_fp16, y = linear_174_cast_fp16)[name = tensor("x_1001_cast_fp16")]; tensor x_1001_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1001_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_51_begin_0 = const()[name = tensor("k_cache_51_begin_0"), val = tensor([25, 0, 0, 0, 0])]; tensor k_cache_51_end_0 = const()[name = tensor("k_cache_51_end_0"), val = tensor([26, 1, 4, 2048, 128])]; tensor k_cache_51_end_mask_0 = const()[name = tensor("k_cache_51_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_51_squeeze_mask_0 = const()[name = tensor("k_cache_51_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_51_cast_fp16 = slice_by_index(begin = k_cache_51_begin_0, end = k_cache_51_end_0, end_mask = k_cache_51_end_mask_0, squeeze_mask = k_cache_51_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_51_cast_fp16")]; tensor v_cache_51_begin_0 = const()[name = tensor("v_cache_51_begin_0"), val = tensor([25, 0, 0, 0, 0])]; tensor v_cache_51_end_0 = const()[name = tensor("v_cache_51_end_0"), val = tensor([26, 1, 4, 2048, 128])]; tensor v_cache_51_end_mask_0 = const()[name = tensor("v_cache_51_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_51_squeeze_mask_0 = const()[name = tensor("v_cache_51_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_51_cast_fp16 = slice_by_index(begin = v_cache_51_begin_0, end = v_cache_51_end_0, end_mask = v_cache_51_end_mask_0, squeeze_mask = v_cache_51_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_51_cast_fp16")]; tensor var_4396 = const()[name = tensor("op_4396"), val = tensor(-1)]; tensor var_4395_promoted = const()[name = tensor("op_4395_promoted"), val = tensor(0x1p+1)]; tensor x_1001_cast_fp16_to_fp32 = cast(dtype = x_1001_cast_fp16_to_fp32_dtype_0, x = x_1001_cast_fp16)[name = tensor("cast_236")]; tensor var_4405 = pow(x = x_1001_cast_fp16_to_fp32, y = var_4395_promoted)[name = tensor("op_4405")]; tensor var_201_axes_0 = const()[name = tensor("var_201_axes_0"), val = tensor([-1])]; tensor var_201_keep_dims_0 = const()[name = tensor("var_201_keep_dims_0"), val = tensor(true)]; tensor var_201 = reduce_mean(axes = var_201_axes_0, keep_dims = var_201_keep_dims_0, x = var_4405)[name = tensor("var_201")]; tensor var_201_to_fp16_dtype_0 = const()[name = tensor("var_201_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4409_to_fp16 = const()[name = tensor("op_4409_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_201_to_fp16 = cast(dtype = var_201_to_fp16_dtype_0, x = var_201)[name = tensor("cast_235")]; tensor var_4410_cast_fp16 = add(x = var_201_to_fp16, y = var_4409_to_fp16)[name = tensor("op_4410_cast_fp16")]; tensor var_4410_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4410_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4411_epsilon_0 = const()[name = tensor("op_4411_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4410_cast_fp16_to_fp32 = cast(dtype = var_4410_cast_fp16_to_fp32_dtype_0, x = var_4410_cast_fp16)[name = tensor("cast_234")]; tensor var_4411 = rsqrt(epsilon = var_4411_epsilon_0, x = var_4410_cast_fp16_to_fp32)[name = tensor("op_4411")]; tensor var_4411_to_fp16_dtype_0 = const()[name = tensor("op_4411_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4411_to_fp16 = cast(dtype = var_4411_to_fp16_dtype_0, x = var_4411)[name = tensor("cast_233")]; tensor x_1007_cast_fp16 = mul(x = x_1001_cast_fp16, y = var_4411_to_fp16)[name = tensor("x_1007_cast_fp16")]; tensor layers_25_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_25_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(445362432)))]; tensor x_1009_cast_fp16 = mul(x = layers_25_input_layernorm_weight_to_fp16, y = x_1007_cast_fp16)[name = tensor("x_1009_cast_fp16")]; tensor layers_25_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_25_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(445363520)))]; tensor linear_175_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_25_self_attn_q_proj_weight_to_fp16, x = x_1009_cast_fp16)[name = tensor("linear_175_cast_fp16")]; tensor var_4425 = const()[name = tensor("op_4425"), val = tensor([1, 1, 12, 128])]; tensor x_1011_cast_fp16 = reshape(shape = var_4425, x = linear_175_cast_fp16)[name = tensor("x_1011_cast_fp16")]; tensor x_1011_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1011_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4395_promoted_1 = const()[name = tensor("op_4395_promoted_1"), val = tensor(0x1p+1)]; tensor x_1011_cast_fp16_to_fp32 = cast(dtype = x_1011_cast_fp16_to_fp32_dtype_0, x = x_1011_cast_fp16)[name = tensor("cast_232")]; tensor var_4429 = pow(x = x_1011_cast_fp16_to_fp32, y = var_4395_promoted_1)[name = tensor("op_4429")]; tensor var_203_axes_0 = const()[name = tensor("var_203_axes_0"), val = tensor([-1])]; tensor var_203_keep_dims_0 = const()[name = tensor("var_203_keep_dims_0"), val = tensor(true)]; tensor var_203 = reduce_mean(axes = var_203_axes_0, keep_dims = var_203_keep_dims_0, x = var_4429)[name = tensor("var_203")]; tensor var_203_to_fp16_dtype_0 = const()[name = tensor("var_203_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4433_to_fp16 = const()[name = tensor("op_4433_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_203_to_fp16 = cast(dtype = var_203_to_fp16_dtype_0, x = var_203)[name = tensor("cast_231")]; tensor var_4434_cast_fp16 = add(x = var_203_to_fp16, y = var_4433_to_fp16)[name = tensor("op_4434_cast_fp16")]; tensor var_4434_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4434_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4435_epsilon_0 = const()[name = tensor("op_4435_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4434_cast_fp16_to_fp32 = cast(dtype = var_4434_cast_fp16_to_fp32_dtype_0, x = var_4434_cast_fp16)[name = tensor("cast_230")]; tensor var_4435 = rsqrt(epsilon = var_4435_epsilon_0, x = var_4434_cast_fp16_to_fp32)[name = tensor("op_4435")]; tensor var_4435_to_fp16_dtype_0 = const()[name = tensor("op_4435_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4435_to_fp16 = cast(dtype = var_4435_to_fp16_dtype_0, x = var_4435)[name = tensor("cast_229")]; tensor x_1017_cast_fp16 = mul(x = x_1011_cast_fp16, y = var_4435_to_fp16)[name = tensor("x_1017_cast_fp16")]; tensor layers_25_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_25_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(446936448)))]; tensor var_4437_cast_fp16 = mul(x = layers_25_self_attn_q_norm_weight_to_fp16, y = x_1017_cast_fp16)[name = tensor("op_4437_cast_fp16")]; tensor q_101_perm_0 = const()[name = tensor("q_101_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_25_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_25_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(446936768)))]; tensor linear_176_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_25_self_attn_k_proj_weight_to_fp16, x = x_1009_cast_fp16)[name = tensor("linear_176_cast_fp16")]; tensor var_4441 = const()[name = tensor("op_4441"), val = tensor([1, 1, 4, 128])]; tensor x_1019_cast_fp16 = reshape(shape = var_4441, x = linear_176_cast_fp16)[name = tensor("x_1019_cast_fp16")]; tensor x_1019_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1019_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4395_promoted_2 = const()[name = tensor("op_4395_promoted_2"), val = tensor(0x1p+1)]; tensor x_1019_cast_fp16_to_fp32 = cast(dtype = x_1019_cast_fp16_to_fp32_dtype_0, x = x_1019_cast_fp16)[name = tensor("cast_228")]; tensor var_4445 = pow(x = x_1019_cast_fp16_to_fp32, y = var_4395_promoted_2)[name = tensor("op_4445")]; tensor var_205_axes_0 = const()[name = tensor("var_205_axes_0"), val = tensor([-1])]; tensor var_205_keep_dims_0 = const()[name = tensor("var_205_keep_dims_0"), val = tensor(true)]; tensor var_205 = reduce_mean(axes = var_205_axes_0, keep_dims = var_205_keep_dims_0, x = var_4445)[name = tensor("var_205")]; tensor var_205_to_fp16_dtype_0 = const()[name = tensor("var_205_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4449_to_fp16 = const()[name = tensor("op_4449_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_205_to_fp16 = cast(dtype = var_205_to_fp16_dtype_0, x = var_205)[name = tensor("cast_227")]; tensor var_4450_cast_fp16 = add(x = var_205_to_fp16, y = var_4449_to_fp16)[name = tensor("op_4450_cast_fp16")]; tensor var_4450_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4450_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4451_epsilon_0 = const()[name = tensor("op_4451_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4450_cast_fp16_to_fp32 = cast(dtype = var_4450_cast_fp16_to_fp32_dtype_0, x = var_4450_cast_fp16)[name = tensor("cast_226")]; tensor var_4451 = rsqrt(epsilon = var_4451_epsilon_0, x = var_4450_cast_fp16_to_fp32)[name = tensor("op_4451")]; tensor var_4451_to_fp16_dtype_0 = const()[name = tensor("op_4451_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4451_to_fp16 = cast(dtype = var_4451_to_fp16_dtype_0, x = var_4451)[name = tensor("cast_225")]; tensor x_1025_cast_fp16 = mul(x = x_1019_cast_fp16, y = var_4451_to_fp16)[name = tensor("x_1025_cast_fp16")]; tensor layers_25_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_25_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(447461120)))]; tensor var_4453_cast_fp16 = mul(x = layers_25_self_attn_k_norm_weight_to_fp16, y = x_1025_cast_fp16)[name = tensor("op_4453_cast_fp16")]; tensor k_101_perm_0 = const()[name = tensor("k_101_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_25_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_25_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(447461440)))]; tensor linear_177_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_25_self_attn_v_proj_weight_to_fp16, x = x_1009_cast_fp16)[name = tensor("linear_177_cast_fp16")]; tensor var_4457 = const()[name = tensor("op_4457"), val = tensor([1, 1, 4, 128])]; tensor var_4458_cast_fp16 = reshape(shape = var_4457, x = linear_177_cast_fp16)[name = tensor("op_4458_cast_fp16")]; tensor v_51_perm_0 = const()[name = tensor("v_51_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_101_cast_fp16 = transpose(perm = q_101_perm_0, x = var_4437_cast_fp16)[name = tensor("transpose_11")]; tensor var_4462_cast_fp16 = mul(x = q_101_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_4462_cast_fp16")]; tensor x1_101_begin_0 = const()[name = tensor("x1_101_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_101_end_0 = const()[name = tensor("x1_101_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_101_end_mask_0 = const()[name = tensor("x1_101_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_101_cast_fp16 = slice_by_index(begin = x1_101_begin_0, end = x1_101_end_0, end_mask = x1_101_end_mask_0, x = q_101_cast_fp16)[name = tensor("x1_101_cast_fp16")]; tensor x2_101_begin_0 = const()[name = tensor("x2_101_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_101_end_0 = const()[name = tensor("x2_101_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_101_end_mask_0 = const()[name = tensor("x2_101_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_101_cast_fp16 = slice_by_index(begin = x2_101_begin_0, end = x2_101_end_0, end_mask = x2_101_end_mask_0, x = q_101_cast_fp16)[name = tensor("x2_101_cast_fp16")]; tensor const_51_promoted_to_fp16 = const()[name = tensor("const_51_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_4465_cast_fp16 = mul(x = x2_101_cast_fp16, y = const_51_promoted_to_fp16)[name = tensor("op_4465_cast_fp16")]; tensor var_4467_interleave_0 = const()[name = tensor("op_4467_interleave_0"), val = tensor(false)]; tensor var_4467_cast_fp16 = concat(axis = var_4396, interleave = var_4467_interleave_0, values = (var_4465_cast_fp16, x1_101_cast_fp16))[name = tensor("op_4467_cast_fp16")]; tensor var_4468_cast_fp16 = mul(x = var_4467_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_4468_cast_fp16")]; tensor q_103_cast_fp16 = add(x = var_4462_cast_fp16, y = var_4468_cast_fp16)[name = tensor("q_103_cast_fp16")]; tensor k_101_cast_fp16 = transpose(perm = k_101_perm_0, x = var_4453_cast_fp16)[name = tensor("transpose_10")]; tensor var_4470_cast_fp16 = mul(x = k_101_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_4470_cast_fp16")]; tensor x1_103_begin_0 = const()[name = tensor("x1_103_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_103_end_0 = const()[name = tensor("x1_103_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_103_end_mask_0 = const()[name = tensor("x1_103_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_103_cast_fp16 = slice_by_index(begin = x1_103_begin_0, end = x1_103_end_0, end_mask = x1_103_end_mask_0, x = k_101_cast_fp16)[name = tensor("x1_103_cast_fp16")]; tensor x2_103_begin_0 = const()[name = tensor("x2_103_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_103_end_0 = const()[name = tensor("x2_103_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_103_end_mask_0 = const()[name = tensor("x2_103_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_103_cast_fp16 = slice_by_index(begin = x2_103_begin_0, end = x2_103_end_0, end_mask = x2_103_end_mask_0, x = k_101_cast_fp16)[name = tensor("x2_103_cast_fp16")]; tensor const_52_promoted_to_fp16 = const()[name = tensor("const_52_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_4473_cast_fp16 = mul(x = x2_103_cast_fp16, y = const_52_promoted_to_fp16)[name = tensor("op_4473_cast_fp16")]; tensor var_4475_interleave_0 = const()[name = tensor("op_4475_interleave_0"), val = tensor(false)]; tensor var_4475_cast_fp16 = concat(axis = var_4396, interleave = var_4475_interleave_0, values = (var_4473_cast_fp16, x1_103_cast_fp16))[name = tensor("op_4475_cast_fp16")]; tensor var_4476_cast_fp16 = mul(x = var_4475_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_4476_cast_fp16")]; tensor k_103_cast_fp16 = add(x = var_4470_cast_fp16, y = var_4476_cast_fp16)[name = tensor("k_103_cast_fp16")]; tensor var_4479_cast_fp16 = mul(x = k_cache_51_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_4479_cast_fp16")]; tensor var_4480_cast_fp16 = mul(x = k_103_cast_fp16, y = var_107_to_fp16)[name = tensor("op_4480_cast_fp16")]; tensor k_full_51_cast_fp16 = add(x = var_4479_cast_fp16, y = var_4480_cast_fp16)[name = tensor("k_full_51_cast_fp16")]; tensor var_4483_cast_fp16 = mul(x = v_cache_51_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_4483_cast_fp16")]; tensor v_51_cast_fp16 = transpose(perm = v_51_perm_0, x = var_4458_cast_fp16)[name = tensor("transpose_9")]; tensor var_4484_cast_fp16 = mul(x = v_51_cast_fp16, y = var_107_to_fp16)[name = tensor("op_4484_cast_fp16")]; tensor v_full_51_cast_fp16 = add(x = var_4483_cast_fp16, y = var_4484_cast_fp16)[name = tensor("v_full_51_cast_fp16")]; tensor var_4486_axes_0 = const()[name = tensor("op_4486_axes_0"), val = tensor([2])]; tensor var_4486_cast_fp16 = expand_dims(axes = var_4486_axes_0, x = k_full_51_cast_fp16)[name = tensor("op_4486_cast_fp16")]; tensor var_4488_reps_0 = const()[name = tensor("op_4488_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_4488_cast_fp16 = tile(reps = var_4488_reps_0, x = var_4486_cast_fp16)[name = tensor("op_4488_cast_fp16")]; tensor var_4489 = const()[name = tensor("op_4489"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_51_cast_fp16 = reshape(shape = var_4489, x = var_4488_cast_fp16)[name = tensor("k_rep_51_cast_fp16")]; tensor var_4491_axes_0 = const()[name = tensor("op_4491_axes_0"), val = tensor([2])]; tensor var_4491_cast_fp16 = expand_dims(axes = var_4491_axes_0, x = v_full_51_cast_fp16)[name = tensor("op_4491_cast_fp16")]; tensor var_4493_reps_0 = const()[name = tensor("op_4493_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_4493_cast_fp16 = tile(reps = var_4493_reps_0, x = var_4491_cast_fp16)[name = tensor("op_4493_cast_fp16")]; tensor var_4494 = const()[name = tensor("op_4494"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_51_cast_fp16 = reshape(shape = var_4494, x = var_4493_cast_fp16)[name = tensor("v_rep_51_cast_fp16")]; tensor var_4497_transpose_x_1 = const()[name = tensor("op_4497_transpose_x_1"), val = tensor(false)]; tensor var_4497_transpose_y_1 = const()[name = tensor("op_4497_transpose_y_1"), val = tensor(true)]; tensor var_4497_cast_fp16 = matmul(transpose_x = var_4497_transpose_x_1, transpose_y = var_4497_transpose_y_1, x = q_103_cast_fp16, y = k_rep_51_cast_fp16)[name = tensor("op_4497_cast_fp16")]; tensor var_4498_to_fp16 = const()[name = tensor("op_4498_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_101_cast_fp16 = mul(x = var_4497_cast_fp16, y = var_4498_to_fp16)[name = tensor("attn_101_cast_fp16")]; tensor input_103_cast_fp16 = add(x = attn_101_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_103_cast_fp16")]; tensor input_103_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_103_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_103_cast_fp16_to_fp32 = cast(dtype = input_103_cast_fp16_to_fp32_dtype_0, x = input_103_cast_fp16)[name = tensor("cast_224")]; tensor attn_103 = softmax(axis = var_4396, x = input_103_cast_fp16_to_fp32)[name = tensor("attn_103")]; tensor out_51_transpose_x_0 = const()[name = tensor("out_51_transpose_x_0"), val = tensor(false)]; tensor out_51_transpose_y_0 = const()[name = tensor("out_51_transpose_y_0"), val = tensor(false)]; tensor attn_103_to_fp16_dtype_0 = const()[name = tensor("attn_103_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_103_to_fp16 = cast(dtype = attn_103_to_fp16_dtype_0, x = attn_103)[name = tensor("cast_223")]; tensor out_51_cast_fp16 = matmul(transpose_x = out_51_transpose_x_0, transpose_y = out_51_transpose_y_0, x = attn_103_to_fp16, y = v_rep_51_cast_fp16)[name = tensor("out_51_cast_fp16")]; tensor var_4503_perm_0 = const()[name = tensor("op_4503_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_4505 = const()[name = tensor("op_4505"), val = tensor([1, 1, 1536])]; tensor var_4503_cast_fp16 = transpose(perm = var_4503_perm_0, x = out_51_cast_fp16)[name = tensor("transpose_8")]; tensor x_1027_cast_fp16 = reshape(shape = var_4505, x = var_4503_cast_fp16)[name = tensor("x_1027_cast_fp16")]; tensor layers_25_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_25_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(447985792)))]; tensor linear_178_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_25_self_attn_o_proj_weight_to_fp16, x = x_1027_cast_fp16)[name = tensor("linear_178_cast_fp16")]; tensor x_1029_cast_fp16 = add(x = x_1001_cast_fp16, y = linear_178_cast_fp16)[name = tensor("x_1029_cast_fp16")]; tensor x_1029_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1029_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4395_promoted_3 = const()[name = tensor("op_4395_promoted_3"), val = tensor(0x1p+1)]; tensor x_1029_cast_fp16_to_fp32 = cast(dtype = x_1029_cast_fp16_to_fp32_dtype_0, x = x_1029_cast_fp16)[name = tensor("cast_222")]; tensor var_4516 = pow(x = x_1029_cast_fp16_to_fp32, y = var_4395_promoted_3)[name = tensor("op_4516")]; tensor var_207_axes_0 = const()[name = tensor("var_207_axes_0"), val = tensor([-1])]; tensor var_207_keep_dims_0 = const()[name = tensor("var_207_keep_dims_0"), val = tensor(true)]; tensor var_207 = reduce_mean(axes = var_207_axes_0, keep_dims = var_207_keep_dims_0, x = var_4516)[name = tensor("var_207")]; tensor var_207_to_fp16_dtype_0 = const()[name = tensor("var_207_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4520_to_fp16 = const()[name = tensor("op_4520_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_207_to_fp16 = cast(dtype = var_207_to_fp16_dtype_0, x = var_207)[name = tensor("cast_221")]; tensor var_4521_cast_fp16 = add(x = var_207_to_fp16, y = var_4520_to_fp16)[name = tensor("op_4521_cast_fp16")]; tensor var_4521_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4521_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4522_epsilon_0 = const()[name = tensor("op_4522_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4521_cast_fp16_to_fp32 = cast(dtype = var_4521_cast_fp16_to_fp32_dtype_0, x = var_4521_cast_fp16)[name = tensor("cast_220")]; tensor var_4522 = rsqrt(epsilon = var_4522_epsilon_0, x = var_4521_cast_fp16_to_fp32)[name = tensor("op_4522")]; tensor var_4522_to_fp16_dtype_0 = const()[name = tensor("op_4522_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4522_to_fp16 = cast(dtype = var_4522_to_fp16_dtype_0, x = var_4522)[name = tensor("cast_219")]; tensor x_1035_cast_fp16 = mul(x = x_1029_cast_fp16, y = var_4522_to_fp16)[name = tensor("x_1035_cast_fp16")]; tensor layers_25_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_25_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(449558720)))]; tensor x_1037_cast_fp16 = mul(x = layers_25_post_attention_layernorm_weight_to_fp16, y = x_1035_cast_fp16)[name = tensor("x_1037_cast_fp16")]; tensor layers_25_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_25_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(449559808)))]; tensor linear_179_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_25_mlp_gate_proj_weight_to_fp16, x = x_1037_cast_fp16)[name = tensor("linear_179_cast_fp16")]; tensor var_4533_cast_fp16 = silu(x = linear_179_cast_fp16)[name = tensor("op_4533_cast_fp16")]; tensor layers_25_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_25_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(451132736)))]; tensor linear_180_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_25_mlp_up_proj_weight_to_fp16, x = x_1037_cast_fp16)[name = tensor("linear_180_cast_fp16")]; tensor x_1039_cast_fp16 = mul(x = var_4533_cast_fp16, y = linear_180_cast_fp16)[name = tensor("x_1039_cast_fp16")]; tensor layers_25_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_25_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(452705664)))]; tensor linear_181_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_25_mlp_down_proj_weight_to_fp16, x = x_1039_cast_fp16)[name = tensor("linear_181_cast_fp16")]; tensor x_1041_cast_fp16 = add(x = x_1029_cast_fp16, y = linear_181_cast_fp16)[name = tensor("x_1041_cast_fp16")]; tensor x_1041_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1041_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_53_begin_0 = const()[name = tensor("k_cache_53_begin_0"), val = tensor([26, 0, 0, 0, 0])]; tensor k_cache_53_end_0 = const()[name = tensor("k_cache_53_end_0"), val = tensor([27, 1, 4, 2048, 128])]; tensor k_cache_53_end_mask_0 = const()[name = tensor("k_cache_53_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_53_squeeze_mask_0 = const()[name = tensor("k_cache_53_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_53_cast_fp16 = slice_by_index(begin = k_cache_53_begin_0, end = k_cache_53_end_0, end_mask = k_cache_53_end_mask_0, squeeze_mask = k_cache_53_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_53_cast_fp16")]; tensor v_cache_53_begin_0 = const()[name = tensor("v_cache_53_begin_0"), val = tensor([26, 0, 0, 0, 0])]; tensor v_cache_53_end_0 = const()[name = tensor("v_cache_53_end_0"), val = tensor([27, 1, 4, 2048, 128])]; tensor v_cache_53_end_mask_0 = const()[name = tensor("v_cache_53_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_53_squeeze_mask_0 = const()[name = tensor("v_cache_53_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_53_cast_fp16 = slice_by_index(begin = v_cache_53_begin_0, end = v_cache_53_end_0, end_mask = v_cache_53_end_mask_0, squeeze_mask = v_cache_53_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_53_cast_fp16")]; tensor var_4565 = const()[name = tensor("op_4565"), val = tensor(-1)]; tensor var_4564_promoted = const()[name = tensor("op_4564_promoted"), val = tensor(0x1p+1)]; tensor x_1041_cast_fp16_to_fp32 = cast(dtype = x_1041_cast_fp16_to_fp32_dtype_0, x = x_1041_cast_fp16)[name = tensor("cast_218")]; tensor var_4574 = pow(x = x_1041_cast_fp16_to_fp32, y = var_4564_promoted)[name = tensor("op_4574")]; tensor var_209_axes_0 = const()[name = tensor("var_209_axes_0"), val = tensor([-1])]; tensor var_209_keep_dims_0 = const()[name = tensor("var_209_keep_dims_0"), val = tensor(true)]; tensor var_209 = reduce_mean(axes = var_209_axes_0, keep_dims = var_209_keep_dims_0, x = var_4574)[name = tensor("var_209")]; tensor var_209_to_fp16_dtype_0 = const()[name = tensor("var_209_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4578_to_fp16 = const()[name = tensor("op_4578_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_209_to_fp16 = cast(dtype = var_209_to_fp16_dtype_0, x = var_209)[name = tensor("cast_217")]; tensor var_4579_cast_fp16 = add(x = var_209_to_fp16, y = var_4578_to_fp16)[name = tensor("op_4579_cast_fp16")]; tensor var_4579_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4579_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4580_epsilon_0 = const()[name = tensor("op_4580_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4579_cast_fp16_to_fp32 = cast(dtype = var_4579_cast_fp16_to_fp32_dtype_0, x = var_4579_cast_fp16)[name = tensor("cast_216")]; tensor var_4580 = rsqrt(epsilon = var_4580_epsilon_0, x = var_4579_cast_fp16_to_fp32)[name = tensor("op_4580")]; tensor var_4580_to_fp16_dtype_0 = const()[name = tensor("op_4580_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4580_to_fp16 = cast(dtype = var_4580_to_fp16_dtype_0, x = var_4580)[name = tensor("cast_215")]; tensor x_1047_cast_fp16 = mul(x = x_1041_cast_fp16, y = var_4580_to_fp16)[name = tensor("x_1047_cast_fp16")]; tensor layers_26_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_26_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(454278592)))]; tensor x_1049_cast_fp16 = mul(x = layers_26_input_layernorm_weight_to_fp16, y = x_1047_cast_fp16)[name = tensor("x_1049_cast_fp16")]; tensor layers_26_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_26_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(454279680)))]; tensor linear_182_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_26_self_attn_q_proj_weight_to_fp16, x = x_1049_cast_fp16)[name = tensor("linear_182_cast_fp16")]; tensor var_4594 = const()[name = tensor("op_4594"), val = tensor([1, 1, 12, 128])]; tensor x_1051_cast_fp16 = reshape(shape = var_4594, x = linear_182_cast_fp16)[name = tensor("x_1051_cast_fp16")]; tensor x_1051_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1051_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4564_promoted_1 = const()[name = tensor("op_4564_promoted_1"), val = tensor(0x1p+1)]; tensor x_1051_cast_fp16_to_fp32 = cast(dtype = x_1051_cast_fp16_to_fp32_dtype_0, x = x_1051_cast_fp16)[name = tensor("cast_214")]; tensor var_4598 = pow(x = x_1051_cast_fp16_to_fp32, y = var_4564_promoted_1)[name = tensor("op_4598")]; tensor var_211_axes_0 = const()[name = tensor("var_211_axes_0"), val = tensor([-1])]; tensor var_211_keep_dims_0 = const()[name = tensor("var_211_keep_dims_0"), val = tensor(true)]; tensor var_211 = reduce_mean(axes = var_211_axes_0, keep_dims = var_211_keep_dims_0, x = var_4598)[name = tensor("var_211")]; tensor var_211_to_fp16_dtype_0 = const()[name = tensor("var_211_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4602_to_fp16 = const()[name = tensor("op_4602_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_211_to_fp16 = cast(dtype = var_211_to_fp16_dtype_0, x = var_211)[name = tensor("cast_213")]; tensor var_4603_cast_fp16 = add(x = var_211_to_fp16, y = var_4602_to_fp16)[name = tensor("op_4603_cast_fp16")]; tensor var_4603_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4603_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4604_epsilon_0 = const()[name = tensor("op_4604_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4603_cast_fp16_to_fp32 = cast(dtype = var_4603_cast_fp16_to_fp32_dtype_0, x = var_4603_cast_fp16)[name = tensor("cast_212")]; tensor var_4604 = rsqrt(epsilon = var_4604_epsilon_0, x = var_4603_cast_fp16_to_fp32)[name = tensor("op_4604")]; tensor var_4604_to_fp16_dtype_0 = const()[name = tensor("op_4604_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4604_to_fp16 = cast(dtype = var_4604_to_fp16_dtype_0, x = var_4604)[name = tensor("cast_211")]; tensor x_1057_cast_fp16 = mul(x = x_1051_cast_fp16, y = var_4604_to_fp16)[name = tensor("x_1057_cast_fp16")]; tensor layers_26_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_26_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(455852608)))]; tensor var_4606_cast_fp16 = mul(x = layers_26_self_attn_q_norm_weight_to_fp16, y = x_1057_cast_fp16)[name = tensor("op_4606_cast_fp16")]; tensor q_105_perm_0 = const()[name = tensor("q_105_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_26_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_26_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(455852928)))]; tensor linear_183_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_26_self_attn_k_proj_weight_to_fp16, x = x_1049_cast_fp16)[name = tensor("linear_183_cast_fp16")]; tensor var_4610 = const()[name = tensor("op_4610"), val = tensor([1, 1, 4, 128])]; tensor x_1059_cast_fp16 = reshape(shape = var_4610, x = linear_183_cast_fp16)[name = tensor("x_1059_cast_fp16")]; tensor x_1059_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1059_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4564_promoted_2 = const()[name = tensor("op_4564_promoted_2"), val = tensor(0x1p+1)]; tensor x_1059_cast_fp16_to_fp32 = cast(dtype = x_1059_cast_fp16_to_fp32_dtype_0, x = x_1059_cast_fp16)[name = tensor("cast_210")]; tensor var_4614 = pow(x = x_1059_cast_fp16_to_fp32, y = var_4564_promoted_2)[name = tensor("op_4614")]; tensor var_213_axes_0 = const()[name = tensor("var_213_axes_0"), val = tensor([-1])]; tensor var_213_keep_dims_0 = const()[name = tensor("var_213_keep_dims_0"), val = tensor(true)]; tensor var_213 = reduce_mean(axes = var_213_axes_0, keep_dims = var_213_keep_dims_0, x = var_4614)[name = tensor("var_213")]; tensor var_213_to_fp16_dtype_0 = const()[name = tensor("var_213_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4618_to_fp16 = const()[name = tensor("op_4618_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_213_to_fp16 = cast(dtype = var_213_to_fp16_dtype_0, x = var_213)[name = tensor("cast_209")]; tensor var_4619_cast_fp16 = add(x = var_213_to_fp16, y = var_4618_to_fp16)[name = tensor("op_4619_cast_fp16")]; tensor var_4619_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4619_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4620_epsilon_0 = const()[name = tensor("op_4620_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4619_cast_fp16_to_fp32 = cast(dtype = var_4619_cast_fp16_to_fp32_dtype_0, x = var_4619_cast_fp16)[name = tensor("cast_208")]; tensor var_4620 = rsqrt(epsilon = var_4620_epsilon_0, x = var_4619_cast_fp16_to_fp32)[name = tensor("op_4620")]; tensor var_4620_to_fp16_dtype_0 = const()[name = tensor("op_4620_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4620_to_fp16 = cast(dtype = var_4620_to_fp16_dtype_0, x = var_4620)[name = tensor("cast_207")]; tensor x_1065_cast_fp16 = mul(x = x_1059_cast_fp16, y = var_4620_to_fp16)[name = tensor("x_1065_cast_fp16")]; tensor layers_26_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_26_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(456377280)))]; tensor var_4622_cast_fp16 = mul(x = layers_26_self_attn_k_norm_weight_to_fp16, y = x_1065_cast_fp16)[name = tensor("op_4622_cast_fp16")]; tensor k_105_perm_0 = const()[name = tensor("k_105_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_26_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_26_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(456377600)))]; tensor linear_184_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_26_self_attn_v_proj_weight_to_fp16, x = x_1049_cast_fp16)[name = tensor("linear_184_cast_fp16")]; tensor var_4626 = const()[name = tensor("op_4626"), val = tensor([1, 1, 4, 128])]; tensor var_4627_cast_fp16 = reshape(shape = var_4626, x = linear_184_cast_fp16)[name = tensor("op_4627_cast_fp16")]; tensor v_53_perm_0 = const()[name = tensor("v_53_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_105_cast_fp16 = transpose(perm = q_105_perm_0, x = var_4606_cast_fp16)[name = tensor("transpose_7")]; tensor var_4631_cast_fp16 = mul(x = q_105_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_4631_cast_fp16")]; tensor x1_105_begin_0 = const()[name = tensor("x1_105_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_105_end_0 = const()[name = tensor("x1_105_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_105_end_mask_0 = const()[name = tensor("x1_105_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_105_cast_fp16 = slice_by_index(begin = x1_105_begin_0, end = x1_105_end_0, end_mask = x1_105_end_mask_0, x = q_105_cast_fp16)[name = tensor("x1_105_cast_fp16")]; tensor x2_105_begin_0 = const()[name = tensor("x2_105_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_105_end_0 = const()[name = tensor("x2_105_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_105_end_mask_0 = const()[name = tensor("x2_105_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_105_cast_fp16 = slice_by_index(begin = x2_105_begin_0, end = x2_105_end_0, end_mask = x2_105_end_mask_0, x = q_105_cast_fp16)[name = tensor("x2_105_cast_fp16")]; tensor const_53_promoted_to_fp16 = const()[name = tensor("const_53_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_4634_cast_fp16 = mul(x = x2_105_cast_fp16, y = const_53_promoted_to_fp16)[name = tensor("op_4634_cast_fp16")]; tensor var_4636_interleave_0 = const()[name = tensor("op_4636_interleave_0"), val = tensor(false)]; tensor var_4636_cast_fp16 = concat(axis = var_4565, interleave = var_4636_interleave_0, values = (var_4634_cast_fp16, x1_105_cast_fp16))[name = tensor("op_4636_cast_fp16")]; tensor var_4637_cast_fp16 = mul(x = var_4636_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_4637_cast_fp16")]; tensor q_107_cast_fp16 = add(x = var_4631_cast_fp16, y = var_4637_cast_fp16)[name = tensor("q_107_cast_fp16")]; tensor k_105_cast_fp16 = transpose(perm = k_105_perm_0, x = var_4622_cast_fp16)[name = tensor("transpose_6")]; tensor var_4639_cast_fp16 = mul(x = k_105_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_4639_cast_fp16")]; tensor x1_107_begin_0 = const()[name = tensor("x1_107_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_107_end_0 = const()[name = tensor("x1_107_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_107_end_mask_0 = const()[name = tensor("x1_107_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_107_cast_fp16 = slice_by_index(begin = x1_107_begin_0, end = x1_107_end_0, end_mask = x1_107_end_mask_0, x = k_105_cast_fp16)[name = tensor("x1_107_cast_fp16")]; tensor x2_107_begin_0 = const()[name = tensor("x2_107_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_107_end_0 = const()[name = tensor("x2_107_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_107_end_mask_0 = const()[name = tensor("x2_107_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_107_cast_fp16 = slice_by_index(begin = x2_107_begin_0, end = x2_107_end_0, end_mask = x2_107_end_mask_0, x = k_105_cast_fp16)[name = tensor("x2_107_cast_fp16")]; tensor const_54_promoted_to_fp16 = const()[name = tensor("const_54_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_4642_cast_fp16 = mul(x = x2_107_cast_fp16, y = const_54_promoted_to_fp16)[name = tensor("op_4642_cast_fp16")]; tensor var_4644_interleave_0 = const()[name = tensor("op_4644_interleave_0"), val = tensor(false)]; tensor var_4644_cast_fp16 = concat(axis = var_4565, interleave = var_4644_interleave_0, values = (var_4642_cast_fp16, x1_107_cast_fp16))[name = tensor("op_4644_cast_fp16")]; tensor var_4645_cast_fp16 = mul(x = var_4644_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_4645_cast_fp16")]; tensor k_107_cast_fp16 = add(x = var_4639_cast_fp16, y = var_4645_cast_fp16)[name = tensor("k_107_cast_fp16")]; tensor var_4648_cast_fp16 = mul(x = k_cache_53_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_4648_cast_fp16")]; tensor var_4649_cast_fp16 = mul(x = k_107_cast_fp16, y = var_107_to_fp16)[name = tensor("op_4649_cast_fp16")]; tensor k_full_53_cast_fp16 = add(x = var_4648_cast_fp16, y = var_4649_cast_fp16)[name = tensor("k_full_53_cast_fp16")]; tensor var_4652_cast_fp16 = mul(x = v_cache_53_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_4652_cast_fp16")]; tensor v_53_cast_fp16 = transpose(perm = v_53_perm_0, x = var_4627_cast_fp16)[name = tensor("transpose_5")]; tensor var_4653_cast_fp16 = mul(x = v_53_cast_fp16, y = var_107_to_fp16)[name = tensor("op_4653_cast_fp16")]; tensor v_full_53_cast_fp16 = add(x = var_4652_cast_fp16, y = var_4653_cast_fp16)[name = tensor("v_full_53_cast_fp16")]; tensor var_4655_axes_0 = const()[name = tensor("op_4655_axes_0"), val = tensor([2])]; tensor var_4655_cast_fp16 = expand_dims(axes = var_4655_axes_0, x = k_full_53_cast_fp16)[name = tensor("op_4655_cast_fp16")]; tensor var_4657_reps_0 = const()[name = tensor("op_4657_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_4657_cast_fp16 = tile(reps = var_4657_reps_0, x = var_4655_cast_fp16)[name = tensor("op_4657_cast_fp16")]; tensor var_4658 = const()[name = tensor("op_4658"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_53_cast_fp16 = reshape(shape = var_4658, x = var_4657_cast_fp16)[name = tensor("k_rep_53_cast_fp16")]; tensor var_4660_axes_0 = const()[name = tensor("op_4660_axes_0"), val = tensor([2])]; tensor var_4660_cast_fp16 = expand_dims(axes = var_4660_axes_0, x = v_full_53_cast_fp16)[name = tensor("op_4660_cast_fp16")]; tensor var_4662_reps_0 = const()[name = tensor("op_4662_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_4662_cast_fp16 = tile(reps = var_4662_reps_0, x = var_4660_cast_fp16)[name = tensor("op_4662_cast_fp16")]; tensor var_4663 = const()[name = tensor("op_4663"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_53_cast_fp16 = reshape(shape = var_4663, x = var_4662_cast_fp16)[name = tensor("v_rep_53_cast_fp16")]; tensor var_4666_transpose_x_1 = const()[name = tensor("op_4666_transpose_x_1"), val = tensor(false)]; tensor var_4666_transpose_y_1 = const()[name = tensor("op_4666_transpose_y_1"), val = tensor(true)]; tensor var_4666_cast_fp16 = matmul(transpose_x = var_4666_transpose_x_1, transpose_y = var_4666_transpose_y_1, x = q_107_cast_fp16, y = k_rep_53_cast_fp16)[name = tensor("op_4666_cast_fp16")]; tensor var_4667_to_fp16 = const()[name = tensor("op_4667_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_105_cast_fp16 = mul(x = var_4666_cast_fp16, y = var_4667_to_fp16)[name = tensor("attn_105_cast_fp16")]; tensor input_107_cast_fp16 = add(x = attn_105_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_107_cast_fp16")]; tensor input_107_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_107_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_107_cast_fp16_to_fp32 = cast(dtype = input_107_cast_fp16_to_fp32_dtype_0, x = input_107_cast_fp16)[name = tensor("cast_206")]; tensor attn_107 = softmax(axis = var_4565, x = input_107_cast_fp16_to_fp32)[name = tensor("attn_107")]; tensor out_53_transpose_x_0 = const()[name = tensor("out_53_transpose_x_0"), val = tensor(false)]; tensor out_53_transpose_y_0 = const()[name = tensor("out_53_transpose_y_0"), val = tensor(false)]; tensor attn_107_to_fp16_dtype_0 = const()[name = tensor("attn_107_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_107_to_fp16 = cast(dtype = attn_107_to_fp16_dtype_0, x = attn_107)[name = tensor("cast_205")]; tensor out_53_cast_fp16 = matmul(transpose_x = out_53_transpose_x_0, transpose_y = out_53_transpose_y_0, x = attn_107_to_fp16, y = v_rep_53_cast_fp16)[name = tensor("out_53_cast_fp16")]; tensor var_4672_perm_0 = const()[name = tensor("op_4672_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_4674 = const()[name = tensor("op_4674"), val = tensor([1, 1, 1536])]; tensor var_4672_cast_fp16 = transpose(perm = var_4672_perm_0, x = out_53_cast_fp16)[name = tensor("transpose_4")]; tensor x_1067_cast_fp16 = reshape(shape = var_4674, x = var_4672_cast_fp16)[name = tensor("x_1067_cast_fp16")]; tensor layers_26_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_26_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(456901952)))]; tensor linear_185_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_26_self_attn_o_proj_weight_to_fp16, x = x_1067_cast_fp16)[name = tensor("linear_185_cast_fp16")]; tensor x_1069_cast_fp16 = add(x = x_1041_cast_fp16, y = linear_185_cast_fp16)[name = tensor("x_1069_cast_fp16")]; tensor x_1069_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1069_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4564_promoted_3 = const()[name = tensor("op_4564_promoted_3"), val = tensor(0x1p+1)]; tensor x_1069_cast_fp16_to_fp32 = cast(dtype = x_1069_cast_fp16_to_fp32_dtype_0, x = x_1069_cast_fp16)[name = tensor("cast_204")]; tensor var_4685 = pow(x = x_1069_cast_fp16_to_fp32, y = var_4564_promoted_3)[name = tensor("op_4685")]; tensor var_215_axes_0 = const()[name = tensor("var_215_axes_0"), val = tensor([-1])]; tensor var_215_keep_dims_0 = const()[name = tensor("var_215_keep_dims_0"), val = tensor(true)]; tensor var_215 = reduce_mean(axes = var_215_axes_0, keep_dims = var_215_keep_dims_0, x = var_4685)[name = tensor("var_215")]; tensor var_215_to_fp16_dtype_0 = const()[name = tensor("var_215_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4689_to_fp16 = const()[name = tensor("op_4689_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_215_to_fp16 = cast(dtype = var_215_to_fp16_dtype_0, x = var_215)[name = tensor("cast_203")]; tensor var_4690_cast_fp16 = add(x = var_215_to_fp16, y = var_4689_to_fp16)[name = tensor("op_4690_cast_fp16")]; tensor var_4690_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4690_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4691_epsilon_0 = const()[name = tensor("op_4691_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4690_cast_fp16_to_fp32 = cast(dtype = var_4690_cast_fp16_to_fp32_dtype_0, x = var_4690_cast_fp16)[name = tensor("cast_202")]; tensor var_4691 = rsqrt(epsilon = var_4691_epsilon_0, x = var_4690_cast_fp16_to_fp32)[name = tensor("op_4691")]; tensor var_4691_to_fp16_dtype_0 = const()[name = tensor("op_4691_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4691_to_fp16 = cast(dtype = var_4691_to_fp16_dtype_0, x = var_4691)[name = tensor("cast_201")]; tensor x_1075_cast_fp16 = mul(x = x_1069_cast_fp16, y = var_4691_to_fp16)[name = tensor("x_1075_cast_fp16")]; tensor layers_26_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_26_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(458474880)))]; tensor x_1077_cast_fp16 = mul(x = layers_26_post_attention_layernorm_weight_to_fp16, y = x_1075_cast_fp16)[name = tensor("x_1077_cast_fp16")]; tensor layers_26_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_26_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(458475968)))]; tensor linear_186_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_26_mlp_gate_proj_weight_to_fp16, x = x_1077_cast_fp16)[name = tensor("linear_186_cast_fp16")]; tensor var_4702_cast_fp16 = silu(x = linear_186_cast_fp16)[name = tensor("op_4702_cast_fp16")]; tensor layers_26_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_26_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(460048896)))]; tensor linear_187_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_26_mlp_up_proj_weight_to_fp16, x = x_1077_cast_fp16)[name = tensor("linear_187_cast_fp16")]; tensor x_1079_cast_fp16 = mul(x = var_4702_cast_fp16, y = linear_187_cast_fp16)[name = tensor("x_1079_cast_fp16")]; tensor layers_26_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_26_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(461621824)))]; tensor linear_188_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_26_mlp_down_proj_weight_to_fp16, x = x_1079_cast_fp16)[name = tensor("linear_188_cast_fp16")]; tensor x_1081_cast_fp16 = add(x = x_1069_cast_fp16, y = linear_188_cast_fp16)[name = tensor("x_1081_cast_fp16")]; tensor x_1081_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1081_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor k_cache_begin_0 = const()[name = tensor("k_cache_begin_0"), val = tensor([27, 0, 0, 0, 0])]; tensor k_cache_end_0 = const()[name = tensor("k_cache_end_0"), val = tensor([28, 1, 4, 2048, 128])]; tensor k_cache_end_mask_0 = const()[name = tensor("k_cache_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor k_cache_squeeze_mask_0 = const()[name = tensor("k_cache_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor k_cache_cast_fp16 = slice_by_index(begin = k_cache_begin_0, end = k_cache_end_0, end_mask = k_cache_end_mask_0, squeeze_mask = k_cache_squeeze_mask_0, x = kv_k_to_fp16)[name = tensor("k_cache_cast_fp16")]; tensor v_cache_begin_0 = const()[name = tensor("v_cache_begin_0"), val = tensor([27, 0, 0, 0, 0])]; tensor v_cache_end_0 = const()[name = tensor("v_cache_end_0"), val = tensor([28, 1, 4, 2048, 128])]; tensor v_cache_end_mask_0 = const()[name = tensor("v_cache_end_mask_0"), val = tensor([false, true, true, true, true])]; tensor v_cache_squeeze_mask_0 = const()[name = tensor("v_cache_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; tensor v_cache_cast_fp16 = slice_by_index(begin = v_cache_begin_0, end = v_cache_end_0, end_mask = v_cache_end_mask_0, squeeze_mask = v_cache_squeeze_mask_0, x = kv_v_to_fp16)[name = tensor("v_cache_cast_fp16")]; tensor var_4734 = const()[name = tensor("op_4734"), val = tensor(-1)]; tensor var_4733_promoted = const()[name = tensor("op_4733_promoted"), val = tensor(0x1p+1)]; tensor x_1081_cast_fp16_to_fp32 = cast(dtype = x_1081_cast_fp16_to_fp32_dtype_0, x = x_1081_cast_fp16)[name = tensor("cast_200")]; tensor var_4743 = pow(x = x_1081_cast_fp16_to_fp32, y = var_4733_promoted)[name = tensor("op_4743")]; tensor var_217_axes_0 = const()[name = tensor("var_217_axes_0"), val = tensor([-1])]; tensor var_217_keep_dims_0 = const()[name = tensor("var_217_keep_dims_0"), val = tensor(true)]; tensor var_217 = reduce_mean(axes = var_217_axes_0, keep_dims = var_217_keep_dims_0, x = var_4743)[name = tensor("var_217")]; tensor var_217_to_fp16_dtype_0 = const()[name = tensor("var_217_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4747_to_fp16 = const()[name = tensor("op_4747_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_217_to_fp16 = cast(dtype = var_217_to_fp16_dtype_0, x = var_217)[name = tensor("cast_199")]; tensor var_4748_cast_fp16 = add(x = var_217_to_fp16, y = var_4747_to_fp16)[name = tensor("op_4748_cast_fp16")]; tensor var_4748_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4748_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4749_epsilon_0 = const()[name = tensor("op_4749_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4748_cast_fp16_to_fp32 = cast(dtype = var_4748_cast_fp16_to_fp32_dtype_0, x = var_4748_cast_fp16)[name = tensor("cast_198")]; tensor var_4749 = rsqrt(epsilon = var_4749_epsilon_0, x = var_4748_cast_fp16_to_fp32)[name = tensor("op_4749")]; tensor var_4749_to_fp16_dtype_0 = const()[name = tensor("op_4749_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4749_to_fp16 = cast(dtype = var_4749_to_fp16_dtype_0, x = var_4749)[name = tensor("cast_197")]; tensor x_1087_cast_fp16 = mul(x = x_1081_cast_fp16, y = var_4749_to_fp16)[name = tensor("x_1087_cast_fp16")]; tensor layers_27_input_layernorm_weight_to_fp16 = const()[name = tensor("layers_27_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(463194752)))]; tensor x_1089_cast_fp16 = mul(x = layers_27_input_layernorm_weight_to_fp16, y = x_1087_cast_fp16)[name = tensor("x_1089_cast_fp16")]; tensor layers_27_self_attn_q_proj_weight_to_fp16 = const()[name = tensor("layers_27_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(463195840)))]; tensor linear_189_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_27_self_attn_q_proj_weight_to_fp16, x = x_1089_cast_fp16)[name = tensor("linear_189_cast_fp16")]; tensor var_4763 = const()[name = tensor("op_4763"), val = tensor([1, 1, 12, 128])]; tensor x_1091_cast_fp16 = reshape(shape = var_4763, x = linear_189_cast_fp16)[name = tensor("x_1091_cast_fp16")]; tensor x_1091_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1091_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4733_promoted_1 = const()[name = tensor("op_4733_promoted_1"), val = tensor(0x1p+1)]; tensor x_1091_cast_fp16_to_fp32 = cast(dtype = x_1091_cast_fp16_to_fp32_dtype_0, x = x_1091_cast_fp16)[name = tensor("cast_196")]; tensor var_4767 = pow(x = x_1091_cast_fp16_to_fp32, y = var_4733_promoted_1)[name = tensor("op_4767")]; tensor var_219_axes_0 = const()[name = tensor("var_219_axes_0"), val = tensor([-1])]; tensor var_219_keep_dims_0 = const()[name = tensor("var_219_keep_dims_0"), val = tensor(true)]; tensor var_219 = reduce_mean(axes = var_219_axes_0, keep_dims = var_219_keep_dims_0, x = var_4767)[name = tensor("var_219")]; tensor var_219_to_fp16_dtype_0 = const()[name = tensor("var_219_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4771_to_fp16 = const()[name = tensor("op_4771_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_219_to_fp16 = cast(dtype = var_219_to_fp16_dtype_0, x = var_219)[name = tensor("cast_195")]; tensor var_4772_cast_fp16 = add(x = var_219_to_fp16, y = var_4771_to_fp16)[name = tensor("op_4772_cast_fp16")]; tensor var_4772_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4772_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4773_epsilon_0 = const()[name = tensor("op_4773_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4772_cast_fp16_to_fp32 = cast(dtype = var_4772_cast_fp16_to_fp32_dtype_0, x = var_4772_cast_fp16)[name = tensor("cast_194")]; tensor var_4773 = rsqrt(epsilon = var_4773_epsilon_0, x = var_4772_cast_fp16_to_fp32)[name = tensor("op_4773")]; tensor var_4773_to_fp16_dtype_0 = const()[name = tensor("op_4773_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4773_to_fp16 = cast(dtype = var_4773_to_fp16_dtype_0, x = var_4773)[name = tensor("cast_193")]; tensor x_1097_cast_fp16 = mul(x = x_1091_cast_fp16, y = var_4773_to_fp16)[name = tensor("x_1097_cast_fp16")]; tensor layers_27_self_attn_q_norm_weight_to_fp16 = const()[name = tensor("layers_27_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(464768768)))]; tensor var_4775_cast_fp16 = mul(x = layers_27_self_attn_q_norm_weight_to_fp16, y = x_1097_cast_fp16)[name = tensor("op_4775_cast_fp16")]; tensor q_109_perm_0 = const()[name = tensor("q_109_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_27_self_attn_k_proj_weight_to_fp16 = const()[name = tensor("layers_27_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(464769088)))]; tensor linear_190_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_27_self_attn_k_proj_weight_to_fp16, x = x_1089_cast_fp16)[name = tensor("linear_190_cast_fp16")]; tensor var_4779 = const()[name = tensor("op_4779"), val = tensor([1, 1, 4, 128])]; tensor x_1099_cast_fp16 = reshape(shape = var_4779, x = linear_190_cast_fp16)[name = tensor("x_1099_cast_fp16")]; tensor x_1099_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1099_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4733_promoted_2 = const()[name = tensor("op_4733_promoted_2"), val = tensor(0x1p+1)]; tensor x_1099_cast_fp16_to_fp32 = cast(dtype = x_1099_cast_fp16_to_fp32_dtype_0, x = x_1099_cast_fp16)[name = tensor("cast_192")]; tensor var_4783 = pow(x = x_1099_cast_fp16_to_fp32, y = var_4733_promoted_2)[name = tensor("op_4783")]; tensor var_221_axes_0 = const()[name = tensor("var_221_axes_0"), val = tensor([-1])]; tensor var_221_keep_dims_0 = const()[name = tensor("var_221_keep_dims_0"), val = tensor(true)]; tensor var_221 = reduce_mean(axes = var_221_axes_0, keep_dims = var_221_keep_dims_0, x = var_4783)[name = tensor("var_221")]; tensor var_221_to_fp16_dtype_0 = const()[name = tensor("var_221_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4787_to_fp16 = const()[name = tensor("op_4787_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_221_to_fp16 = cast(dtype = var_221_to_fp16_dtype_0, x = var_221)[name = tensor("cast_191")]; tensor var_4788_cast_fp16 = add(x = var_221_to_fp16, y = var_4787_to_fp16)[name = tensor("op_4788_cast_fp16")]; tensor var_4788_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4788_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4789_epsilon_0 = const()[name = tensor("op_4789_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4788_cast_fp16_to_fp32 = cast(dtype = var_4788_cast_fp16_to_fp32_dtype_0, x = var_4788_cast_fp16)[name = tensor("cast_190")]; tensor var_4789 = rsqrt(epsilon = var_4789_epsilon_0, x = var_4788_cast_fp16_to_fp32)[name = tensor("op_4789")]; tensor var_4789_to_fp16_dtype_0 = const()[name = tensor("op_4789_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4789_to_fp16 = cast(dtype = var_4789_to_fp16_dtype_0, x = var_4789)[name = tensor("cast_189")]; tensor x_1105_cast_fp16 = mul(x = x_1099_cast_fp16, y = var_4789_to_fp16)[name = tensor("x_1105_cast_fp16")]; tensor layers_27_self_attn_k_norm_weight_to_fp16 = const()[name = tensor("layers_27_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(465293440)))]; tensor var_4791_cast_fp16 = mul(x = layers_27_self_attn_k_norm_weight_to_fp16, y = x_1105_cast_fp16)[name = tensor("op_4791_cast_fp16")]; tensor k_109_perm_0 = const()[name = tensor("k_109_perm_0"), val = tensor([0, 2, 1, 3])]; tensor layers_27_self_attn_v_proj_weight_to_fp16 = const()[name = tensor("layers_27_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(465293760)))]; tensor linear_191_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_27_self_attn_v_proj_weight_to_fp16, x = x_1089_cast_fp16)[name = tensor("linear_191_cast_fp16")]; tensor var_4795 = const()[name = tensor("op_4795"), val = tensor([1, 1, 4, 128])]; tensor var_4796_cast_fp16 = reshape(shape = var_4795, x = linear_191_cast_fp16)[name = tensor("op_4796_cast_fp16")]; tensor v_perm_0 = const()[name = tensor("v_perm_0"), val = tensor([0, 2, 1, 3])]; tensor q_109_cast_fp16 = transpose(perm = q_109_perm_0, x = var_4775_cast_fp16)[name = tensor("transpose_3")]; tensor var_4800_cast_fp16 = mul(x = q_109_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_4800_cast_fp16")]; tensor x1_109_begin_0 = const()[name = tensor("x1_109_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_109_end_0 = const()[name = tensor("x1_109_end_0"), val = tensor([1, 12, 1, 64])]; tensor x1_109_end_mask_0 = const()[name = tensor("x1_109_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_109_cast_fp16 = slice_by_index(begin = x1_109_begin_0, end = x1_109_end_0, end_mask = x1_109_end_mask_0, x = q_109_cast_fp16)[name = tensor("x1_109_cast_fp16")]; tensor x2_109_begin_0 = const()[name = tensor("x2_109_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_109_end_0 = const()[name = tensor("x2_109_end_0"), val = tensor([1, 12, 1, 128])]; tensor x2_109_end_mask_0 = const()[name = tensor("x2_109_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_109_cast_fp16 = slice_by_index(begin = x2_109_begin_0, end = x2_109_end_0, end_mask = x2_109_end_mask_0, x = q_109_cast_fp16)[name = tensor("x2_109_cast_fp16")]; tensor const_55_promoted_to_fp16 = const()[name = tensor("const_55_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_4803_cast_fp16 = mul(x = x2_109_cast_fp16, y = const_55_promoted_to_fp16)[name = tensor("op_4803_cast_fp16")]; tensor var_4805_interleave_0 = const()[name = tensor("op_4805_interleave_0"), val = tensor(false)]; tensor var_4805_cast_fp16 = concat(axis = var_4734, interleave = var_4805_interleave_0, values = (var_4803_cast_fp16, x1_109_cast_fp16))[name = tensor("op_4805_cast_fp16")]; tensor var_4806_cast_fp16 = mul(x = var_4805_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_4806_cast_fp16")]; tensor q_cast_fp16 = add(x = var_4800_cast_fp16, y = var_4806_cast_fp16)[name = tensor("q_cast_fp16")]; tensor k_109_cast_fp16 = transpose(perm = k_109_perm_0, x = var_4791_cast_fp16)[name = tensor("transpose_2")]; tensor var_4808_cast_fp16 = mul(x = k_109_cast_fp16, y = cos_3_cast_fp16)[name = tensor("op_4808_cast_fp16")]; tensor x1_begin_0 = const()[name = tensor("x1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_end_0 = const()[name = tensor("x1_end_0"), val = tensor([1, 4, 1, 64])]; tensor x1_end_mask_0 = const()[name = tensor("x1_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_cast_fp16 = slice_by_index(begin = x1_begin_0, end = x1_end_0, end_mask = x1_end_mask_0, x = k_109_cast_fp16)[name = tensor("x1_cast_fp16")]; tensor x2_begin_0 = const()[name = tensor("x2_begin_0"), val = tensor([0, 0, 0, 64])]; tensor x2_end_0 = const()[name = tensor("x2_end_0"), val = tensor([1, 4, 1, 128])]; tensor x2_end_mask_0 = const()[name = tensor("x2_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_cast_fp16 = slice_by_index(begin = x2_begin_0, end = x2_end_0, end_mask = x2_end_mask_0, x = k_109_cast_fp16)[name = tensor("x2_cast_fp16")]; tensor const_56_promoted_to_fp16 = const()[name = tensor("const_56_promoted_to_fp16"), val = tensor(-0x1p+0)]; tensor var_4811_cast_fp16 = mul(x = x2_cast_fp16, y = const_56_promoted_to_fp16)[name = tensor("op_4811_cast_fp16")]; tensor var_4813_interleave_0 = const()[name = tensor("op_4813_interleave_0"), val = tensor(false)]; tensor var_4813_cast_fp16 = concat(axis = var_4734, interleave = var_4813_interleave_0, values = (var_4811_cast_fp16, x1_cast_fp16))[name = tensor("op_4813_cast_fp16")]; tensor var_4814_cast_fp16 = mul(x = var_4813_cast_fp16, y = sin_3_cast_fp16)[name = tensor("op_4814_cast_fp16")]; tensor k_cast_fp16 = add(x = var_4808_cast_fp16, y = var_4814_cast_fp16)[name = tensor("k_cast_fp16")]; tensor var_4817_cast_fp16 = mul(x = k_cache_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_4817_cast_fp16")]; tensor var_4818_cast_fp16 = mul(x = k_cast_fp16, y = var_107_to_fp16)[name = tensor("op_4818_cast_fp16")]; tensor k_full_cast_fp16 = add(x = var_4817_cast_fp16, y = var_4818_cast_fp16)[name = tensor("k_full_cast_fp16")]; tensor var_4821_cast_fp16 = mul(x = v_cache_cast_fp16, y = var_253_cast_fp16)[name = tensor("op_4821_cast_fp16")]; tensor v_cast_fp16 = transpose(perm = v_perm_0, x = var_4796_cast_fp16)[name = tensor("transpose_1")]; tensor var_4822_cast_fp16 = mul(x = v_cast_fp16, y = var_107_to_fp16)[name = tensor("op_4822_cast_fp16")]; tensor v_full_cast_fp16 = add(x = var_4821_cast_fp16, y = var_4822_cast_fp16)[name = tensor("v_full_cast_fp16")]; tensor var_4824_axes_0 = const()[name = tensor("op_4824_axes_0"), val = tensor([2])]; tensor var_4824_cast_fp16 = expand_dims(axes = var_4824_axes_0, x = k_full_cast_fp16)[name = tensor("op_4824_cast_fp16")]; tensor var_4826_reps_0 = const()[name = tensor("op_4826_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_4826_cast_fp16 = tile(reps = var_4826_reps_0, x = var_4824_cast_fp16)[name = tensor("op_4826_cast_fp16")]; tensor var_4827 = const()[name = tensor("op_4827"), val = tensor([1, 12, 2048, 128])]; tensor k_rep_cast_fp16 = reshape(shape = var_4827, x = var_4826_cast_fp16)[name = tensor("k_rep_cast_fp16")]; tensor var_4829_axes_0 = const()[name = tensor("op_4829_axes_0"), val = tensor([2])]; tensor var_4829_cast_fp16 = expand_dims(axes = var_4829_axes_0, x = v_full_cast_fp16)[name = tensor("op_4829_cast_fp16")]; tensor var_4831_reps_0 = const()[name = tensor("op_4831_reps_0"), val = tensor([1, 1, 3, 1, 1])]; tensor var_4831_cast_fp16 = tile(reps = var_4831_reps_0, x = var_4829_cast_fp16)[name = tensor("op_4831_cast_fp16")]; tensor var_4832 = const()[name = tensor("op_4832"), val = tensor([1, 12, 2048, 128])]; tensor v_rep_cast_fp16 = reshape(shape = var_4832, x = var_4831_cast_fp16)[name = tensor("v_rep_cast_fp16")]; tensor var_4835_transpose_x_1 = const()[name = tensor("op_4835_transpose_x_1"), val = tensor(false)]; tensor var_4835_transpose_y_1 = const()[name = tensor("op_4835_transpose_y_1"), val = tensor(true)]; tensor var_4835_cast_fp16 = matmul(transpose_x = var_4835_transpose_x_1, transpose_y = var_4835_transpose_y_1, x = q_cast_fp16, y = k_rep_cast_fp16)[name = tensor("op_4835_cast_fp16")]; tensor var_4836_to_fp16 = const()[name = tensor("op_4836_to_fp16"), val = tensor(0x1.6ap-4)]; tensor attn_109_cast_fp16 = mul(x = var_4835_cast_fp16, y = var_4836_to_fp16)[name = tensor("attn_109_cast_fp16")]; tensor input_111_cast_fp16 = add(x = attn_109_cast_fp16, y = attn_mask_cast_fp16)[name = tensor("input_111_cast_fp16")]; tensor input_111_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("input_111_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor input_111_cast_fp16_to_fp32 = cast(dtype = input_111_cast_fp16_to_fp32_dtype_0, x = input_111_cast_fp16)[name = tensor("cast_188")]; tensor attn = softmax(axis = var_4734, x = input_111_cast_fp16_to_fp32)[name = tensor("attn")]; tensor out_transpose_x_0 = const()[name = tensor("out_transpose_x_0"), val = tensor(false)]; tensor out_transpose_y_0 = const()[name = tensor("out_transpose_y_0"), val = tensor(false)]; tensor attn_to_fp16_dtype_0 = const()[name = tensor("attn_to_fp16_dtype_0"), val = tensor("fp16")]; tensor attn_to_fp16 = cast(dtype = attn_to_fp16_dtype_0, x = attn)[name = tensor("cast_187")]; tensor out_cast_fp16 = matmul(transpose_x = out_transpose_x_0, transpose_y = out_transpose_y_0, x = attn_to_fp16, y = v_rep_cast_fp16)[name = tensor("out_cast_fp16")]; tensor var_4841_perm_0 = const()[name = tensor("op_4841_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_4843 = const()[name = tensor("op_4843"), val = tensor([1, 1, 1536])]; tensor var_4841_cast_fp16 = transpose(perm = var_4841_perm_0, x = out_cast_fp16)[name = tensor("transpose_0")]; tensor x_1107_cast_fp16 = reshape(shape = var_4843, x = var_4841_cast_fp16)[name = tensor("x_1107_cast_fp16")]; tensor layers_27_self_attn_o_proj_weight_to_fp16 = const()[name = tensor("layers_27_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(465818112)))]; tensor linear_192_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_27_self_attn_o_proj_weight_to_fp16, x = x_1107_cast_fp16)[name = tensor("linear_192_cast_fp16")]; tensor x_1109_cast_fp16 = add(x = x_1081_cast_fp16, y = linear_192_cast_fp16)[name = tensor("x_1109_cast_fp16")]; tensor x_1109_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1109_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4733_promoted_3 = const()[name = tensor("op_4733_promoted_3"), val = tensor(0x1p+1)]; tensor x_1109_cast_fp16_to_fp32 = cast(dtype = x_1109_cast_fp16_to_fp32_dtype_0, x = x_1109_cast_fp16)[name = tensor("cast_186")]; tensor var_4854 = pow(x = x_1109_cast_fp16_to_fp32, y = var_4733_promoted_3)[name = tensor("op_4854")]; tensor var_223_axes_0 = const()[name = tensor("var_223_axes_0"), val = tensor([-1])]; tensor var_223_keep_dims_0 = const()[name = tensor("var_223_keep_dims_0"), val = tensor(true)]; tensor var_223 = reduce_mean(axes = var_223_axes_0, keep_dims = var_223_keep_dims_0, x = var_4854)[name = tensor("var_223")]; tensor var_223_to_fp16_dtype_0 = const()[name = tensor("var_223_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4858_to_fp16 = const()[name = tensor("op_4858_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_223_to_fp16 = cast(dtype = var_223_to_fp16_dtype_0, x = var_223)[name = tensor("cast_185")]; tensor var_4859_cast_fp16 = add(x = var_223_to_fp16, y = var_4858_to_fp16)[name = tensor("op_4859_cast_fp16")]; tensor var_4859_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4859_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4860_epsilon_0 = const()[name = tensor("op_4860_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4859_cast_fp16_to_fp32 = cast(dtype = var_4859_cast_fp16_to_fp32_dtype_0, x = var_4859_cast_fp16)[name = tensor("cast_184")]; tensor var_4860 = rsqrt(epsilon = var_4860_epsilon_0, x = var_4859_cast_fp16_to_fp32)[name = tensor("op_4860")]; tensor var_4860_to_fp16_dtype_0 = const()[name = tensor("op_4860_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4860_to_fp16 = cast(dtype = var_4860_to_fp16_dtype_0, x = var_4860)[name = tensor("cast_183")]; tensor x_1115_cast_fp16 = mul(x = x_1109_cast_fp16, y = var_4860_to_fp16)[name = tensor("x_1115_cast_fp16")]; tensor layers_27_post_attention_layernorm_weight_to_fp16 = const()[name = tensor("layers_27_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(467391040)))]; tensor x_1117_cast_fp16 = mul(x = layers_27_post_attention_layernorm_weight_to_fp16, y = x_1115_cast_fp16)[name = tensor("x_1117_cast_fp16")]; tensor layers_27_mlp_gate_proj_weight_to_fp16 = const()[name = tensor("layers_27_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(467392128)))]; tensor linear_193_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_27_mlp_gate_proj_weight_to_fp16, x = x_1117_cast_fp16)[name = tensor("linear_193_cast_fp16")]; tensor var_4871_cast_fp16 = silu(x = linear_193_cast_fp16)[name = tensor("op_4871_cast_fp16")]; tensor layers_27_mlp_up_proj_weight_to_fp16 = const()[name = tensor("layers_27_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(468965056)))]; tensor linear_194_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = layers_27_mlp_up_proj_weight_to_fp16, x = x_1117_cast_fp16)[name = tensor("linear_194_cast_fp16")]; tensor x_1119_cast_fp16 = mul(x = var_4871_cast_fp16, y = linear_194_cast_fp16)[name = tensor("x_1119_cast_fp16")]; tensor layers_27_mlp_down_proj_weight_to_fp16 = const()[name = tensor("layers_27_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(470537984)))]; tensor linear_195_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_27_mlp_down_proj_weight_to_fp16, x = x_1119_cast_fp16)[name = tensor("linear_195_cast_fp16")]; tensor x_1121_cast_fp16 = add(x = x_1109_cast_fp16, y = linear_195_cast_fp16)[name = tensor("x_1121_cast_fp16")]; tensor x_1121_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("x_1121_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4885_promoted = const()[name = tensor("op_4885_promoted"), val = tensor(0x1p+1)]; tensor x_1121_cast_fp16_to_fp32 = cast(dtype = x_1121_cast_fp16_to_fp32_dtype_0, x = x_1121_cast_fp16)[name = tensor("cast_182")]; tensor var_4891 = pow(x = x_1121_cast_fp16_to_fp32, y = var_4885_promoted)[name = tensor("op_4891")]; tensor var_axes_0 = const()[name = tensor("var_axes_0"), val = tensor([-1])]; tensor var_keep_dims_0 = const()[name = tensor("var_keep_dims_0"), val = tensor(true)]; tensor var = reduce_mean(axes = var_axes_0, keep_dims = var_keep_dims_0, x = var_4891)[name = tensor("var")]; tensor var_to_fp16_dtype_0 = const()[name = tensor("var_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4895_to_fp16 = const()[name = tensor("op_4895_to_fp16"), val = tensor(0x1.1p-20)]; tensor var_to_fp16 = cast(dtype = var_to_fp16_dtype_0, x = var)[name = tensor("cast_181")]; tensor var_4896_cast_fp16 = add(x = var_to_fp16, y = var_4895_to_fp16)[name = tensor("op_4896_cast_fp16")]; tensor var_4896_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4896_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4897_epsilon_0 = const()[name = tensor("op_4897_epsilon_0"), val = tensor(0x1.197998p-40)]; tensor var_4896_cast_fp16_to_fp32 = cast(dtype = var_4896_cast_fp16_to_fp32_dtype_0, x = var_4896_cast_fp16)[name = tensor("cast_180")]; tensor var_4897 = rsqrt(epsilon = var_4897_epsilon_0, x = var_4896_cast_fp16_to_fp32)[name = tensor("op_4897")]; tensor var_4897_to_fp16_dtype_0 = const()[name = tensor("op_4897_to_fp16_dtype_0"), val = tensor("fp16")]; tensor var_4897_to_fp16 = cast(dtype = var_4897_to_fp16_dtype_0, x = var_4897)[name = tensor("cast_179")]; tensor x_cast_fp16 = mul(x = x_1121_cast_fp16, y = var_4897_to_fp16)[name = tensor("x_cast_fp16")]; tensor norm_weight_to_fp16 = const()[name = tensor("norm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(472110912)))]; tensor hidden_cast_fp16 = mul(x = norm_weight_to_fp16, y = x_cast_fp16)[name = tensor("hidden_cast_fp16")]; tensor linear_196_bias_0_to_fp16 = const()[name = tensor("linear_196_bias_0_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(472112000)))]; tensor linear_196_cast_fp16 = linear(bias = linear_196_bias_0_to_fp16, weight = tok_weight_to_fp16, x = hidden_cast_fp16)[name = tensor("linear_196_cast_fp16")]; tensor var_4903_axes_0 = const()[name = tensor("op_4903_axes_0"), val = tensor([1])]; tensor var_4903_cast_fp16 = squeeze(axes = var_4903_axes_0, x = linear_196_cast_fp16)[name = tensor("op_4903_cast_fp16")]; tensor var_4903_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4903_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4906_axis_0 = const()[name = tensor("op_4906_axis_0"), val = tensor(0)]; tensor var_4906_cast_fp16 = stack(axis = var_4906_axis_0, values = (k_full_1_cast_fp16, k_full_3_cast_fp16, k_full_5_cast_fp16, k_full_7_cast_fp16, k_full_9_cast_fp16, k_full_11_cast_fp16, k_full_13_cast_fp16, k_full_15_cast_fp16, k_full_17_cast_fp16, k_full_19_cast_fp16, k_full_21_cast_fp16, k_full_23_cast_fp16, k_full_25_cast_fp16, k_full_27_cast_fp16, k_full_29_cast_fp16, k_full_31_cast_fp16, k_full_33_cast_fp16, k_full_35_cast_fp16, k_full_37_cast_fp16, k_full_39_cast_fp16, k_full_41_cast_fp16, k_full_43_cast_fp16, k_full_45_cast_fp16, k_full_47_cast_fp16, k_full_49_cast_fp16, k_full_51_cast_fp16, k_full_53_cast_fp16, k_full_cast_fp16))[name = tensor("op_4906_cast_fp16")]; tensor var_4906_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4906_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor var_4909_axis_0 = const()[name = tensor("op_4909_axis_0"), val = tensor(0)]; tensor var_4909_cast_fp16 = stack(axis = var_4909_axis_0, values = (v_full_1_cast_fp16, v_full_3_cast_fp16, v_full_5_cast_fp16, v_full_7_cast_fp16, v_full_9_cast_fp16, v_full_11_cast_fp16, v_full_13_cast_fp16, v_full_15_cast_fp16, v_full_17_cast_fp16, v_full_19_cast_fp16, v_full_21_cast_fp16, v_full_23_cast_fp16, v_full_25_cast_fp16, v_full_27_cast_fp16, v_full_29_cast_fp16, v_full_31_cast_fp16, v_full_33_cast_fp16, v_full_35_cast_fp16, v_full_37_cast_fp16, v_full_39_cast_fp16, v_full_41_cast_fp16, v_full_43_cast_fp16, v_full_45_cast_fp16, v_full_47_cast_fp16, v_full_49_cast_fp16, v_full_51_cast_fp16, v_full_53_cast_fp16, v_full_cast_fp16))[name = tensor("op_4909_cast_fp16")]; tensor var_4909_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_4909_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; tensor kv_v_out = cast(dtype = var_4909_cast_fp16_to_fp32_dtype_0, x = var_4909_cast_fp16)[name = tensor("cast_176")]; tensor kv_k_out = cast(dtype = var_4906_cast_fp16_to_fp32_dtype_0, x = var_4906_cast_fp16)[name = tensor("cast_177")]; tensor logits = cast(dtype = var_4903_cast_fp16_to_fp32_dtype_0, x = var_4903_cast_fp16)[name = tensor("cast_178")]; } -> (logits, kv_k_out, kv_v_out); }