lithium0003 commited on Dec 13, 2025

Commit

8a23912

1 Parent(s): 1f717f0

20251213

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

ggml-base-encoder.mlmodelc/analytics/coremldata.bin +1 -1
ggml-base-encoder.mlmodelc/coremldata.bin +2 -2
ggml-base-encoder.mlmodelc/metadata.json +22 -20
ggml-base-encoder.mlmodelc/model.mil +0 -0
ggml-base-encoder.mlmodelc/weights/weight.bin +1 -1
ggml-large-v3-turbo-encoder.mlmodelc/model0/analytics/coremldata.bin → ggml-base-q8_0.bin +2 -2
ggml-large-v2-encoder.mlmodelc/metadata.json +0 -66
ggml-large-v2-encoder.mlmodelc/model.mil +0 -0
ggml-large-v2-q8_0.bin +0 -3
ggml-large-v3-encoder.mlmodelc/analytics/coremldata.bin +1 -1
ggml-large-v3-encoder.mlmodelc/coremldata.bin +2 -2
ggml-large-v3-encoder.mlmodelc/metadata.json +22 -20
ggml-large-v3-encoder.mlmodelc/model.mil +0 -0
ggml-large-v3-encoder.mlmodelc/weights/weight.bin +2 -2
ggml-large-v3-turbo-encoder.mlmodelc/analytics/coremldata.bin +2 -2
ggml-large-v3-turbo-encoder.mlmodelc/coremldata.bin +2 -2
ggml-large-v3-turbo-encoder.mlmodelc/metadata.json +10 -16
ggml-large-v3-turbo-encoder.mlmodelc/model.mil +0 -0
ggml-large-v3-turbo-encoder.mlmodelc/model0/coremldata.bin +0 -3
ggml-large-v3-turbo-encoder.mlmodelc/model0/model.mil +0 -0
ggml-large-v3-turbo-encoder.mlmodelc/model0/weights/0-weight.bin +0 -3
ggml-large-v3-turbo-encoder.mlmodelc/model1/analytics/coremldata.bin +0 -3
ggml-large-v3-turbo-encoder.mlmodelc/model1/coremldata.bin +0 -3
ggml-large-v3-turbo-encoder.mlmodelc/model1/model.mil +0 -0
ggml-large-v3-turbo-encoder.mlmodelc/model1/weights/1-weight.bin +0 -3
{ggml-large-v2-encoder.mlmodelc → ggml-large-v3-turbo-encoder.mlmodelc}/weights/weight.bin +2 -2
ggml-medium-encoder.mlmodelc/analytics/coremldata.bin +1 -1
ggml-medium-encoder.mlmodelc/coremldata.bin +2 -2
ggml-medium-encoder.mlmodelc/metadata.json +22 -20
ggml-medium-encoder.mlmodelc/model.mil +0 -0
ggml-medium-encoder.mlmodelc/weights/weight.bin +1 -1
ggml-base.bin → ggml-medium-q8_0.bin +2 -2
ggml-medium.bin +0 -3
ggml-small-encoder.mlmodelc/analytics/coremldata.bin +1 -1
ggml-small-encoder.mlmodelc/coremldata.bin +2 -2
ggml-small-encoder.mlmodelc/metadata.json +22 -20
ggml-small-encoder.mlmodelc/model.mil +0 -0
ggml-small-encoder.mlmodelc/weights/weight.bin +1 -1
ggml-large-v2-encoder.mlmodelc/coremldata.bin → ggml-small-q8_0.bin +2 -2
ggml-small.bin +0 -3
ggml-tiny-encoder.mlmodelc/analytics/coremldata.bin +1 -1
ggml-tiny-encoder.mlmodelc/coremldata.bin +2 -2
ggml-tiny-encoder.mlmodelc/metadata.json +22 -20
ggml-tiny-encoder.mlmodelc/model.mil +221 -265
ggml-tiny-encoder.mlmodelc/weights/weight.bin +1 -1
ggml-large-v2-encoder.mlmodelc/analytics/coremldata.bin → ggml-tiny-q8_0.bin +2 -2
ggml-tiny.bin +0 -3
index/base +1 -1
index/large-v2 +0 -6
index/large-v3-turbo +2 -8

ggml-base-encoder.mlmodelc/analytics/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:31fb009a1caa38a49165cb418f454ef1e5d3cd8e7a1bc37a721575d800a8c712
 size 243

 version https://git-lfs.github.com/spec/v1
+oid sha256:6ce0ee7deaec2f6b2c5927b6a124f866047ab9bfd04e92c30aadae7b9b45e0d7
 size 243

ggml-base-encoder.mlmodelc/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:b353632cc6cb8774fdd42eab32407ce491f6e786a1e15c30fc9c58c7e39cd437
-size 318

 version https://git-lfs.github.com/spec/v1
+oid sha256:98b532b5a0af2ecdf289928a9b20e6a5aa780da7fd5833fde5c4f90dfa8747ef
+size 379

ggml-base-encoder.mlmodelc/metadata.json CHANGED Viewed

@@ -17,36 +17,38 @@
     "modelParameters" : [
     ],
-    "specificationVersion" : 8,
     "mlProgramOperationTypeHistogram" : {
-      "Ios17.layerNorm" : 13,
-      "Ios17.reshape" : 24,
-      "Ios17.conv" : 2,
-      "Ios17.linear" : 36,
-      "Ios17.add" : 13,
-      "Ios17.matmul" : 12,
-      "Ios16.gelu" : 8,
-      "Ios16.softmax" : 6,
-      "Ios17.mul" : 12,
-      "Ios17.transpose" : 25
     },
     "computePrecision" : "Mixed (Float16, Int32)",
     "isUpdatable" : "0",
     "availability" : {
-      "macOS" : "14.0",
-      "tvOS" : "17.0",
-      "visionOS" : "1.0",
-      "watchOS" : "10.0",
-      "iOS" : "17.0",
-      "macCatalyst" : "17.0"
     },
     "modelType" : {
       "name" : "MLModelType_mlProgram"
     },
     "userDefinedMetadata" : {
-      "com.github.apple.coremltools.source_dialect" : "TorchScript",
-      "com.github.apple.coremltools.source" : "torch==2.2.2",
-      "com.github.apple.coremltools.version" : "7.2"
     },
     "inputSchema" : [
       {

     "modelParameters" : [
     ],
+    "specificationVersion" : 9,
     "mlProgramOperationTypeHistogram" : {
+      "Ios18.scaledDotProductAttention" : 6,
+      "Ios18.linear" : 36,
+      "Ios18.gelu" : 8,
+      "Ios18.layerNorm" : 13,
+      "Ios18.transpose" : 25,
+      "Ios18.conv" : 2,
+      "Ios18.add" : 13,
+      "Ios18.reshape" : 24
     },
     "computePrecision" : "Mixed (Float16, Int32)",
     "isUpdatable" : "0",
+    "stateSchema" : [
+    ],
     "availability" : {
+      "macOS" : "15.0",
+      "tvOS" : "18.0",
+      "visionOS" : "2.0",
+      "watchOS" : "11.0",
+      "iOS" : "18.0",
+      "macCatalyst" : "18.0"
     },
     "modelType" : {
       "name" : "MLModelType_mlProgram"
     },
     "userDefinedMetadata" : {
+      "com.github.apple.coremltools.conversion_date" : "2025-12-13",
+      "com.github.apple.coremltools.source" : "torch==2.9.1",
+      "com.github.apple.coremltools.version" : "9.0",
+      "com.github.apple.coremltools.source_dialect" : "TorchScript"
     },
     "inputSchema" : [
       {

ggml-base-encoder.mlmodelc/model.mil CHANGED Viewed

The diff for this file is too large to render. See raw diff

ggml-base-encoder.mlmodelc/weights/weight.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:b072fd4aebbd60ad4e35df05d3c2ea47267c00f25ac82f4dbcf49fb38e19faec
 size 41188544

 version https://git-lfs.github.com/spec/v1
+oid sha256:c45bee989219532c4cec616d439c51f280ac9d7b04f7847c4b7d7daba1d47523
 size 41188544

ggml-large-v3-turbo-encoder.mlmodelc/model0/analytics/coremldata.bin → ggml-base-q8_0.bin RENAMED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:5a8281049b2a65a3be541cfd9f949e84b8fe1c5251ce90e46da1626fed54e58a
-size 108

 version https://git-lfs.github.com/spec/v1
+oid sha256:f1a911a31e5812ebffef4c517c19186627a0203b7e95b5dbb6db884a9dc446ef
+size 56663708

ggml-large-v2-encoder.mlmodelc/metadata.json DELETED Viewed

@@ -1,66 +0,0 @@
-[
-  {
-    "metadataOutputVersion" : "3.0",
-    "storagePrecision" : "Float16",
-    "outputSchema" : [
-      {
-        "hasShapeFlexibility" : "0",
-        "isOptional" : "0",
-        "dataType" : "Float16",
-        "formattedType" : "MultiArray (Float16 1 × 1500 × 1280)",
-        "shortDescription" : "",
-        "shape" : "[1, 1500, 1280]",
-        "name" : "output",
-        "type" : "MultiArray"
-      }
-    ],
-    "modelParameters" : [
-    ],
-    "specificationVersion" : 8,
-    "mlProgramOperationTypeHistogram" : {
-      "Ios17.layerNorm" : 65,
-      "Ios17.reshape" : 128,
-      "Ios17.conv" : 2,
-      "Ios17.linear" : 192,
-      "Ios17.add" : 65,
-      "Ios17.matmul" : 64,
-      "Ios16.gelu" : 34,
-      "Ios16.softmax" : 32,
-      "Ios17.mul" : 64,
-      "Ios17.transpose" : 129
-    },
-    "computePrecision" : "Mixed (Float16, Int32)",
-    "isUpdatable" : "0",
-    "availability" : {
-      "macOS" : "14.0",
-      "tvOS" : "17.0",
-      "visionOS" : "1.0",
-      "watchOS" : "10.0",
-      "iOS" : "17.0",
-      "macCatalyst" : "17.0"
-    },
-    "modelType" : {
-      "name" : "MLModelType_mlProgram"
-    },
-    "userDefinedMetadata" : {
-      "com.github.apple.coremltools.source_dialect" : "TorchScript",
-      "com.github.apple.coremltools.source" : "torch==2.3.0",
-      "com.github.apple.coremltools.version" : "7.2"
-    },
-    "inputSchema" : [
-      {
-        "hasShapeFlexibility" : "0",
-        "isOptional" : "0",
-        "dataType" : "Float16",
-        "formattedType" : "MultiArray (Float16 1 × 80 × 3000)",
-        "shortDescription" : "",
-        "shape" : "[1, 80, 3000]",
-        "name" : "logmel_data",
-        "type" : "MultiArray"
-      }
-    ],
-    "generatedClassName" : "ggml_large_v2_encoder",
-    "method" : "predict"
-  }
-]

ggml-large-v2-encoder.mlmodelc/model.mil DELETED Viewed

The diff for this file is too large to render. See raw diff

ggml-large-v2-q8_0.bin DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:3a0b8cd189b25520643b628d08dd100a8f0c884c048c25950302bd01a259ae58
-size 967527492

ggml-large-v3-encoder.mlmodelc/analytics/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:2f3c468ec7667cd29ab246467a25b2b93f67b72690248d50d5477ea32e2af932
 size 243

 version https://git-lfs.github.com/spec/v1
+oid sha256:2c012cd81c638ceccf50d4eb96a26f7a440a3a0c677a35120b927a047d1bccc5
 size 243

ggml-large-v3-encoder.mlmodelc/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:f8cf0ad791b6d5613788de024fd72ea2af5e4c08b696fa2528f5ed25a25b5b4b
-size 319

 version https://git-lfs.github.com/spec/v1
+oid sha256:435c32326d0bd2151a3fefa83c80af6c7bab4b3f797aeb6e5f18ab131dd2459b
+size 380

ggml-large-v3-encoder.mlmodelc/metadata.json CHANGED Viewed

@@ -17,36 +17,38 @@
     "modelParameters" : [
     ],
-    "specificationVersion" : 8,
     "mlProgramOperationTypeHistogram" : {
-      "Ios17.layerNorm" : 65,
-      "Ios17.reshape" : 128,
-      "Ios17.conv" : 2,
-      "Ios17.linear" : 192,
-      "Ios17.add" : 65,
-      "Ios17.matmul" : 64,
-      "Ios16.gelu" : 34,
-      "Ios16.softmax" : 32,
-      "Ios17.mul" : 64,
-      "Ios17.transpose" : 129
     },
     "computePrecision" : "Mixed (Float16, Int32)",
     "isUpdatable" : "0",
     "availability" : {
-      "macOS" : "14.0",
-      "tvOS" : "17.0",
-      "visionOS" : "1.0",
-      "watchOS" : "10.0",
-      "iOS" : "17.0",
-      "macCatalyst" : "17.0"
     },
     "modelType" : {
       "name" : "MLModelType_mlProgram"
     },
     "userDefinedMetadata" : {
-      "com.github.apple.coremltools.source_dialect" : "TorchScript",
-      "com.github.apple.coremltools.source" : "torch==2.3.0",
-      "com.github.apple.coremltools.version" : "7.2"
     },
     "inputSchema" : [
       {

     "modelParameters" : [
     ],
+    "specificationVersion" : 9,
     "mlProgramOperationTypeHistogram" : {
+      "Ios18.scaledDotProductAttention" : 32,
+      "Ios18.linear" : 192,
+      "Ios18.gelu" : 34,
+      "Ios18.layerNorm" : 65,
+      "Ios18.transpose" : 129,
+      "Ios18.conv" : 2,
+      "Ios18.add" : 65,
+      "Ios18.reshape" : 128
     },
     "computePrecision" : "Mixed (Float16, Int32)",
     "isUpdatable" : "0",
+    "stateSchema" : [
+    ],
     "availability" : {
+      "macOS" : "15.0",
+      "tvOS" : "18.0",
+      "visionOS" : "2.0",
+      "watchOS" : "11.0",
+      "iOS" : "18.0",
+      "macCatalyst" : "18.0"
     },
     "modelType" : {
       "name" : "MLModelType_mlProgram"
     },
     "userDefinedMetadata" : {
+      "com.github.apple.coremltools.conversion_date" : "2025-12-13",
+      "com.github.apple.coremltools.source" : "torch==2.9.1",
+      "com.github.apple.coremltools.version" : "9.0",
+      "com.github.apple.coremltools.source_dialect" : "TorchScript"
     },
     "inputSchema" : [
       {

ggml-large-v3-encoder.mlmodelc/model.mil CHANGED Viewed

The diff for this file is too large to render. See raw diff

ggml-large-v3-encoder.mlmodelc/weights/weight.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:086e3063e54313291f2d4199d1303e64e466d638384c9469ac3f6ad95ce0c895
-size 1064174656

 version https://git-lfs.github.com/spec/v1
+oid sha256:7cdd15ff2f714bdca758058f70490c89617f764ddc1b25602a6c60eaa408856a
+size 1273971776

ggml-large-v3-turbo-encoder.mlmodelc/analytics/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:7be9acf764f6e452c949d3a3b27bf0afe3609629db9ba9dd8069127ad8e43c20
-size 202

 version https://git-lfs.github.com/spec/v1
+oid sha256:521d0f794f87e8fea91108f66d373485c89fce52d861e3cca3e4a5a01b6734fd
+size 243

ggml-large-v3-turbo-encoder.mlmodelc/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:46dc321dd0ff6005125dc0365c3e0ecb2413f838328888df48578af4d2869749
-size 197

 version https://git-lfs.github.com/spec/v1
+oid sha256:d188b1f3d2d1c8e15381f6c75bb2a2caa03a5094c30136fe897b6a1d91ce907c
+size 380

ggml-large-v3-turbo-encoder.mlmodelc/metadata.json CHANGED Viewed

@@ -19,17 +19,16 @@
     ],
     "specificationVersion" : 9,
     "mlProgramOperationTypeHistogram" : {
-      "Ios18.reshape" : 128,
       "Ios18.linear" : 192,
       "Ios18.gelu" : 34,
       "Ios18.layerNorm" : 65,
       "Ios18.transpose" : 129,
       "Ios18.conv" : 2,
-      "Ios18.cast" : 4,
-      "Ios18.scaledDotProductAttention" : 32,
-      "Ios18.add" : 65
     },
-    "computePrecision" : "Mixed (Float16, Float32, Int32)",
     "isUpdatable" : "0",
     "stateSchema" : [
@@ -43,18 +42,13 @@
       "macCatalyst" : "18.0"
     },
     "modelType" : {
-      "name" : "MLModelType_pipeline",
-      "structure" : [
-        {
-          "name" : "MLModelType_mlProgram"
-        },
-        {
-          "name" : "MLModelType_mlProgram"
-        }
-      ]
     },
     "userDefinedMetadata" : {
     },
     "inputSchema" : [
       {
@@ -68,7 +62,7 @@
         "type" : "MultiArray"
       }
     ],
-    "generatedClassName" : "chunked_pipeline",
     "method" : "predict"
   }
 ]

     ],
     "specificationVersion" : 9,
     "mlProgramOperationTypeHistogram" : {
+      "Ios18.scaledDotProductAttention" : 32,
       "Ios18.linear" : 192,
       "Ios18.gelu" : 34,
       "Ios18.layerNorm" : 65,
       "Ios18.transpose" : 129,
       "Ios18.conv" : 2,
+      "Ios18.add" : 65,
+      "Ios18.reshape" : 128
     },
+    "computePrecision" : "Mixed (Float16, Int32)",
     "isUpdatable" : "0",
     "stateSchema" : [
       "macCatalyst" : "18.0"
     },
     "modelType" : {
+      "name" : "MLModelType_mlProgram"
     },
     "userDefinedMetadata" : {
+      "com.github.apple.coremltools.conversion_date" : "2025-12-13",
+      "com.github.apple.coremltools.source" : "torch==2.9.1",
+      "com.github.apple.coremltools.version" : "9.0",
+      "com.github.apple.coremltools.source_dialect" : "TorchScript"
     },
     "inputSchema" : [
       {
         "type" : "MultiArray"
       }
     ],
+    "generatedClassName" : "ggml_large_v3_turbo_encoder",
     "method" : "predict"
   }
 ]

ggml-large-v3-turbo-encoder.mlmodelc/model.mil ADDED Viewed

The diff for this file is too large to render. See raw diff

ggml-large-v3-turbo-encoder.mlmodelc/model0/coremldata.bin DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:a2b0461e225831cc34e0017a300f867929784559e2ee471f01ddfd3452381076
-size 201

ggml-large-v3-turbo-encoder.mlmodelc/model0/model.mil DELETED Viewed

The diff for this file is too large to render. See raw diff

ggml-large-v3-turbo-encoder.mlmodelc/model0/weights/0-weight.bin DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:6d26bc07916a863ace6c1bb750aad546bdc7b16c12ebe198d3b511134b09eb68
-size 644314048

ggml-large-v3-turbo-encoder.mlmodelc/model1/analytics/coremldata.bin DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:5a8281049b2a65a3be541cfd9f949e84b8fe1c5251ce90e46da1626fed54e58a
-size 108

ggml-large-v3-turbo-encoder.mlmodelc/model1/coremldata.bin DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:ee5eef16c4adf0aac1e84f5676c8942aa2741fe919f65d72b1b23bd9dd417ab2
-size 196

ggml-large-v3-turbo-encoder.mlmodelc/model1/model.mil DELETED Viewed

The diff for this file is too large to render. See raw diff

ggml-large-v3-turbo-encoder.mlmodelc/model1/weights/1-weight.bin DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:28e9965c3b4ed7db29d226652d29ca6f09a35ee68d2cd61100e7b0f4d0a7095c
-size 629660416

{ggml-large-v2-encoder.mlmodelc → ggml-large-v3-turbo-encoder.mlmodelc}/weights/weight.bin RENAMED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:38680c45f8b18d9586187777968cf81e3e7b7af6baa88a52fc1a06bdb8798548
-size 1063806016

 version https://git-lfs.github.com/spec/v1
+oid sha256:e77bb855473f07c5e56a2343dcf1f3af9c80b6e61ce4247268d67f929e47a1c8
+size 1273971776

ggml-medium-encoder.mlmodelc/analytics/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:975604071b5e16602d99864f0c9bd082c0114ced9fea1cce6367ee7a33e99591
 size 243

 version https://git-lfs.github.com/spec/v1
+oid sha256:f3328e031445e499158cd4fc091b4ca7da84476e629db98372b1c52ed49e8f49
 size 243

ggml-medium-encoder.mlmodelc/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:e781c7fcb346060d55fa9f2066f27fdc0e58b6018c50400488144d06e6dd7109
-size 318

 version https://git-lfs.github.com/spec/v1
+oid sha256:1085b1ad3f7ed6de1553948e97404683f50b555a59ee26d06b8c425e8f8f30d1
+size 379

ggml-medium-encoder.mlmodelc/metadata.json CHANGED Viewed

@@ -17,36 +17,38 @@
     "modelParameters" : [
     ],
-    "specificationVersion" : 8,
     "mlProgramOperationTypeHistogram" : {
-      "Ios17.layerNorm" : 49,
-      "Ios17.reshape" : 96,
-      "Ios17.conv" : 2,
-      "Ios17.linear" : 144,
-      "Ios17.add" : 49,
-      "Ios17.matmul" : 48,
-      "Ios16.gelu" : 26,
-      "Ios16.softmax" : 24,
-      "Ios17.mul" : 48,
-      "Ios17.transpose" : 97
     },
     "computePrecision" : "Mixed (Float16, Int32)",
     "isUpdatable" : "0",
     "availability" : {
-      "macOS" : "14.0",
-      "tvOS" : "17.0",
-      "visionOS" : "1.0",
-      "watchOS" : "10.0",
-      "iOS" : "17.0",
-      "macCatalyst" : "17.0"
     },
     "modelType" : {
       "name" : "MLModelType_mlProgram"
     },
     "userDefinedMetadata" : {
-      "com.github.apple.coremltools.source_dialect" : "TorchScript",
-      "com.github.apple.coremltools.source" : "torch==2.2.2",
-      "com.github.apple.coremltools.version" : "7.2"
     },
     "inputSchema" : [
       {

     "modelParameters" : [
     ],
+    "specificationVersion" : 9,
     "mlProgramOperationTypeHistogram" : {
+      "Ios18.scaledDotProductAttention" : 24,
+      "Ios18.linear" : 144,
+      "Ios18.gelu" : 26,
+      "Ios18.layerNorm" : 49,
+      "Ios18.transpose" : 97,
+      "Ios18.conv" : 2,
+      "Ios18.add" : 49,
+      "Ios18.reshape" : 96
     },
     "computePrecision" : "Mixed (Float16, Int32)",
     "isUpdatable" : "0",
+    "stateSchema" : [
+    ],
     "availability" : {
+      "macOS" : "15.0",
+      "tvOS" : "18.0",
+      "visionOS" : "2.0",
+      "watchOS" : "11.0",
+      "iOS" : "18.0",
+      "macCatalyst" : "18.0"
     },
     "modelType" : {
       "name" : "MLModelType_mlProgram"
     },
     "userDefinedMetadata" : {
+      "com.github.apple.coremltools.conversion_date" : "2025-12-13",
+      "com.github.apple.coremltools.source" : "torch==2.9.1",
+      "com.github.apple.coremltools.version" : "9.0",
+      "com.github.apple.coremltools.source_dialect" : "TorchScript"
     },
     "inputSchema" : [
       {

ggml-medium-encoder.mlmodelc/model.mil CHANGED Viewed

The diff for this file is too large to render. See raw diff

ggml-medium-encoder.mlmodelc/weights/weight.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:cbbdb8a10cee8abc0d0c0d86af33ea4b1d8453380df103fd49272a28143a892b
 size 614458432

 version https://git-lfs.github.com/spec/v1
+oid sha256:2cd1baef4c7d8260ea817ea56705b3700155c01e8d3ea4bc8e364a8674a88d15
 size 614458432

ggml-base.bin → ggml-medium-q8_0.bin RENAMED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:84ceff52bae6817d2b3c3f5c96e6cfad66352047dd390b86b8aaf0e6f30dff9c
-size 105151868

 version https://git-lfs.github.com/spec/v1
+oid sha256:1562f96f3dd57b9517f286ea4b5843ed70759bce397035ffe84f0bb2d407d62b
+size 488364756

ggml-medium.bin DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:076f80d119389b32fb3ff7ee78330418de19970090183807dfa3da56a4f69fda
-size 915642516

ggml-small-encoder.mlmodelc/analytics/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:0e33d81d8cf4a206337049c6f0379ed5370da2b0fc3d822a6065b2c021664a46
 size 243

 version https://git-lfs.github.com/spec/v1
+oid sha256:ad9997a136336812138a5f02e4977a2a83c5112a5c2928beb7b43bafc55853df
 size 243

ggml-small-encoder.mlmodelc/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:f6643c80fd2477e972234f11c3d21cc415963809f6bc560980f83bda4c62f31b
-size 318

 version https://git-lfs.github.com/spec/v1
+oid sha256:c6253da76ccc8aefef97c4bc85a4f2a5687e4ed4db03a41c34efff4a7ec9de3e
+size 379

ggml-small-encoder.mlmodelc/metadata.json CHANGED Viewed

@@ -17,36 +17,38 @@
     "modelParameters" : [
     ],
-    "specificationVersion" : 8,
     "mlProgramOperationTypeHistogram" : {
-      "Ios17.layerNorm" : 25,
-      "Ios17.reshape" : 48,
-      "Ios17.conv" : 2,
-      "Ios17.linear" : 72,
-      "Ios17.add" : 25,
-      "Ios17.matmul" : 24,
-      "Ios16.gelu" : 14,
-      "Ios16.softmax" : 12,
-      "Ios17.mul" : 24,
-      "Ios17.transpose" : 49
     },
     "computePrecision" : "Mixed (Float16, Int32)",
     "isUpdatable" : "0",
     "availability" : {
-      "macOS" : "14.0",
-      "tvOS" : "17.0",
-      "visionOS" : "1.0",
-      "watchOS" : "10.0",
-      "iOS" : "17.0",
-      "macCatalyst" : "17.0"
     },
     "modelType" : {
       "name" : "MLModelType_mlProgram"
     },
     "userDefinedMetadata" : {
-      "com.github.apple.coremltools.source_dialect" : "TorchScript",
-      "com.github.apple.coremltools.source" : "torch==2.2.2",
-      "com.github.apple.coremltools.version" : "7.2"
     },
     "inputSchema" : [
       {

     "modelParameters" : [
     ],
+    "specificationVersion" : 9,
     "mlProgramOperationTypeHistogram" : {
+      "Ios18.scaledDotProductAttention" : 12,
+      "Ios18.linear" : 72,
+      "Ios18.gelu" : 14,
+      "Ios18.layerNorm" : 25,
+      "Ios18.transpose" : 49,
+      "Ios18.conv" : 2,
+      "Ios18.add" : 25,
+      "Ios18.reshape" : 48
     },
     "computePrecision" : "Mixed (Float16, Int32)",
     "isUpdatable" : "0",
+    "stateSchema" : [
+    ],
     "availability" : {
+      "macOS" : "15.0",
+      "tvOS" : "18.0",
+      "visionOS" : "2.0",
+      "watchOS" : "11.0",
+      "iOS" : "18.0",
+      "macCatalyst" : "18.0"
     },
     "modelType" : {
       "name" : "MLModelType_mlProgram"
     },
     "userDefinedMetadata" : {
+      "com.github.apple.coremltools.conversion_date" : "2025-12-13",
+      "com.github.apple.coremltools.source" : "torch==2.9.1",
+      "com.github.apple.coremltools.version" : "9.0",
+      "com.github.apple.coremltools.source_dialect" : "TorchScript"
     },
     "inputSchema" : [
       {

ggml-small-encoder.mlmodelc/model.mil CHANGED Viewed

The diff for this file is too large to render. See raw diff

ggml-small-encoder.mlmodelc/weights/weight.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:c2362f377002638f1f3ec3a413d80c9d82410b537d776a1a75d6571c8e40ace6
 size 176321856

 version https://git-lfs.github.com/spec/v1
+oid sha256:4d3ab676977d57b06993ee7ebc638fc8568a99ddb11cb7a445328ce50fbd8b36
 size 176321856

ggml-large-v2-encoder.mlmodelc/coremldata.bin → ggml-small-q8_0.bin RENAMED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:ae8e3d2e9d47c7254f44d42290d3bfc325144bd7671d7631b3eb222484a1006e
-size 318

 version https://git-lfs.github.com/spec/v1
+oid sha256:2f9f98684c133761578bb32314e4b978c203bc6d4d1b6f5553c356104b9a7571
+size 165242356

ggml-small.bin DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:42ea1b1ba6ffa8cde9da77664db1ac5e24378f9d3d0363ca0104deebedac7732
-size 308753476

ggml-tiny-encoder.mlmodelc/analytics/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:3e6edb6be522938b84fd688c4aef4180f652783aaa157e274bd1ba221c2f9322
 size 243

 version https://git-lfs.github.com/spec/v1
+oid sha256:e4caac0b38c9cca17df0f212e16a01d7046e2f454767565a9161988640492be8
 size 243

ggml-tiny-encoder.mlmodelc/coremldata.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:4a2e684990429a39584b0adf645ea565057cb6bdcfd404f2dba45a1031e4faaa
-size 318

 version https://git-lfs.github.com/spec/v1
+oid sha256:de0977d50f2b62990f3fe458c5f94e6e05147e197dfce5ce9c6fadc370ea0e7f
+size 379

ggml-tiny-encoder.mlmodelc/metadata.json CHANGED Viewed

@@ -17,36 +17,38 @@
     "modelParameters" : [
     ],
-    "specificationVersion" : 8,
     "mlProgramOperationTypeHistogram" : {
-      "Ios17.layerNorm" : 9,
-      "Ios17.reshape" : 16,
-      "Ios17.conv" : 2,
-      "Ios17.linear" : 24,
-      "Ios17.add" : 9,
-      "Ios17.matmul" : 8,
-      "Ios16.gelu" : 6,
-      "Ios16.softmax" : 4,
-      "Ios17.mul" : 8,
-      "Ios17.transpose" : 17
     },
     "computePrecision" : "Mixed (Float16, Int32)",
     "isUpdatable" : "0",
     "availability" : {
-      "macOS" : "14.0",
-      "tvOS" : "17.0",
-      "visionOS" : "1.0",
-      "watchOS" : "10.0",
-      "iOS" : "17.0",
-      "macCatalyst" : "17.0"
     },
     "modelType" : {
       "name" : "MLModelType_mlProgram"
     },
     "userDefinedMetadata" : {
-      "com.github.apple.coremltools.source_dialect" : "TorchScript",
-      "com.github.apple.coremltools.source" : "torch==2.2.2",
-      "com.github.apple.coremltools.version" : "7.2"
     },
     "inputSchema" : [
       {

     "modelParameters" : [
     ],
+    "specificationVersion" : 9,
     "mlProgramOperationTypeHistogram" : {
+      "Ios18.scaledDotProductAttention" : 4,
+      "Ios18.linear" : 24,
+      "Ios18.gelu" : 6,
+      "Ios18.layerNorm" : 9,
+      "Ios18.transpose" : 17,
+      "Ios18.conv" : 2,
+      "Ios18.add" : 9,
+      "Ios18.reshape" : 16
     },
     "computePrecision" : "Mixed (Float16, Int32)",
     "isUpdatable" : "0",
+    "stateSchema" : [
+    ],
     "availability" : {
+      "macOS" : "15.0",
+      "tvOS" : "18.0",
+      "visionOS" : "2.0",
+      "watchOS" : "11.0",
+      "iOS" : "18.0",
+      "macCatalyst" : "18.0"
     },
     "modelType" : {
       "name" : "MLModelType_mlProgram"
     },
     "userDefinedMetadata" : {
+      "com.github.apple.coremltools.conversion_date" : "2025-12-13",
+      "com.github.apple.coremltools.source" : "torch==2.9.1",
+      "com.github.apple.coremltools.version" : "9.0",
+      "com.github.apple.coremltools.source_dialect" : "TorchScript"
     },
     "inputSchema" : [
       {

ggml-tiny-encoder.mlmodelc/model.mil CHANGED Viewed

@@ -1,268 +1,224 @@
-program(1.0)
-[buildInfo = dict<tensor<string, []>, tensor<string, []>>({{"coremlc-component-MIL", "3304.5.2"}, {"coremlc-version", "3304.6.2"}, {"coremltools-component-torch", "2.2.2"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "7.2"}})]
 {
-    func main<ios17>(tensor<fp16, [1, 80, 3000]> logmel_data) {
-            tensor<int32, []> var_16 = const()[name = tensor<string, []>("op_16"), val = tensor<int32, []>(1)];
-            tensor<int32, [1]> var_24 = const()[name = tensor<string, []>("op_24"), val = tensor<int32, [1]>([1])];
-            tensor<int32, [1]> var_26 = const()[name = tensor<string, []>("op_26"), val = tensor<int32, [1]>([1])];
-            tensor<string, []> var_28_pad_type_0 = const()[name = tensor<string, []>("op_28_pad_type_0"), val = tensor<string, []>("custom")];
-            tensor<int32, [2]> var_28_pad_0 = const()[name = tensor<string, []>("op_28_pad_0"), val = tensor<int32, [2]>([1, 1])];
-            tensor<fp16, [384, 80, 3]> weight_3_to_fp16 = const()[name = tensor<string, []>("weight_3_to_fp16"), val = tensor<fp16, [384, 80, 3]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(64)))];
-            tensor<fp16, [384]> bias_3_to_fp16 = const()[name = tensor<string, []>("bias_3_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(184448)))];
-            tensor<fp16, [1, 384, 3000]> var_28_cast_fp16 = conv(bias = bias_3_to_fp16, dilations = var_26, groups = var_16, pad = var_28_pad_0, pad_type = var_28_pad_type_0, strides = var_24, weight = weight_3_to_fp16, x = logmel_data)[name = tensor<string, []>("op_28_cast_fp16")];
-            tensor<string, []> input_1_mode_0 = const()[name = tensor<string, []>("input_1_mode_0"), val = tensor<string, []>("EXACT")];
-            tensor<fp16, [1, 384, 3000]> input_1_cast_fp16 = gelu(mode = input_1_mode_0, x = var_28_cast_fp16)[name = tensor<string, []>("input_1_cast_fp16")];
-            tensor<int32, []> var_33 = const()[name = tensor<string, []>("op_33"), val = tensor<int32, []>(1)];
-            tensor<int32, [1]> var_42 = const()[name = tensor<string, []>("op_42"), val = tensor<int32, [1]>([2])];
-            tensor<int32, [1]> var_44 = const()[name = tensor<string, []>("op_44"), val = tensor<int32, [1]>([1])];
-            tensor<string, []> var_46_pad_type_0 = const()[name = tensor<string, []>("op_46_pad_type_0"), val = tensor<string, []>("custom")];
-            tensor<int32, [2]> var_46_pad_0 = const()[name = tensor<string, []>("op_46_pad_0"), val = tensor<int32, [2]>([1, 1])];
-            tensor<fp16, [384, 384, 3]> weight_7_to_fp16 = const()[name = tensor<string, []>("weight_7_to_fp16"), val = tensor<fp16, [384, 384, 3]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(185280)))];
-            tensor<fp16, [384]> bias_7_to_fp16 = const()[name = tensor<string, []>("bias_7_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(1070080)))];
-            tensor<fp16, [1, 384, 1500]> var_46_cast_fp16 = conv(bias = bias_7_to_fp16, dilations = var_44, groups = var_33, pad = var_46_pad_0, pad_type = var_46_pad_type_0, strides = var_42, weight = weight_7_to_fp16, x = input_1_cast_fp16)[name = tensor<string, []>("op_46_cast_fp16")];
-            tensor<string, []> x_3_mode_0 = const()[name = tensor<string, []>("x_3_mode_0"), val = tensor<string, []>("EXACT")];
-            tensor<fp16, [1, 384, 1500]> x_3_cast_fp16 = gelu(mode = x_3_mode_0, x = var_46_cast_fp16)[name = tensor<string, []>("x_3_cast_fp16")];
-            tensor<int32, [3]> var_52 = const()[name = tensor<string, []>("op_52"), val = tensor<int32, [3]>([0, 2, 1])];
-            tensor<fp16, [1500, 384]> positional_embedding_to_fp16 = const()[name = tensor<string, []>("positional_embedding_to_fp16"), val = tensor<fp16, [1500, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(1070912)))];
-            tensor<fp16, [1, 1500, 384]> transpose_40 = transpose(perm = var_52, x = x_3_cast_fp16)[name = tensor<string, []>("transpose_40")];
-            tensor<fp16, [1, 1500, 384]> var_55_cast_fp16 = add(x = transpose_40, y = positional_embedding_to_fp16)[name = tensor<string, []>("op_55_cast_fp16")];
-            tensor<int32, []> var_67 = const()[name = tensor<string, []>("op_67"), val = tensor<int32, []>(-1)];
-            tensor<int32, [1]> var_83_axes_0 = const()[name = tensor<string, []>("op_83_axes_0"), val = tensor<int32, [1]>([-1])];
-            tensor<fp16, [384]> blocks_0_attn_ln_weight_to_fp16 = const()[name = tensor<string, []>("blocks_0_attn_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(2222976)))];
-            tensor<fp16, [384]> blocks_0_attn_ln_bias_to_fp16 = const()[name = tensor<string, []>("blocks_0_attn_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(2223808)))];
-            tensor<fp16, []> var_73_to_fp16 = const()[name = tensor<string, []>("op_73_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
-            tensor<fp16, [1, 1500, 384]> var_83_cast_fp16 = layer_norm(axes = var_83_axes_0, beta = blocks_0_attn_ln_bias_to_fp16, epsilon = var_73_to_fp16, gamma = blocks_0_attn_ln_weight_to_fp16, x = var_55_cast_fp16)[name = tensor<string, []>("op_83_cast_fp16")];
-            tensor<fp16, [384, 384]> var_94_to_fp16 = const()[name = tensor<string, []>("op_94_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(2224640)))];
-            tensor<fp16, [384]> var_95_to_fp16 = const()[name = tensor<string, []>("op_95_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(2519616)))];
-            tensor<fp16, [1, 1500, 384]> linear_0_cast_fp16 = linear(bias = var_95_to_fp16, weight = var_94_to_fp16, x = var_83_cast_fp16)[name = tensor<string, []>("linear_0_cast_fp16")];
-            tensor<fp16, [384, 384]> var_98_to_fp16 = const()[name = tensor<string, []>("op_98_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(2520448)))];
-            tensor<fp16, [384]> linear_1_bias_0_to_fp16 = const()[name = tensor<string, []>("linear_1_bias_0_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(2815424)))];
-            tensor<fp16, [1, 1500, 384]> linear_1_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = var_98_to_fp16, x = var_83_cast_fp16)[name = tensor<string, []>("linear_1_cast_fp16")];
-            tensor<fp16, [384, 384]> var_102_to_fp16 = const()[name = tensor<string, []>("op_102_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(2816256)))];
-            tensor<fp16, [384]> var_103_to_fp16 = const()[name = tensor<string, []>("op_103_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(3111232)))];
-            tensor<fp16, [1, 1500, 384]> linear_2_cast_fp16 = linear(bias = var_103_to_fp16, weight = var_102_to_fp16, x = var_83_cast_fp16)[name = tensor<string, []>("linear_2_cast_fp16")];
-            tensor<int32, [4]> var_111 = const()[name = tensor<string, []>("op_111"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_112_cast_fp16 = reshape(shape = var_111, x = linear_0_cast_fp16)[name = tensor<string, []>("op_112_cast_fp16")];
-            tensor<fp16, [1, 1, 1, 1]> const_28_to_fp16 = const()[name = tensor<string, []>("const_28_to_fp16"), val = tensor<fp16, [1, 1, 1, 1]>([[[[0x1.6ap-2]]]])];
-            tensor<fp16, [1, 1500, 6, 64]> q_3_cast_fp16 = mul(x = var_112_cast_fp16, y = const_28_to_fp16)[name = tensor<string, []>("q_3_cast_fp16")];
-            tensor<int32, [4]> var_118 = const()[name = tensor<string, []>("op_118"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_119_cast_fp16 = reshape(shape = var_118, x = linear_1_cast_fp16)[name = tensor<string, []>("op_119_cast_fp16")];
-            tensor<fp16, [1, 1, 1, 1]> const_29_to_fp16 = const()[name = tensor<string, []>("const_29_to_fp16"), val = tensor<fp16, [1, 1, 1, 1]>([[[[0x1.6ap-2]]]])];
-            tensor<fp16, [1, 1500, 6, 64]> k_3_cast_fp16 = mul(x = var_119_cast_fp16, y = const_29_to_fp16)[name = tensor<string, []>("k_3_cast_fp16")];
-            tensor<int32, [4]> var_125 = const()[name = tensor<string, []>("op_125"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_126_cast_fp16 = reshape(shape = var_125, x = linear_2_cast_fp16)[name = tensor<string, []>("op_126_cast_fp16")];
-            tensor<int32, [4]> var_127 = const()[name = tensor<string, []>("op_127"), val = tensor<int32, [4]>([0, 2, 1, 3])];
-            tensor<bool, []> qk_1_transpose_x_0 = const()[name = tensor<string, []>("qk_1_transpose_x_0"), val = tensor<bool, []>(false)];
-            tensor<bool, []> qk_1_transpose_y_0 = const()[name = tensor<string, []>("qk_1_transpose_y_0"), val = tensor<bool, []>(false)];
-            tensor<int32, [4]> transpose_16_perm_0 = const()[name = tensor<string, []>("transpose_16_perm_0"), val = tensor<int32, [4]>([0, 2, -3, -1])];
-            tensor<int32, [4]> transpose_17_perm_0 = const()[name = tensor<string, []>("transpose_17_perm_0"), val = tensor<int32, [4]>([0, 2, -1, -3])];
-            tensor<fp16, [1, 6, 64, 1500]> transpose_37 = transpose(perm = transpose_17_perm_0, x = k_3_cast_fp16)[name = tensor<string, []>("transpose_37")];
-            tensor<fp16, [1, 6, 1500, 64]> transpose_38 = transpose(perm = transpose_16_perm_0, x = q_3_cast_fp16)[name = tensor<string, []>("transpose_38")];
-            tensor<fp16, [1, 6, 1500, 1500]> qk_1_cast_fp16 = matmul(transpose_x = qk_1_transpose_x_0, transpose_y = qk_1_transpose_y_0, x = transpose_38, y = transpose_37)[name = tensor<string, []>("qk_1_cast_fp16")];
-            tensor<fp16, [1, 6, 1500, 1500]> var_131_cast_fp16 = softmax(axis = var_67, x = qk_1_cast_fp16)[name = tensor<string, []>("op_131_cast_fp16")];
-            tensor<bool, []> var_133_transpose_x_0 = const()[name = tensor<string, []>("op_133_transpose_x_0"), val = tensor<bool, []>(false)];
-            tensor<bool, []> var_133_transpose_y_0 = const()[name = tensor<string, []>("op_133_transpose_y_0"), val = tensor<bool, []>(false)];
-            tensor<fp16, [1, 6, 1500, 64]> transpose_39 = transpose(perm = var_127, x = var_126_cast_fp16)[name = tensor<string, []>("transpose_39")];
-            tensor<fp16, [1, 6, 1500, 64]> var_133_cast_fp16 = matmul(transpose_x = var_133_transpose_x_0, transpose_y = var_133_transpose_y_0, x = var_131_cast_fp16, y = transpose_39)[name = tensor<string, []>("op_133_cast_fp16")];
-            tensor<int32, [4]> var_134 = const()[name = tensor<string, []>("op_134"), val = tensor<int32, [4]>([0, 2, 1, 3])];
-            tensor<int32, [3]> concat_0 = const()[name = tensor<string, []>("concat_0"), val = tensor<int32, [3]>([1, 1500, 384])];
-            tensor<fp16, [1, 1500, 6, 64]> transpose_36 = transpose(perm = var_134, x = var_133_cast_fp16)[name = tensor<string, []>("transpose_36")];
-            tensor<fp16, [1, 1500, 384]> x_11_cast_fp16 = reshape(shape = concat_0, x = transpose_36)[name = tensor<string, []>("x_11_cast_fp16")];
-            tensor<fp16, [384, 384]> var_139_to_fp16 = const()[name = tensor<string, []>("op_139_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(3112064)))];
-            tensor<fp16, [384]> var_140_to_fp16 = const()[name = tensor<string, []>("op_140_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(3407040)))];
-            tensor<fp16, [1, 1500, 384]> linear_3_cast_fp16 = linear(bias = var_140_to_fp16, weight = var_139_to_fp16, x = x_11_cast_fp16)[name = tensor<string, []>("linear_3_cast_fp16")];
-            tensor<fp16, [1, 1500, 384]> x_13_cast_fp16 = add(x = var_55_cast_fp16, y = linear_3_cast_fp16)[name = tensor<string, []>("x_13_cast_fp16")];
-            tensor<int32, [1]> var_147_axes_0 = const()[name = tensor<string, []>("op_147_axes_0"), val = tensor<int32, [1]>([-1])];
-            tensor<fp16, [384]> blocks_0_mlp_ln_weight_to_fp16 = const()[name = tensor<string, []>("blocks_0_mlp_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(3407872)))];
-            tensor<fp16, [384]> blocks_0_mlp_ln_bias_to_fp16 = const()[name = tensor<string, []>("blocks_0_mlp_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(3408704)))];
-            tensor<fp16, [1, 1500, 384]> var_147_cast_fp16 = layer_norm(axes = var_147_axes_0, beta = blocks_0_mlp_ln_bias_to_fp16, epsilon = var_73_to_fp16, gamma = blocks_0_mlp_ln_weight_to_fp16, x = x_13_cast_fp16)[name = tensor<string, []>("op_147_cast_fp16")];
-            tensor<fp16, [1536, 384]> var_156_to_fp16 = const()[name = tensor<string, []>("op_156_to_fp16"), val = tensor<fp16, [1536, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(3409536)))];
-            tensor<fp16, [1536]> var_157_to_fp16 = const()[name = tensor<string, []>("op_157_to_fp16"), val = tensor<fp16, [1536]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(4589248)))];
-            tensor<fp16, [1, 1500, 1536]> linear_4_cast_fp16 = linear(bias = var_157_to_fp16, weight = var_156_to_fp16, x = var_147_cast_fp16)[name = tensor<string, []>("linear_4_cast_fp16")];
-            tensor<string, []> x_17_mode_0 = const()[name = tensor<string, []>("x_17_mode_0"), val = tensor<string, []>("EXACT")];
-            tensor<fp16, [1, 1500, 1536]> x_17_cast_fp16 = gelu(mode = x_17_mode_0, x = linear_4_cast_fp16)[name = tensor<string, []>("x_17_cast_fp16")];
-            tensor<fp16, [384, 1536]> var_162_to_fp16 = const()[name = tensor<string, []>("op_162_to_fp16"), val = tensor<fp16, [384, 1536]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(4592384)))];
-            tensor<fp16, [384]> var_163_to_fp16 = const()[name = tensor<string, []>("op_163_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(5772096)))];
-            tensor<fp16, [1, 1500, 384]> linear_5_cast_fp16 = linear(bias = var_163_to_fp16, weight = var_162_to_fp16, x = x_17_cast_fp16)[name = tensor<string, []>("linear_5_cast_fp16")];
-            tensor<fp16, [1, 1500, 384]> x_19_cast_fp16 = add(x = x_13_cast_fp16, y = linear_5_cast_fp16)[name = tensor<string, []>("x_19_cast_fp16")];
-            tensor<int32, []> var_172 = const()[name = tensor<string, []>("op_172"), val = tensor<int32, []>(-1)];
-            tensor<int32, [1]> var_188_axes_0 = const()[name = tensor<string, []>("op_188_axes_0"), val = tensor<int32, [1]>([-1])];
-            tensor<fp16, [384]> blocks_1_attn_ln_weight_to_fp16 = const()[name = tensor<string, []>("blocks_1_attn_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(5772928)))];
-            tensor<fp16, [384]> blocks_1_attn_ln_bias_to_fp16 = const()[name = tensor<string, []>("blocks_1_attn_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(5773760)))];
-            tensor<fp16, []> var_178_to_fp16 = const()[name = tensor<string, []>("op_178_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
-            tensor<fp16, [1, 1500, 384]> var_188_cast_fp16 = layer_norm(axes = var_188_axes_0, beta = blocks_1_attn_ln_bias_to_fp16, epsilon = var_178_to_fp16, gamma = blocks_1_attn_ln_weight_to_fp16, x = x_19_cast_fp16)[name = tensor<string, []>("op_188_cast_fp16")];
-            tensor<fp16, [384, 384]> var_199_to_fp16 = const()[name = tensor<string, []>("op_199_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(5774592)))];
-            tensor<fp16, [384]> var_200_to_fp16 = const()[name = tensor<string, []>("op_200_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(6069568)))];
-            tensor<fp16, [1, 1500, 384]> linear_6_cast_fp16 = linear(bias = var_200_to_fp16, weight = var_199_to_fp16, x = var_188_cast_fp16)[name = tensor<string, []>("linear_6_cast_fp16")];
-            tensor<fp16, [384, 384]> var_203_to_fp16 = const()[name = tensor<string, []>("op_203_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(6070400)))];
-            tensor<fp16, [1, 1500, 384]> linear_7_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = var_203_to_fp16, x = var_188_cast_fp16)[name = tensor<string, []>("linear_7_cast_fp16")];
-            tensor<fp16, [384, 384]> var_207_to_fp16 = const()[name = tensor<string, []>("op_207_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(6365376)))];
-            tensor<fp16, [384]> var_208_to_fp16 = const()[name = tensor<string, []>("op_208_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(6660352)))];
-            tensor<fp16, [1, 1500, 384]> linear_8_cast_fp16 = linear(bias = var_208_to_fp16, weight = var_207_to_fp16, x = var_188_cast_fp16)[name = tensor<string, []>("linear_8_cast_fp16")];
-            tensor<int32, [4]> var_216 = const()[name = tensor<string, []>("op_216"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_217_cast_fp16 = reshape(shape = var_216, x = linear_6_cast_fp16)[name = tensor<string, []>("op_217_cast_fp16")];
-            tensor<fp16, [1, 1, 1, 1]> const_30_to_fp16 = const()[name = tensor<string, []>("const_30_to_fp16"), val = tensor<fp16, [1, 1, 1, 1]>([[[[0x1.6ap-2]]]])];
-            tensor<fp16, [1, 1500, 6, 64]> q_7_cast_fp16 = mul(x = var_217_cast_fp16, y = const_30_to_fp16)[name = tensor<string, []>("q_7_cast_fp16")];
-            tensor<int32, [4]> var_223 = const()[name = tensor<string, []>("op_223"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_224_cast_fp16 = reshape(shape = var_223, x = linear_7_cast_fp16)[name = tensor<string, []>("op_224_cast_fp16")];
-            tensor<fp16, [1, 1, 1, 1]> const_31_to_fp16 = const()[name = tensor<string, []>("const_31_to_fp16"), val = tensor<fp16, [1, 1, 1, 1]>([[[[0x1.6ap-2]]]])];
-            tensor<fp16, [1, 1500, 6, 64]> k_7_cast_fp16 = mul(x = var_224_cast_fp16, y = const_31_to_fp16)[name = tensor<string, []>("k_7_cast_fp16")];
-            tensor<int32, [4]> var_230 = const()[name = tensor<string, []>("op_230"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_231_cast_fp16 = reshape(shape = var_230, x = linear_8_cast_fp16)[name = tensor<string, []>("op_231_cast_fp16")];
-            tensor<int32, [4]> var_232 = const()[name = tensor<string, []>("op_232"), val = tensor<int32, [4]>([0, 2, 1, 3])];
-            tensor<bool, []> qk_3_transpose_x_0 = const()[name = tensor<string, []>("qk_3_transpose_x_0"), val = tensor<bool, []>(false)];
-            tensor<bool, []> qk_3_transpose_y_0 = const()[name = tensor<string, []>("qk_3_transpose_y_0"), val = tensor<bool, []>(false)];
-            tensor<int32, [4]> transpose_18_perm_0 = const()[name = tensor<string, []>("transpose_18_perm_0"), val = tensor<int32, [4]>([0, 2, -3, -1])];
-            tensor<int32, [4]> transpose_19_perm_0 = const()[name = tensor<string, []>("transpose_19_perm_0"), val = tensor<int32, [4]>([0, 2, -1, -3])];
-            tensor<fp16, [1, 6, 64, 1500]> transpose_33 = transpose(perm = transpose_19_perm_0, x = k_7_cast_fp16)[name = tensor<string, []>("transpose_33")];
-            tensor<fp16, [1, 6, 1500, 64]> transpose_34 = transpose(perm = transpose_18_perm_0, x = q_7_cast_fp16)[name = tensor<string, []>("transpose_34")];
-            tensor<fp16, [1, 6, 1500, 1500]> qk_3_cast_fp16 = matmul(transpose_x = qk_3_transpose_x_0, transpose_y = qk_3_transpose_y_0, x = transpose_34, y = transpose_33)[name = tensor<string, []>("qk_3_cast_fp16")];
-            tensor<fp16, [1, 6, 1500, 1500]> var_236_cast_fp16 = softmax(axis = var_172, x = qk_3_cast_fp16)[name = tensor<string, []>("op_236_cast_fp16")];
-            tensor<bool, []> var_238_transpose_x_0 = const()[name = tensor<string, []>("op_238_transpose_x_0"), val = tensor<bool, []>(false)];
-            tensor<bool, []> var_238_transpose_y_0 = const()[name = tensor<string, []>("op_238_transpose_y_0"), val = tensor<bool, []>(false)];
-            tensor<fp16, [1, 6, 1500, 64]> transpose_35 = transpose(perm = var_232, x = var_231_cast_fp16)[name = tensor<string, []>("transpose_35")];
-            tensor<fp16, [1, 6, 1500, 64]> var_238_cast_fp16 = matmul(transpose_x = var_238_transpose_x_0, transpose_y = var_238_transpose_y_0, x = var_236_cast_fp16, y = transpose_35)[name = tensor<string, []>("op_238_cast_fp16")];
-            tensor<int32, [4]> var_239 = const()[name = tensor<string, []>("op_239"), val = tensor<int32, [4]>([0, 2, 1, 3])];
-            tensor<int32, [3]> concat_1 = const()[name = tensor<string, []>("concat_1"), val = tensor<int32, [3]>([1, 1500, 384])];
-            tensor<fp16, [1, 1500, 6, 64]> transpose_32 = transpose(perm = var_239, x = var_238_cast_fp16)[name = tensor<string, []>("transpose_32")];
-            tensor<fp16, [1, 1500, 384]> x_23_cast_fp16 = reshape(shape = concat_1, x = transpose_32)[name = tensor<string, []>("x_23_cast_fp16")];
-            tensor<fp16, [384, 384]> var_244_to_fp16 = const()[name = tensor<string, []>("op_244_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(6661184)))];
-            tensor<fp16, [384]> var_245_to_fp16 = const()[name = tensor<string, []>("op_245_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(6956160)))];
-            tensor<fp16, [1, 1500, 384]> linear_9_cast_fp16 = linear(bias = var_245_to_fp16, weight = var_244_to_fp16, x = x_23_cast_fp16)[name = tensor<string, []>("linear_9_cast_fp16")];
-            tensor<fp16, [1, 1500, 384]> x_25_cast_fp16 = add(x = x_19_cast_fp16, y = linear_9_cast_fp16)[name = tensor<string, []>("x_25_cast_fp16")];
-            tensor<int32, [1]> var_252_axes_0 = const()[name = tensor<string, []>("op_252_axes_0"), val = tensor<int32, [1]>([-1])];
-            tensor<fp16, [384]> blocks_1_mlp_ln_weight_to_fp16 = const()[name = tensor<string, []>("blocks_1_mlp_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(6956992)))];
-            tensor<fp16, [384]> blocks_1_mlp_ln_bias_to_fp16 = const()[name = tensor<string, []>("blocks_1_mlp_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(6957824)))];
-            tensor<fp16, [1, 1500, 384]> var_252_cast_fp16 = layer_norm(axes = var_252_axes_0, beta = blocks_1_mlp_ln_bias_to_fp16, epsilon = var_178_to_fp16, gamma = blocks_1_mlp_ln_weight_to_fp16, x = x_25_cast_fp16)[name = tensor<string, []>("op_252_cast_fp16")];
-            tensor<fp16, [1536, 384]> var_261_to_fp16 = const()[name = tensor<string, []>("op_261_to_fp16"), val = tensor<fp16, [1536, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(6958656)))];
-            tensor<fp16, [1536]> var_262_to_fp16 = const()[name = tensor<string, []>("op_262_to_fp16"), val = tensor<fp16, [1536]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(8138368)))];
-            tensor<fp16, [1, 1500, 1536]> linear_10_cast_fp16 = linear(bias = var_262_to_fp16, weight = var_261_to_fp16, x = var_252_cast_fp16)[name = tensor<string, []>("linear_10_cast_fp16")];
-            tensor<string, []> x_29_mode_0 = const()[name = tensor<string, []>("x_29_mode_0"), val = tensor<string, []>("EXACT")];
-            tensor<fp16, [1, 1500, 1536]> x_29_cast_fp16 = gelu(mode = x_29_mode_0, x = linear_10_cast_fp16)[name = tensor<string, []>("x_29_cast_fp16")];
-            tensor<fp16, [384, 1536]> var_267_to_fp16 = const()[name = tensor<string, []>("op_267_to_fp16"), val = tensor<fp16, [384, 1536]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(8141504)))];
-            tensor<fp16, [384]> var_268_to_fp16 = const()[name = tensor<string, []>("op_268_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(9321216)))];
-            tensor<fp16, [1, 1500, 384]> linear_11_cast_fp16 = linear(bias = var_268_to_fp16, weight = var_267_to_fp16, x = x_29_cast_fp16)[name = tensor<string, []>("linear_11_cast_fp16")];
-            tensor<fp16, [1, 1500, 384]> x_31_cast_fp16 = add(x = x_25_cast_fp16, y = linear_11_cast_fp16)[name = tensor<string, []>("x_31_cast_fp16")];
-            tensor<int32, []> var_277 = const()[name = tensor<string, []>("op_277"), val = tensor<int32, []>(-1)];
-            tensor<int32, [1]> var_293_axes_0 = const()[name = tensor<string, []>("op_293_axes_0"), val = tensor<int32, [1]>([-1])];
-            tensor<fp16, [384]> blocks_2_attn_ln_weight_to_fp16 = const()[name = tensor<string, []>("blocks_2_attn_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(9322048)))];
-            tensor<fp16, [384]> blocks_2_attn_ln_bias_to_fp16 = const()[name = tensor<string, []>("blocks_2_attn_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(9322880)))];
-            tensor<fp16, []> var_283_to_fp16 = const()[name = tensor<string, []>("op_283_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
-            tensor<fp16, [1, 1500, 384]> var_293_cast_fp16 = layer_norm(axes = var_293_axes_0, beta = blocks_2_attn_ln_bias_to_fp16, epsilon = var_283_to_fp16, gamma = blocks_2_attn_ln_weight_to_fp16, x = x_31_cast_fp16)[name = tensor<string, []>("op_293_cast_fp16")];
-            tensor<fp16, [384, 384]> var_304_to_fp16 = const()[name = tensor<string, []>("op_304_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(9323712)))];
-            tensor<fp16, [384]> var_305_to_fp16 = const()[name = tensor<string, []>("op_305_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(9618688)))];
-            tensor<fp16, [1, 1500, 384]> linear_12_cast_fp16 = linear(bias = var_305_to_fp16, weight = var_304_to_fp16, x = var_293_cast_fp16)[name = tensor<string, []>("linear_12_cast_fp16")];
-            tensor<fp16, [384, 384]> var_308_to_fp16 = const()[name = tensor<string, []>("op_308_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(9619520)))];
-            tensor<fp16, [1, 1500, 384]> linear_13_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = var_308_to_fp16, x = var_293_cast_fp16)[name = tensor<string, []>("linear_13_cast_fp16")];
-            tensor<fp16, [384, 384]> var_312_to_fp16 = const()[name = tensor<string, []>("op_312_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(9914496)))];
-            tensor<fp16, [384]> var_313_to_fp16 = const()[name = tensor<string, []>("op_313_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(10209472)))];
-            tensor<fp16, [1, 1500, 384]> linear_14_cast_fp16 = linear(bias = var_313_to_fp16, weight = var_312_to_fp16, x = var_293_cast_fp16)[name = tensor<string, []>("linear_14_cast_fp16")];
-            tensor<int32, [4]> var_321 = const()[name = tensor<string, []>("op_321"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_322_cast_fp16 = reshape(shape = var_321, x = linear_12_cast_fp16)[name = tensor<string, []>("op_322_cast_fp16")];
-            tensor<fp16, [1, 1, 1, 1]> const_32_to_fp16 = const()[name = tensor<string, []>("const_32_to_fp16"), val = tensor<fp16, [1, 1, 1, 1]>([[[[0x1.6ap-2]]]])];
-            tensor<fp16, [1, 1500, 6, 64]> q_11_cast_fp16 = mul(x = var_322_cast_fp16, y = const_32_to_fp16)[name = tensor<string, []>("q_11_cast_fp16")];
-            tensor<int32, [4]> var_328 = const()[name = tensor<string, []>("op_328"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_329_cast_fp16 = reshape(shape = var_328, x = linear_13_cast_fp16)[name = tensor<string, []>("op_329_cast_fp16")];
-            tensor<fp16, [1, 1, 1, 1]> const_33_to_fp16 = const()[name = tensor<string, []>("const_33_to_fp16"), val = tensor<fp16, [1, 1, 1, 1]>([[[[0x1.6ap-2]]]])];
-            tensor<fp16, [1, 1500, 6, 64]> k_11_cast_fp16 = mul(x = var_329_cast_fp16, y = const_33_to_fp16)[name = tensor<string, []>("k_11_cast_fp16")];
-            tensor<int32, [4]> var_335 = const()[name = tensor<string, []>("op_335"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_336_cast_fp16 = reshape(shape = var_335, x = linear_14_cast_fp16)[name = tensor<string, []>("op_336_cast_fp16")];
-            tensor<int32, [4]> var_337 = const()[name = tensor<string, []>("op_337"), val = tensor<int32, [4]>([0, 2, 1, 3])];
-            tensor<bool, []> qk_5_transpose_x_0 = const()[name = tensor<string, []>("qk_5_transpose_x_0"), val = tensor<bool, []>(false)];
-            tensor<bool, []> qk_5_transpose_y_0 = const()[name = tensor<string, []>("qk_5_transpose_y_0"), val = tensor<bool, []>(false)];
-            tensor<int32, [4]> transpose_20_perm_0 = const()[name = tensor<string, []>("transpose_20_perm_0"), val = tensor<int32, [4]>([0, 2, -3, -1])];
-            tensor<int32, [4]> transpose_21_perm_0 = const()[name = tensor<string, []>("transpose_21_perm_0"), val = tensor<int32, [4]>([0, 2, -1, -3])];
-            tensor<fp16, [1, 6, 64, 1500]> transpose_29 = transpose(perm = transpose_21_perm_0, x = k_11_cast_fp16)[name = tensor<string, []>("transpose_29")];
-            tensor<fp16, [1, 6, 1500, 64]> transpose_30 = transpose(perm = transpose_20_perm_0, x = q_11_cast_fp16)[name = tensor<string, []>("transpose_30")];
-            tensor<fp16, [1, 6, 1500, 1500]> qk_5_cast_fp16 = matmul(transpose_x = qk_5_transpose_x_0, transpose_y = qk_5_transpose_y_0, x = transpose_30, y = transpose_29)[name = tensor<string, []>("qk_5_cast_fp16")];
-            tensor<fp16, [1, 6, 1500, 1500]> var_341_cast_fp16 = softmax(axis = var_277, x = qk_5_cast_fp16)[name = tensor<string, []>("op_341_cast_fp16")];
-            tensor<bool, []> var_343_transpose_x_0 = const()[name = tensor<string, []>("op_343_transpose_x_0"), val = tensor<bool, []>(false)];
-            tensor<bool, []> var_343_transpose_y_0 = const()[name = tensor<string, []>("op_343_transpose_y_0"), val = tensor<bool, []>(false)];
-            tensor<fp16, [1, 6, 1500, 64]> transpose_31 = transpose(perm = var_337, x = var_336_cast_fp16)[name = tensor<string, []>("transpose_31")];
-            tensor<fp16, [1, 6, 1500, 64]> var_343_cast_fp16 = matmul(transpose_x = var_343_transpose_x_0, transpose_y = var_343_transpose_y_0, x = var_341_cast_fp16, y = transpose_31)[name = tensor<string, []>("op_343_cast_fp16")];
-            tensor<int32, [4]> var_344 = const()[name = tensor<string, []>("op_344"), val = tensor<int32, [4]>([0, 2, 1, 3])];
-            tensor<int32, [3]> concat_2 = const()[name = tensor<string, []>("concat_2"), val = tensor<int32, [3]>([1, 1500, 384])];
-            tensor<fp16, [1, 1500, 6, 64]> transpose_28 = transpose(perm = var_344, x = var_343_cast_fp16)[name = tensor<string, []>("transpose_28")];
-            tensor<fp16, [1, 1500, 384]> x_35_cast_fp16 = reshape(shape = concat_2, x = transpose_28)[name = tensor<string, []>("x_35_cast_fp16")];
-            tensor<fp16, [384, 384]> var_349_to_fp16 = const()[name = tensor<string, []>("op_349_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(10210304)))];
-            tensor<fp16, [384]> var_350_to_fp16 = const()[name = tensor<string, []>("op_350_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(10505280)))];
-            tensor<fp16, [1, 1500, 384]> linear_15_cast_fp16 = linear(bias = var_350_to_fp16, weight = var_349_to_fp16, x = x_35_cast_fp16)[name = tensor<string, []>("linear_15_cast_fp16")];
-            tensor<fp16, [1, 1500, 384]> x_37_cast_fp16 = add(x = x_31_cast_fp16, y = linear_15_cast_fp16)[name = tensor<string, []>("x_37_cast_fp16")];
-            tensor<int32, [1]> var_357_axes_0 = const()[name = tensor<string, []>("op_357_axes_0"), val = tensor<int32, [1]>([-1])];
-            tensor<fp16, [384]> blocks_2_mlp_ln_weight_to_fp16 = const()[name = tensor<string, []>("blocks_2_mlp_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(10506112)))];
-            tensor<fp16, [384]> blocks_2_mlp_ln_bias_to_fp16 = const()[name = tensor<string, []>("blocks_2_mlp_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(10506944)))];
-            tensor<fp16, [1, 1500, 384]> var_357_cast_fp16 = layer_norm(axes = var_357_axes_0, beta = blocks_2_mlp_ln_bias_to_fp16, epsilon = var_283_to_fp16, gamma = blocks_2_mlp_ln_weight_to_fp16, x = x_37_cast_fp16)[name = tensor<string, []>("op_357_cast_fp16")];
-            tensor<fp16, [1536, 384]> var_366_to_fp16 = const()[name = tensor<string, []>("op_366_to_fp16"), val = tensor<fp16, [1536, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(10507776)))];
-            tensor<fp16, [1536]> var_367_to_fp16 = const()[name = tensor<string, []>("op_367_to_fp16"), val = tensor<fp16, [1536]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(11687488)))];
-            tensor<fp16, [1, 1500, 1536]> linear_16_cast_fp16 = linear(bias = var_367_to_fp16, weight = var_366_to_fp16, x = var_357_cast_fp16)[name = tensor<string, []>("linear_16_cast_fp16")];
-            tensor<string, []> x_41_mode_0 = const()[name = tensor<string, []>("x_41_mode_0"), val = tensor<string, []>("EXACT")];
-            tensor<fp16, [1, 1500, 1536]> x_41_cast_fp16 = gelu(mode = x_41_mode_0, x = linear_16_cast_fp16)[name = tensor<string, []>("x_41_cast_fp16")];
-            tensor<fp16, [384, 1536]> var_372_to_fp16 = const()[name = tensor<string, []>("op_372_to_fp16"), val = tensor<fp16, [384, 1536]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(11690624)))];
-            tensor<fp16, [384]> var_373_to_fp16 = const()[name = tensor<string, []>("op_373_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(12870336)))];
-            tensor<fp16, [1, 1500, 384]> linear_17_cast_fp16 = linear(bias = var_373_to_fp16, weight = var_372_to_fp16, x = x_41_cast_fp16)[name = tensor<string, []>("linear_17_cast_fp16")];
-            tensor<fp16, [1, 1500, 384]> x_43_cast_fp16 = add(x = x_37_cast_fp16, y = linear_17_cast_fp16)[name = tensor<string, []>("x_43_cast_fp16")];
-            tensor<int32, []> var_382 = const()[name = tensor<string, []>("op_382"), val = tensor<int32, []>(-1)];
-            tensor<int32, [1]> var_398_axes_0 = const()[name = tensor<string, []>("op_398_axes_0"), val = tensor<int32, [1]>([-1])];
-            tensor<fp16, [384]> blocks_3_attn_ln_weight_to_fp16 = const()[name = tensor<string, []>("blocks_3_attn_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(12871168)))];
-            tensor<fp16, [384]> blocks_3_attn_ln_bias_to_fp16 = const()[name = tensor<string, []>("blocks_3_attn_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(12872000)))];
-            tensor<fp16, []> var_388_to_fp16 = const()[name = tensor<string, []>("op_388_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
-            tensor<fp16, [1, 1500, 384]> var_398_cast_fp16 = layer_norm(axes = var_398_axes_0, beta = blocks_3_attn_ln_bias_to_fp16, epsilon = var_388_to_fp16, gamma = blocks_3_attn_ln_weight_to_fp16, x = x_43_cast_fp16)[name = tensor<string, []>("op_398_cast_fp16")];
-            tensor<fp16, [384, 384]> var_409_to_fp16 = const()[name = tensor<string, []>("op_409_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(12872832)))];
-            tensor<fp16, [384]> var_410_to_fp16 = const()[name = tensor<string, []>("op_410_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(13167808)))];
-            tensor<fp16, [1, 1500, 384]> linear_18_cast_fp16 = linear(bias = var_410_to_fp16, weight = var_409_to_fp16, x = var_398_cast_fp16)[name = tensor<string, []>("linear_18_cast_fp16")];
-            tensor<fp16, [384, 384]> var_413_to_fp16 = const()[name = tensor<string, []>("op_413_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(13168640)))];
-            tensor<fp16, [1, 1500, 384]> linear_19_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = var_413_to_fp16, x = var_398_cast_fp16)[name = tensor<string, []>("linear_19_cast_fp16")];
-            tensor<fp16, [384, 384]> var_417_to_fp16 = const()[name = tensor<string, []>("op_417_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(13463616)))];
-            tensor<fp16, [384]> var_418_to_fp16 = const()[name = tensor<string, []>("op_418_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(13758592)))];
-            tensor<fp16, [1, 1500, 384]> linear_20_cast_fp16 = linear(bias = var_418_to_fp16, weight = var_417_to_fp16, x = var_398_cast_fp16)[name = tensor<string, []>("linear_20_cast_fp16")];
-            tensor<int32, [4]> var_426 = const()[name = tensor<string, []>("op_426"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_427_cast_fp16 = reshape(shape = var_426, x = linear_18_cast_fp16)[name = tensor<string, []>("op_427_cast_fp16")];
-            tensor<fp16, [1, 1, 1, 1]> const_34_to_fp16 = const()[name = tensor<string, []>("const_34_to_fp16"), val = tensor<fp16, [1, 1, 1, 1]>([[[[0x1.6ap-2]]]])];
-            tensor<fp16, [1, 1500, 6, 64]> q_cast_fp16 = mul(x = var_427_cast_fp16, y = const_34_to_fp16)[name = tensor<string, []>("q_cast_fp16")];
-            tensor<int32, [4]> var_433 = const()[name = tensor<string, []>("op_433"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_434_cast_fp16 = reshape(shape = var_433, x = linear_19_cast_fp16)[name = tensor<string, []>("op_434_cast_fp16")];
-            tensor<fp16, [1, 1, 1, 1]> const_35_to_fp16 = const()[name = tensor<string, []>("const_35_to_fp16"), val = tensor<fp16, [1, 1, 1, 1]>([[[[0x1.6ap-2]]]])];
-            tensor<fp16, [1, 1500, 6, 64]> k_cast_fp16 = mul(x = var_434_cast_fp16, y = const_35_to_fp16)[name = tensor<string, []>("k_cast_fp16")];
-            tensor<int32, [4]> var_440 = const()[name = tensor<string, []>("op_440"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
-            tensor<fp16, [1, 1500, 6, 64]> var_441_cast_fp16 = reshape(shape = var_440, x = linear_20_cast_fp16)[name = tensor<string, []>("op_441_cast_fp16")];
-            tensor<int32, [4]> var_442 = const()[name = tensor<string, []>("op_442"), val = tensor<int32, [4]>([0, 2, 1, 3])];
-            tensor<bool, []> qk_transpose_x_0 = const()[name = tensor<string, []>("qk_transpose_x_0"), val = tensor<bool, []>(false)];
-            tensor<bool, []> qk_transpose_y_0 = const()[name = tensor<string, []>("qk_transpose_y_0"), val = tensor<bool, []>(false)];
-            tensor<int32, [4]> transpose_22_perm_0 = const()[name = tensor<string, []>("transpose_22_perm_0"), val = tensor<int32, [4]>([0, 2, -3, -1])];
-            tensor<int32, [4]> transpose_23_perm_0 = const()[name = tensor<string, []>("transpose_23_perm_0"), val = tensor<int32, [4]>([0, 2, -1, -3])];
-            tensor<fp16, [1, 6, 64, 1500]> transpose_25 = transpose(perm = transpose_23_perm_0, x = k_cast_fp16)[name = tensor<string, []>("transpose_25")];
-            tensor<fp16, [1, 6, 1500, 64]> transpose_26 = transpose(perm = transpose_22_perm_0, x = q_cast_fp16)[name = tensor<string, []>("transpose_26")];
-            tensor<fp16, [1, 6, 1500, 1500]> qk_cast_fp16 = matmul(transpose_x = qk_transpose_x_0, transpose_y = qk_transpose_y_0, x = transpose_26, y = transpose_25)[name = tensor<string, []>("qk_cast_fp16")];
-            tensor<fp16, [1, 6, 1500, 1500]> var_446_cast_fp16 = softmax(axis = var_382, x = qk_cast_fp16)[name = tensor<string, []>("op_446_cast_fp16")];
-            tensor<bool, []> var_448_transpose_x_0 = const()[name = tensor<string, []>("op_448_transpose_x_0"), val = tensor<bool, []>(false)];
-            tensor<bool, []> var_448_transpose_y_0 = const()[name = tensor<string, []>("op_448_transpose_y_0"), val = tensor<bool, []>(false)];
-            tensor<fp16, [1, 6, 1500, 64]> transpose_27 = transpose(perm = var_442, x = var_441_cast_fp16)[name = tensor<string, []>("transpose_27")];
-            tensor<fp16, [1, 6, 1500, 64]> var_448_cast_fp16 = matmul(transpose_x = var_448_transpose_x_0, transpose_y = var_448_transpose_y_0, x = var_446_cast_fp16, y = transpose_27)[name = tensor<string, []>("op_448_cast_fp16")];
-            tensor<int32, [4]> var_449 = const()[name = tensor<string, []>("op_449"), val = tensor<int32, [4]>([0, 2, 1, 3])];
-            tensor<int32, [3]> concat_3 = const()[name = tensor<string, []>("concat_3"), val = tensor<int32, [3]>([1, 1500, 384])];
-            tensor<fp16, [1, 1500, 6, 64]> transpose_24 = transpose(perm = var_449, x = var_448_cast_fp16)[name = tensor<string, []>("transpose_24")];
-            tensor<fp16, [1, 1500, 384]> x_47_cast_fp16 = reshape(shape = concat_3, x = transpose_24)[name = tensor<string, []>("x_47_cast_fp16")];
-            tensor<fp16, [384, 384]> var_454_to_fp16 = const()[name = tensor<string, []>("op_454_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(13759424)))];
-            tensor<fp16, [384]> var_455_to_fp16 = const()[name = tensor<string, []>("op_455_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(14054400)))];
-            tensor<fp16, [1, 1500, 384]> linear_21_cast_fp16 = linear(bias = var_455_to_fp16, weight = var_454_to_fp16, x = x_47_cast_fp16)[name = tensor<string, []>("linear_21_cast_fp16")];
-            tensor<fp16, [1, 1500, 384]> x_49_cast_fp16 = add(x = x_43_cast_fp16, y = linear_21_cast_fp16)[name = tensor<string, []>("x_49_cast_fp16")];
-            tensor<int32, [1]> var_462_axes_0 = const()[name = tensor<string, []>("op_462_axes_0"), val = tensor<int32, [1]>([-1])];
-            tensor<fp16, [384]> blocks_3_mlp_ln_weight_to_fp16 = const()[name = tensor<string, []>("blocks_3_mlp_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(14055232)))];
-            tensor<fp16, [384]> blocks_3_mlp_ln_bias_to_fp16 = const()[name = tensor<string, []>("blocks_3_mlp_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(14056064)))];
-            tensor<fp16, [1, 1500, 384]> var_462_cast_fp16 = layer_norm(axes = var_462_axes_0, beta = blocks_3_mlp_ln_bias_to_fp16, epsilon = var_388_to_fp16, gamma = blocks_3_mlp_ln_weight_to_fp16, x = x_49_cast_fp16)[name = tensor<string, []>("op_462_cast_fp16")];
-            tensor<fp16, [1536, 384]> var_471_to_fp16 = const()[name = tensor<string, []>("op_471_to_fp16"), val = tensor<fp16, [1536, 384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(14056896)))];
-            tensor<fp16, [1536]> var_472_to_fp16 = const()[name = tensor<string, []>("op_472_to_fp16"), val = tensor<fp16, [1536]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(15236608)))];
-            tensor<fp16, [1, 1500, 1536]> linear_22_cast_fp16 = linear(bias = var_472_to_fp16, weight = var_471_to_fp16, x = var_462_cast_fp16)[name = tensor<string, []>("linear_22_cast_fp16")];
-            tensor<string, []> x_53_mode_0 = const()[name = tensor<string, []>("x_53_mode_0"), val = tensor<string, []>("EXACT")];
-            tensor<fp16, [1, 1500, 1536]> x_53_cast_fp16 = gelu(mode = x_53_mode_0, x = linear_22_cast_fp16)[name = tensor<string, []>("x_53_cast_fp16")];
-            tensor<fp16, [384, 1536]> var_477_to_fp16 = const()[name = tensor<string, []>("op_477_to_fp16"), val = tensor<fp16, [384, 1536]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(15239744)))];
-            tensor<fp16, [384]> var_478_to_fp16 = const()[name = tensor<string, []>("op_478_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(16419456)))];
-            tensor<fp16, [1, 1500, 384]> linear_23_cast_fp16 = linear(bias = var_478_to_fp16, weight = var_477_to_fp16, x = x_53_cast_fp16)[name = tensor<string, []>("linear_23_cast_fp16")];
-            tensor<fp16, [1, 1500, 384]> x_cast_fp16 = add(x = x_49_cast_fp16, y = linear_23_cast_fp16)[name = tensor<string, []>("x_cast_fp16")];
-            tensor<int32, [1]> var_491_axes_0 = const()[name = tensor<string, []>("op_491_axes_0"), val = tensor<int32, [1]>([-1])];
-            tensor<fp16, [384]> ln_post_weight_to_fp16 = const()[name = tensor<string, []>("ln_post_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(16420288)))];
-            tensor<fp16, [384]> ln_post_bias_to_fp16 = const()[name = tensor<string, []>("ln_post_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(16421120)))];
-            tensor<fp16, []> var_482_to_fp16 = const()[name = tensor<string, []>("op_482_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
-            tensor<fp16, [1, 1500, 384]> output = layer_norm(axes = var_491_axes_0, beta = ln_post_bias_to_fp16, epsilon = var_482_to_fp16, gamma = ln_post_weight_to_fp16, x = x_cast_fp16)[name = tensor<string, []>("op_491_cast_fp16")];
         } -> (output);
 }

+program(1.3)
+[buildInfo = dict<string, string>({{"coremlc-component-MIL", "3510.2.1"}, {"coremlc-version", "3500.32.1"}, {"coremltools-component-torch", "2.9.1"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "9.0"}})]
 {
+    func main<ios18>(tensor<fp16, [1, 80, 3000]> logmel_data) {
+            string var_28_pad_type_0 = const()[name = string("op_28_pad_type_0"), val = string("custom")];
+            tensor<int32, [2]> var_28_pad_0 = const()[name = string("op_28_pad_0"), val = tensor<int32, [2]>([1, 1])];
+            tensor<int32, [1]> var_28_strides_0 = const()[name = string("op_28_strides_0"), val = tensor<int32, [1]>([1])];
+            tensor<int32, [1]> var_28_dilations_0 = const()[name = string("op_28_dilations_0"), val = tensor<int32, [1]>([1])];
+            int32 var_28_groups_0 = const()[name = string("op_28_groups_0"), val = int32(1)];
+            tensor<fp16, [384, 80, 3]> const_0_to_fp16 = const()[name = string("const_0_to_fp16"), val = tensor<fp16, [384, 80, 3]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64)))];
+            tensor<fp16, [384]> const_1_to_fp16 = const()[name = string("const_1_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(184448)))];
+            tensor<fp16, [1, 384, 3000]> var_28_cast_fp16 = conv(bias = const_1_to_fp16, dilations = var_28_dilations_0, groups = var_28_groups_0, pad = var_28_pad_0, pad_type = var_28_pad_type_0, strides = var_28_strides_0, weight = const_0_to_fp16, x = logmel_data)[name = string("op_28_cast_fp16")];
+            string input_1_mode_0 = const()[name = string("input_1_mode_0"), val = string("EXACT")];
+            tensor<fp16, [1, 384, 3000]> input_1_cast_fp16 = gelu(mode = input_1_mode_0, x = var_28_cast_fp16)[name = string("input_1_cast_fp16")];
+            string var_46_pad_type_0 = const()[name = string("op_46_pad_type_0"), val = string("custom")];
+            tensor<int32, [2]> var_46_pad_0 = const()[name = string("op_46_pad_0"), val = tensor<int32, [2]>([1, 1])];
+            tensor<int32, [1]> var_46_strides_0 = const()[name = string("op_46_strides_0"), val = tensor<int32, [1]>([2])];
+            tensor<int32, [1]> var_46_dilations_0 = const()[name = string("op_46_dilations_0"), val = tensor<int32, [1]>([1])];
+            int32 var_46_groups_0 = const()[name = string("op_46_groups_0"), val = int32(1)];
+            tensor<fp16, [384, 384, 3]> const_2_to_fp16 = const()[name = string("const_2_to_fp16"), val = tensor<fp16, [384, 384, 3]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(185280)))];
+            tensor<fp16, [384]> const_3_to_fp16 = const()[name = string("const_3_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1070080)))];
+            tensor<fp16, [1, 384, 1500]> var_46_cast_fp16 = conv(bias = const_3_to_fp16, dilations = var_46_dilations_0, groups = var_46_groups_0, pad = var_46_pad_0, pad_type = var_46_pad_type_0, strides = var_46_strides_0, weight = const_2_to_fp16, x = input_1_cast_fp16)[name = string("op_46_cast_fp16")];
+            string x_3_mode_0 = const()[name = string("x_3_mode_0"), val = string("EXACT")];
+            tensor<fp16, [1, 384, 1500]> x_3_cast_fp16 = gelu(mode = x_3_mode_0, x = var_46_cast_fp16)[name = string("x_3_cast_fp16")];
+            tensor<int32, [3]> var_52 = const()[name = string("op_52"), val = tensor<int32, [3]>([0, 2, 1])];
+            tensor<fp16, [1500, 384]> positional_embedding_to_fp16 = const()[name = string("positional_embedding_to_fp16"), val = tensor<fp16, [1500, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1070912)))];
+            tensor<fp16, [1, 1500, 384]> x_5_cast_fp16 = transpose(perm = var_52, x = x_3_cast_fp16)[name = string("transpose_52")];
+            tensor<fp16, [1, 1500, 384]> var_55_cast_fp16 = add(x = x_5_cast_fp16, y = positional_embedding_to_fp16)[name = string("op_55_cast_fp16")];
+            tensor<int32, [1]> var_82_axes_0 = const()[name = string("op_82_axes_0"), val = tensor<int32, [1]>([-1])];
+            tensor<fp16, [384]> blocks_0_attn_ln_weight_to_fp16 = const()[name = string("blocks_0_attn_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2222976)))];
+            tensor<fp16, [384]> blocks_0_attn_ln_bias_to_fp16 = const()[name = string("blocks_0_attn_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2223808)))];
+            fp16 var_72_to_fp16 = const()[name = string("op_72_to_fp16"), val = fp16(0x1.5p-17)];
+            tensor<fp16, [1, 1500, 384]> var_82_cast_fp16 = layer_norm(axes = var_82_axes_0, beta = blocks_0_attn_ln_bias_to_fp16, epsilon = var_72_to_fp16, gamma = blocks_0_attn_ln_weight_to_fp16, x = var_55_cast_fp16)[name = string("op_82_cast_fp16")];
+            tensor<fp16, [384, 384]> const_4_to_fp16 = const()[name = string("const_4_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2224640)))];
+            tensor<fp16, [384]> const_5_to_fp16 = const()[name = string("const_5_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2519616)))];
+            tensor<fp16, [1, 1500, 384]> linear_0_cast_fp16 = linear(bias = const_5_to_fp16, weight = const_4_to_fp16, x = var_82_cast_fp16)[name = string("linear_0_cast_fp16")];
+            tensor<fp16, [384, 384]> const_6_to_fp16 = const()[name = string("const_6_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2520448)))];
+            tensor<fp16, [384]> linear_1_bias_0_to_fp16 = const()[name = string("linear_1_bias_0_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2815424)))];
+            tensor<fp16, [1, 1500, 384]> linear_1_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = const_6_to_fp16, x = var_82_cast_fp16)[name = string("linear_1_cast_fp16")];
+            tensor<fp16, [384, 384]> const_7_to_fp16 = const()[name = string("const_7_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2816256)))];
+            tensor<fp16, [384]> const_8_to_fp16 = const()[name = string("const_8_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3111232)))];
+            tensor<fp16, [1, 1500, 384]> linear_2_cast_fp16 = linear(bias = const_8_to_fp16, weight = const_7_to_fp16, x = var_82_cast_fp16)[name = string("linear_2_cast_fp16")];
+            tensor<int32, [4]> var_106 = const()[name = string("op_106"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_107_cast_fp16 = reshape(shape = var_106, x = linear_0_cast_fp16)[name = string("op_107_cast_fp16")];
+            tensor<int32, [4]> var_112 = const()[name = string("op_112"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_113_cast_fp16 = reshape(shape = var_112, x = linear_1_cast_fp16)[name = string("op_113_cast_fp16")];
+            tensor<int32, [4]> var_118 = const()[name = string("op_118"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_119_cast_fp16 = reshape(shape = var_118, x = linear_2_cast_fp16)[name = string("op_119_cast_fp16")];
+            tensor<int32, [4]> transpose_24_perm_0 = const()[name = string("transpose_24_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [4]> transpose_25_perm_0 = const()[name = string("transpose_25_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [4]> transpose_26_perm_0 = const()[name = string("transpose_26_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_26 = transpose(perm = transpose_26_perm_0, x = var_119_cast_fp16)[name = string("transpose_49")];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_25 = transpose(perm = transpose_25_perm_0, x = var_113_cast_fp16)[name = string("transpose_50")];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_24 = transpose(perm = transpose_24_perm_0, x = var_107_cast_fp16)[name = string("transpose_51")];
+            tensor<fp16, [1, 6, 1500, 64]> a_1_cast_fp16 = scaled_dot_product_attention(key = transpose_25, query = transpose_24, value = transpose_26)[name = string("a_1_cast_fp16")];
+            tensor<int32, [4]> var_123 = const()[name = string("op_123"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [3]> concat_0 = const()[name = string("concat_0"), val = tensor<int32, [3]>([1, 1500, 384])];
+            tensor<fp16, [1, 1500, 6, 64]> var_124_cast_fp16 = transpose(perm = var_123, x = a_1_cast_fp16)[name = string("transpose_48")];
+            tensor<fp16, [1, 1500, 384]> x_11_cast_fp16 = reshape(shape = concat_0, x = var_124_cast_fp16)[name = string("x_11_cast_fp16")];
+            tensor<fp16, [384, 384]> const_15_to_fp16 = const()[name = string("const_15_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3112064)))];
+            tensor<fp16, [384]> const_16_to_fp16 = const()[name = string("const_16_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3407040)))];
+            tensor<fp16, [1, 1500, 384]> linear_3_cast_fp16 = linear(bias = const_16_to_fp16, weight = const_15_to_fp16, x = x_11_cast_fp16)[name = string("linear_3_cast_fp16")];
+            tensor<fp16, [1, 1500, 384]> x_13_cast_fp16 = add(x = var_55_cast_fp16, y = linear_3_cast_fp16)[name = string("x_13_cast_fp16")];
+            tensor<int32, [1]> var_136_axes_0 = const()[name = string("op_136_axes_0"), val = tensor<int32, [1]>([-1])];
+            tensor<fp16, [384]> blocks_0_mlp_ln_weight_to_fp16 = const()[name = string("blocks_0_mlp_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3407872)))];
+            tensor<fp16, [384]> blocks_0_mlp_ln_bias_to_fp16 = const()[name = string("blocks_0_mlp_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3408704)))];
+            tensor<fp16, [1, 1500, 384]> var_136_cast_fp16 = layer_norm(axes = var_136_axes_0, beta = blocks_0_mlp_ln_bias_to_fp16, epsilon = var_72_to_fp16, gamma = blocks_0_mlp_ln_weight_to_fp16, x = x_13_cast_fp16)[name = string("op_136_cast_fp16")];
+            tensor<fp16, [1536, 384]> const_17_to_fp16 = const()[name = string("const_17_to_fp16"), val = tensor<fp16, [1536, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3409536)))];
+            tensor<fp16, [1536]> const_18_to_fp16 = const()[name = string("const_18_to_fp16"), val = tensor<fp16, [1536]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4589248)))];
+            tensor<fp16, [1, 1500, 1536]> linear_4_cast_fp16 = linear(bias = const_18_to_fp16, weight = const_17_to_fp16, x = var_136_cast_fp16)[name = string("linear_4_cast_fp16")];
+            string x_17_mode_0 = const()[name = string("x_17_mode_0"), val = string("EXACT")];
+            tensor<fp16, [1, 1500, 1536]> x_17_cast_fp16 = gelu(mode = x_17_mode_0, x = linear_4_cast_fp16)[name = string("x_17_cast_fp16")];
+            tensor<fp16, [384, 1536]> const_19_to_fp16 = const()[name = string("const_19_to_fp16"), val = tensor<fp16, [384, 1536]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4592384)))];
+            tensor<fp16, [384]> const_20_to_fp16 = const()[name = string("const_20_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5772096)))];
+            tensor<fp16, [1, 1500, 384]> linear_5_cast_fp16 = linear(bias = const_20_to_fp16, weight = const_19_to_fp16, x = x_17_cast_fp16)[name = string("linear_5_cast_fp16")];
+            tensor<fp16, [1, 1500, 384]> x_19_cast_fp16 = add(x = x_13_cast_fp16, y = linear_5_cast_fp16)[name = string("x_19_cast_fp16")];
+            tensor<int32, [1]> var_176_axes_0 = const()[name = string("op_176_axes_0"), val = tensor<int32, [1]>([-1])];
+            tensor<fp16, [384]> blocks_1_attn_ln_weight_to_fp16 = const()[name = string("blocks_1_attn_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5772928)))];
+            tensor<fp16, [384]> blocks_1_attn_ln_bias_to_fp16 = const()[name = string("blocks_1_attn_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5773760)))];
+            fp16 var_166_to_fp16 = const()[name = string("op_166_to_fp16"), val = fp16(0x1.5p-17)];
+            tensor<fp16, [1, 1500, 384]> var_176_cast_fp16 = layer_norm(axes = var_176_axes_0, beta = blocks_1_attn_ln_bias_to_fp16, epsilon = var_166_to_fp16, gamma = blocks_1_attn_ln_weight_to_fp16, x = x_19_cast_fp16)[name = string("op_176_cast_fp16")];
+            tensor<fp16, [384, 384]> const_21_to_fp16 = const()[name = string("const_21_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5774592)))];
+            tensor<fp16, [384]> const_22_to_fp16 = const()[name = string("const_22_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6069568)))];
+            tensor<fp16, [1, 1500, 384]> linear_6_cast_fp16 = linear(bias = const_22_to_fp16, weight = const_21_to_fp16, x = var_176_cast_fp16)[name = string("linear_6_cast_fp16")];
+            tensor<fp16, [384, 384]> const_23_to_fp16 = const()[name = string("const_23_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6070400)))];
+            tensor<fp16, [1, 1500, 384]> linear_7_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = const_23_to_fp16, x = var_176_cast_fp16)[name = string("linear_7_cast_fp16")];
+            tensor<fp16, [384, 384]> const_24_to_fp16 = const()[name = string("const_24_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6365376)))];
+            tensor<fp16, [384]> const_25_to_fp16 = const()[name = string("const_25_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6660352)))];
+            tensor<fp16, [1, 1500, 384]> linear_8_cast_fp16 = linear(bias = const_25_to_fp16, weight = const_24_to_fp16, x = var_176_cast_fp16)[name = string("linear_8_cast_fp16")];
+            tensor<int32, [4]> var_200 = const()[name = string("op_200"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_201_cast_fp16 = reshape(shape = var_200, x = linear_6_cast_fp16)[name = string("op_201_cast_fp16")];
+            tensor<int32, [4]> var_206 = const()[name = string("op_206"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_207_cast_fp16 = reshape(shape = var_206, x = linear_7_cast_fp16)[name = string("op_207_cast_fp16")];
+            tensor<int32, [4]> var_212 = const()[name = string("op_212"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_213_cast_fp16 = reshape(shape = var_212, x = linear_8_cast_fp16)[name = string("op_213_cast_fp16")];
+            tensor<int32, [4]> transpose_27_perm_0 = const()[name = string("transpose_27_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [4]> transpose_28_perm_0 = const()[name = string("transpose_28_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [4]> transpose_29_perm_0 = const()[name = string("transpose_29_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_29 = transpose(perm = transpose_29_perm_0, x = var_213_cast_fp16)[name = string("transpose_45")];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_28 = transpose(perm = transpose_28_perm_0, x = var_207_cast_fp16)[name = string("transpose_46")];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_27 = transpose(perm = transpose_27_perm_0, x = var_201_cast_fp16)[name = string("transpose_47")];
+            tensor<fp16, [1, 6, 1500, 64]> a_3_cast_fp16 = scaled_dot_product_attention(key = transpose_28, query = transpose_27, value = transpose_29)[name = string("a_3_cast_fp16")];
+            tensor<int32, [4]> var_217 = const()[name = string("op_217"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [3]> concat_1 = const()[name = string("concat_1"), val = tensor<int32, [3]>([1, 1500, 384])];
+            tensor<fp16, [1, 1500, 6, 64]> var_218_cast_fp16 = transpose(perm = var_217, x = a_3_cast_fp16)[name = string("transpose_44")];
+            tensor<fp16, [1, 1500, 384]> x_23_cast_fp16 = reshape(shape = concat_1, x = var_218_cast_fp16)[name = string("x_23_cast_fp16")];
+            tensor<fp16, [384, 384]> const_32_to_fp16 = const()[name = string("const_32_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6661184)))];
+            tensor<fp16, [384]> const_33_to_fp16 = const()[name = string("const_33_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6956160)))];
+            tensor<fp16, [1, 1500, 384]> linear_9_cast_fp16 = linear(bias = const_33_to_fp16, weight = const_32_to_fp16, x = x_23_cast_fp16)[name = string("linear_9_cast_fp16")];
+            tensor<fp16, [1, 1500, 384]> x_25_cast_fp16 = add(x = x_19_cast_fp16, y = linear_9_cast_fp16)[name = string("x_25_cast_fp16")];
+            tensor<int32, [1]> var_230_axes_0 = const()[name = string("op_230_axes_0"), val = tensor<int32, [1]>([-1])];
+            tensor<fp16, [384]> blocks_1_mlp_ln_weight_to_fp16 = const()[name = string("blocks_1_mlp_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6956992)))];
+            tensor<fp16, [384]> blocks_1_mlp_ln_bias_to_fp16 = const()[name = string("blocks_1_mlp_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6957824)))];
+            tensor<fp16, [1, 1500, 384]> var_230_cast_fp16 = layer_norm(axes = var_230_axes_0, beta = blocks_1_mlp_ln_bias_to_fp16, epsilon = var_166_to_fp16, gamma = blocks_1_mlp_ln_weight_to_fp16, x = x_25_cast_fp16)[name = string("op_230_cast_fp16")];
+            tensor<fp16, [1536, 384]> const_34_to_fp16 = const()[name = string("const_34_to_fp16"), val = tensor<fp16, [1536, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6958656)))];
+            tensor<fp16, [1536]> const_35_to_fp16 = const()[name = string("const_35_to_fp16"), val = tensor<fp16, [1536]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8138368)))];
+            tensor<fp16, [1, 1500, 1536]> linear_10_cast_fp16 = linear(bias = const_35_to_fp16, weight = const_34_to_fp16, x = var_230_cast_fp16)[name = string("linear_10_cast_fp16")];
+            string x_29_mode_0 = const()[name = string("x_29_mode_0"), val = string("EXACT")];
+            tensor<fp16, [1, 1500, 1536]> x_29_cast_fp16 = gelu(mode = x_29_mode_0, x = linear_10_cast_fp16)[name = string("x_29_cast_fp16")];
+            tensor<fp16, [384, 1536]> const_36_to_fp16 = const()[name = string("const_36_to_fp16"), val = tensor<fp16, [384, 1536]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8141504)))];
+            tensor<fp16, [384]> const_37_to_fp16 = const()[name = string("const_37_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9321216)))];
+            tensor<fp16, [1, 1500, 384]> linear_11_cast_fp16 = linear(bias = const_37_to_fp16, weight = const_36_to_fp16, x = x_29_cast_fp16)[name = string("linear_11_cast_fp16")];
+            tensor<fp16, [1, 1500, 384]> x_31_cast_fp16 = add(x = x_25_cast_fp16, y = linear_11_cast_fp16)[name = string("x_31_cast_fp16")];
+            tensor<int32, [1]> var_270_axes_0 = const()[name = string("op_270_axes_0"), val = tensor<int32, [1]>([-1])];
+            tensor<fp16, [384]> blocks_2_attn_ln_weight_to_fp16 = const()[name = string("blocks_2_attn_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9322048)))];
+            tensor<fp16, [384]> blocks_2_attn_ln_bias_to_fp16 = const()[name = string("blocks_2_attn_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9322880)))];
+            fp16 var_260_to_fp16 = const()[name = string("op_260_to_fp16"), val = fp16(0x1.5p-17)];
+            tensor<fp16, [1, 1500, 384]> var_270_cast_fp16 = layer_norm(axes = var_270_axes_0, beta = blocks_2_attn_ln_bias_to_fp16, epsilon = var_260_to_fp16, gamma = blocks_2_attn_ln_weight_to_fp16, x = x_31_cast_fp16)[name = string("op_270_cast_fp16")];
+            tensor<fp16, [384, 384]> const_38_to_fp16 = const()[name = string("const_38_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9323712)))];
+            tensor<fp16, [384]> const_39_to_fp16 = const()[name = string("const_39_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9618688)))];
+            tensor<fp16, [1, 1500, 384]> linear_12_cast_fp16 = linear(bias = const_39_to_fp16, weight = const_38_to_fp16, x = var_270_cast_fp16)[name = string("linear_12_cast_fp16")];
+            tensor<fp16, [384, 384]> const_40_to_fp16 = const()[name = string("const_40_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9619520)))];
+            tensor<fp16, [1, 1500, 384]> linear_13_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = const_40_to_fp16, x = var_270_cast_fp16)[name = string("linear_13_cast_fp16")];
+            tensor<fp16, [384, 384]> const_41_to_fp16 = const()[name = string("const_41_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9914496)))];
+            tensor<fp16, [384]> const_42_to_fp16 = const()[name = string("const_42_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10209472)))];
+            tensor<fp16, [1, 1500, 384]> linear_14_cast_fp16 = linear(bias = const_42_to_fp16, weight = const_41_to_fp16, x = var_270_cast_fp16)[name = string("linear_14_cast_fp16")];
+            tensor<int32, [4]> var_294 = const()[name = string("op_294"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_295_cast_fp16 = reshape(shape = var_294, x = linear_12_cast_fp16)[name = string("op_295_cast_fp16")];
+            tensor<int32, [4]> var_300 = const()[name = string("op_300"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_301_cast_fp16 = reshape(shape = var_300, x = linear_13_cast_fp16)[name = string("op_301_cast_fp16")];
+            tensor<int32, [4]> var_306 = const()[name = string("op_306"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_307_cast_fp16 = reshape(shape = var_306, x = linear_14_cast_fp16)[name = string("op_307_cast_fp16")];
+            tensor<int32, [4]> transpose_30_perm_0 = const()[name = string("transpose_30_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [4]> transpose_31_perm_0 = const()[name = string("transpose_31_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [4]> transpose_32_perm_0 = const()[name = string("transpose_32_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_32 = transpose(perm = transpose_32_perm_0, x = var_307_cast_fp16)[name = string("transpose_41")];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_31 = transpose(perm = transpose_31_perm_0, x = var_301_cast_fp16)[name = string("transpose_42")];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_30 = transpose(perm = transpose_30_perm_0, x = var_295_cast_fp16)[name = string("transpose_43")];
+            tensor<fp16, [1, 6, 1500, 64]> a_5_cast_fp16 = scaled_dot_product_attention(key = transpose_31, query = transpose_30, value = transpose_32)[name = string("a_5_cast_fp16")];
+            tensor<int32, [4]> var_311 = const()[name = string("op_311"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [3]> concat_2 = const()[name = string("concat_2"), val = tensor<int32, [3]>([1, 1500, 384])];
+            tensor<fp16, [1, 1500, 6, 64]> var_312_cast_fp16 = transpose(perm = var_311, x = a_5_cast_fp16)[name = string("transpose_40")];
+            tensor<fp16, [1, 1500, 384]> x_35_cast_fp16 = reshape(shape = concat_2, x = var_312_cast_fp16)[name = string("x_35_cast_fp16")];
+            tensor<fp16, [384, 384]> const_49_to_fp16 = const()[name = string("const_49_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10210304)))];
+            tensor<fp16, [384]> const_50_to_fp16 = const()[name = string("const_50_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10505280)))];
+            tensor<fp16, [1, 1500, 384]> linear_15_cast_fp16 = linear(bias = const_50_to_fp16, weight = const_49_to_fp16, x = x_35_cast_fp16)[name = string("linear_15_cast_fp16")];
+            tensor<fp16, [1, 1500, 384]> x_37_cast_fp16 = add(x = x_31_cast_fp16, y = linear_15_cast_fp16)[name = string("x_37_cast_fp16")];
+            tensor<int32, [1]> var_324_axes_0 = const()[name = string("op_324_axes_0"), val = tensor<int32, [1]>([-1])];
+            tensor<fp16, [384]> blocks_2_mlp_ln_weight_to_fp16 = const()[name = string("blocks_2_mlp_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10506112)))];
+            tensor<fp16, [384]> blocks_2_mlp_ln_bias_to_fp16 = const()[name = string("blocks_2_mlp_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10506944)))];
+            tensor<fp16, [1, 1500, 384]> var_324_cast_fp16 = layer_norm(axes = var_324_axes_0, beta = blocks_2_mlp_ln_bias_to_fp16, epsilon = var_260_to_fp16, gamma = blocks_2_mlp_ln_weight_to_fp16, x = x_37_cast_fp16)[name = string("op_324_cast_fp16")];
+            tensor<fp16, [1536, 384]> const_51_to_fp16 = const()[name = string("const_51_to_fp16"), val = tensor<fp16, [1536, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10507776)))];
+            tensor<fp16, [1536]> const_52_to_fp16 = const()[name = string("const_52_to_fp16"), val = tensor<fp16, [1536]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11687488)))];
+            tensor<fp16, [1, 1500, 1536]> linear_16_cast_fp16 = linear(bias = const_52_to_fp16, weight = const_51_to_fp16, x = var_324_cast_fp16)[name = string("linear_16_cast_fp16")];
+            string x_41_mode_0 = const()[name = string("x_41_mode_0"), val = string("EXACT")];
+            tensor<fp16, [1, 1500, 1536]> x_41_cast_fp16 = gelu(mode = x_41_mode_0, x = linear_16_cast_fp16)[name = string("x_41_cast_fp16")];
+            tensor<fp16, [384, 1536]> const_53_to_fp16 = const()[name = string("const_53_to_fp16"), val = tensor<fp16, [384, 1536]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11690624)))];
+            tensor<fp16, [384]> const_54_to_fp16 = const()[name = string("const_54_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12870336)))];
+            tensor<fp16, [1, 1500, 384]> linear_17_cast_fp16 = linear(bias = const_54_to_fp16, weight = const_53_to_fp16, x = x_41_cast_fp16)[name = string("linear_17_cast_fp16")];
+            tensor<fp16, [1, 1500, 384]> x_43_cast_fp16 = add(x = x_37_cast_fp16, y = linear_17_cast_fp16)[name = string("x_43_cast_fp16")];
+            tensor<int32, [1]> var_364_axes_0 = const()[name = string("op_364_axes_0"), val = tensor<int32, [1]>([-1])];
+            tensor<fp16, [384]> blocks_3_attn_ln_weight_to_fp16 = const()[name = string("blocks_3_attn_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12871168)))];
+            tensor<fp16, [384]> blocks_3_attn_ln_bias_to_fp16 = const()[name = string("blocks_3_attn_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12872000)))];
+            fp16 var_354_to_fp16 = const()[name = string("op_354_to_fp16"), val = fp16(0x1.5p-17)];
+            tensor<fp16, [1, 1500, 384]> var_364_cast_fp16 = layer_norm(axes = var_364_axes_0, beta = blocks_3_attn_ln_bias_to_fp16, epsilon = var_354_to_fp16, gamma = blocks_3_attn_ln_weight_to_fp16, x = x_43_cast_fp16)[name = string("op_364_cast_fp16")];
+            tensor<fp16, [384, 384]> const_55_to_fp16 = const()[name = string("const_55_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12872832)))];
+            tensor<fp16, [384]> const_56_to_fp16 = const()[name = string("const_56_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13167808)))];
+            tensor<fp16, [1, 1500, 384]> linear_18_cast_fp16 = linear(bias = const_56_to_fp16, weight = const_55_to_fp16, x = var_364_cast_fp16)[name = string("linear_18_cast_fp16")];
+            tensor<fp16, [384, 384]> const_57_to_fp16 = const()[name = string("const_57_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13168640)))];
+            tensor<fp16, [1, 1500, 384]> linear_19_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = const_57_to_fp16, x = var_364_cast_fp16)[name = string("linear_19_cast_fp16")];
+            tensor<fp16, [384, 384]> const_58_to_fp16 = const()[name = string("const_58_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13463616)))];
+            tensor<fp16, [384]> const_59_to_fp16 = const()[name = string("const_59_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13758592)))];
+            tensor<fp16, [1, 1500, 384]> linear_20_cast_fp16 = linear(bias = const_59_to_fp16, weight = const_58_to_fp16, x = var_364_cast_fp16)[name = string("linear_20_cast_fp16")];
+            tensor<int32, [4]> var_388 = const()[name = string("op_388"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_389_cast_fp16 = reshape(shape = var_388, x = linear_18_cast_fp16)[name = string("op_389_cast_fp16")];
+            tensor<int32, [4]> var_394 = const()[name = string("op_394"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_395_cast_fp16 = reshape(shape = var_394, x = linear_19_cast_fp16)[name = string("op_395_cast_fp16")];
+            tensor<int32, [4]> var_400 = const()[name = string("op_400"), val = tensor<int32, [4]>([1, 1500, 6, -1])];
+            tensor<fp16, [1, 1500, 6, 64]> var_401_cast_fp16 = reshape(shape = var_400, x = linear_20_cast_fp16)[name = string("op_401_cast_fp16")];
+            tensor<int32, [4]> transpose_33_perm_0 = const()[name = string("transpose_33_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [4]> transpose_34_perm_0 = const()[name = string("transpose_34_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [4]> transpose_35_perm_0 = const()[name = string("transpose_35_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_35 = transpose(perm = transpose_35_perm_0, x = var_401_cast_fp16)[name = string("transpose_37")];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_34 = transpose(perm = transpose_34_perm_0, x = var_395_cast_fp16)[name = string("transpose_38")];
+            tensor<fp16, [1, 6, 1500, 64]> transpose_33 = transpose(perm = transpose_33_perm_0, x = var_389_cast_fp16)[name = string("transpose_39")];
+            tensor<fp16, [1, 6, 1500, 64]> a_cast_fp16 = scaled_dot_product_attention(key = transpose_34, query = transpose_33, value = transpose_35)[name = string("a_cast_fp16")];
+            tensor<int32, [4]> var_405 = const()[name = string("op_405"), val = tensor<int32, [4]>([0, 2, 1, 3])];
+            tensor<int32, [3]> concat_3 = const()[name = string("concat_3"), val = tensor<int32, [3]>([1, 1500, 384])];
+            tensor<fp16, [1, 1500, 6, 64]> var_406_cast_fp16 = transpose(perm = var_405, x = a_cast_fp16)[name = string("transpose_36")];
+            tensor<fp16, [1, 1500, 384]> x_47_cast_fp16 = reshape(shape = concat_3, x = var_406_cast_fp16)[name = string("x_47_cast_fp16")];
+            tensor<fp16, [384, 384]> const_66_to_fp16 = const()[name = string("const_66_to_fp16"), val = tensor<fp16, [384, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13759424)))];
+            tensor<fp16, [384]> const_67_to_fp16 = const()[name = string("const_67_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14054400)))];
+            tensor<fp16, [1, 1500, 384]> linear_21_cast_fp16 = linear(bias = const_67_to_fp16, weight = const_66_to_fp16, x = x_47_cast_fp16)[name = string("linear_21_cast_fp16")];
+            tensor<fp16, [1, 1500, 384]> x_49_cast_fp16 = add(x = x_43_cast_fp16, y = linear_21_cast_fp16)[name = string("x_49_cast_fp16")];
+            tensor<int32, [1]> var_418_axes_0 = const()[name = string("op_418_axes_0"), val = tensor<int32, [1]>([-1])];
+            tensor<fp16, [384]> blocks_3_mlp_ln_weight_to_fp16 = const()[name = string("blocks_3_mlp_ln_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14055232)))];
+            tensor<fp16, [384]> blocks_3_mlp_ln_bias_to_fp16 = const()[name = string("blocks_3_mlp_ln_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14056064)))];
+            tensor<fp16, [1, 1500, 384]> var_418_cast_fp16 = layer_norm(axes = var_418_axes_0, beta = blocks_3_mlp_ln_bias_to_fp16, epsilon = var_354_to_fp16, gamma = blocks_3_mlp_ln_weight_to_fp16, x = x_49_cast_fp16)[name = string("op_418_cast_fp16")];
+            tensor<fp16, [1536, 384]> const_68_to_fp16 = const()[name = string("const_68_to_fp16"), val = tensor<fp16, [1536, 384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14056896)))];
+            tensor<fp16, [1536]> const_69_to_fp16 = const()[name = string("const_69_to_fp16"), val = tensor<fp16, [1536]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15236608)))];
+            tensor<fp16, [1, 1500, 1536]> linear_22_cast_fp16 = linear(bias = const_69_to_fp16, weight = const_68_to_fp16, x = var_418_cast_fp16)[name = string("linear_22_cast_fp16")];
+            string x_53_mode_0 = const()[name = string("x_53_mode_0"), val = string("EXACT")];
+            tensor<fp16, [1, 1500, 1536]> x_53_cast_fp16 = gelu(mode = x_53_mode_0, x = linear_22_cast_fp16)[name = string("x_53_cast_fp16")];
+            tensor<fp16, [384, 1536]> const_70_to_fp16 = const()[name = string("const_70_to_fp16"), val = tensor<fp16, [384, 1536]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15239744)))];
+            tensor<fp16, [384]> const_71_to_fp16 = const()[name = string("const_71_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(16419456)))];
+            tensor<fp16, [1, 1500, 384]> linear_23_cast_fp16 = linear(bias = const_71_to_fp16, weight = const_70_to_fp16, x = x_53_cast_fp16)[name = string("linear_23_cast_fp16")];
+            tensor<fp16, [1, 1500, 384]> x_cast_fp16 = add(x = x_49_cast_fp16, y = linear_23_cast_fp16)[name = string("x_cast_fp16")];
+            tensor<int32, [1]> var_447_axes_0 = const()[name = string("op_447_axes_0"), val = tensor<int32, [1]>([-1])];
+            tensor<fp16, [384]> ln_post_weight_to_fp16 = const()[name = string("ln_post_weight_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(16420288)))];
+            tensor<fp16, [384]> ln_post_bias_to_fp16 = const()[name = string("ln_post_bias_to_fp16"), val = tensor<fp16, [384]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(16421120)))];
+            fp16 var_438_to_fp16 = const()[name = string("op_438_to_fp16"), val = fp16(0x1.5p-17)];
+            tensor<fp16, [1, 1500, 384]> output = layer_norm(axes = var_447_axes_0, beta = ln_post_bias_to_fp16, epsilon = var_438_to_fp16, gamma = ln_post_weight_to_fp16, x = x_cast_fp16)[name = string("op_447_cast_fp16")];
         } -> (output);
 }

ggml-tiny-encoder.mlmodelc/weights/weight.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:cb82c7078e49e07a517e598cce7e3b6dded7397efc495be1992ef822570284a4
 size 16421952

 version https://git-lfs.github.com/spec/v1
+oid sha256:4efa9bb81afaf12ac6d7cf7a3a4ba1e6b92f05f96ae77fd55cf725e2ecd3a5fd
 size 16421952

ggml-large-v2-encoder.mlmodelc/analytics/coremldata.bin → ggml-tiny-q8_0.bin RENAMED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:ad0b061176b791284f765ca1e72c9c7fed728ef6cf0af875a51c93fd1a8e525f
-size 243

 version https://git-lfs.github.com/spec/v1
+oid sha256:98bfdeb46504a115c4c0dcbaf14713b003152b2ce23f14f639d7f6d39dee043a
+size 32561100

ggml-tiny.bin DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:c7e4b084cefeebeed66fb9d096a29b836125edbb8456fea5a9c77b4efc085323
-size 60079860

index/base CHANGED Viewed

@@ -3,4 +3,4 @@ ggml-base-encoder.mlmodelc/metadata.json
 ggml-base-encoder.mlmodelc/model.mil
 ggml-base-encoder.mlmodelc/coremldata.bin
 ggml-base-encoder.mlmodelc/analytics/coremldata.bin
-ggml-base.bin

 ggml-base-encoder.mlmodelc/model.mil
 ggml-base-encoder.mlmodelc/coremldata.bin
 ggml-base-encoder.mlmodelc/analytics/coremldata.bin
+ggml-base-q8_0.bin

index/large-v2 DELETED Viewed

@@ -1,6 +0,0 @@
-ggml-large-v2-encoder.mlmodelc/weights/weight.bin
-ggml-large-v2-encoder.mlmodelc/metadata.json
-ggml-large-v2-encoder.mlmodelc/model.mil
-ggml-large-v2-encoder.mlmodelc/coremldata.bin
-ggml-large-v2-encoder.mlmodelc/analytics/coremldata.bin
-ggml-large-v2-q8_0.bin

index/large-v3-turbo CHANGED Viewed

@@ -1,12 +1,6 @@
 ggml-large-v3-turbo-encoder.mlmodelc/metadata.json
-ggml-large-v3-turbo-encoder.mlmodelc/model0/weights/0-weight.bin
-ggml-large-v3-turbo-encoder.mlmodelc/model0/model.mil
-ggml-large-v3-turbo-encoder.mlmodelc/model0/coremldata.bin
-ggml-large-v3-turbo-encoder.mlmodelc/model0/analytics/coremldata.bin
-ggml-large-v3-turbo-encoder.mlmodelc/model1/weights/1-weight.bin
-ggml-large-v3-turbo-encoder.mlmodelc/model1/model.mil
-ggml-large-v3-turbo-encoder.mlmodelc/model1/coremldata.bin
-ggml-large-v3-turbo-encoder.mlmodelc/model1/analytics/coremldata.bin
 ggml-large-v3-turbo-encoder.mlmodelc/coremldata.bin
 ggml-large-v3-turbo-encoder.mlmodelc/analytics/coremldata.bin
 ggml-large-v3-turbo-q8_0.bin

+ggml-large-v3-turbo-encoder.mlmodelc/weights/weight.bin
 ggml-large-v3-turbo-encoder.mlmodelc/metadata.json
+ggml-large-v3-turbo-encoder.mlmodelc/model.mil
 ggml-large-v3-turbo-encoder.mlmodelc/coremldata.bin
 ggml-large-v3-turbo-encoder.mlmodelc/analytics/coremldata.bin
 ggml-large-v3-turbo-q8_0.bin