{ "op": "ai.onnx.NonZero", "tunableSpace": {}, "cases": [ { "name": "f32_1024_capacity", "inputs": { "x": { "dtype": "float32", "shape": [1024], "data": { "kind": "fillFloat32", "sinStep": 0.01, "cosStep": 0.02 } } }, "outputs": { "y": { "dtype": "uint32", "shape": [1, 1024] } } }, { "name": "nonzero-rank2-u32-1m", "preset": "smoke", "inputs": { "x": { "dtype": "uint32", "shape": [1024, 1024], "dist": "linearMod", "mod": 4 } }, "outputs": { "y": { "dtype": "uint32", "shape": [2, 786432] } }, "bench": { "primary": true, "metrics": [ { "type": "bandwidth", "value": "4 * numel(shapes.x) + 4 * numel(shapes.y) + 4 * (4 * ceilDiv(numel(shapes.x), 4) + 5 * ceilDiv(numel(shapes.x), 1024))" } ] } }, { "name": "nonzero-exact-capacity-1024-f32", "preset": "smoke", "provenance": { "notes": "Legacy case name retained for longitudinal comparisons. At the measured crossover, 1024 elements now select the parallel scan. The traffic metric uses algorithm-independent logical I/O so an all-variant comparison weighs serial and parallel kernels fairly." }, "inputs": { "x": { "dtype": "float32", "shape": [1024], "dist": "constant", "value": 1 } }, "outputs": { "y": { "dtype": "uint32", "shape": [1, 1024] } }, "bench": { "metrics": [ { "type": "bandwidth", "value": "4 * numel(shapes.x) + 4 * numel(shapes.y) + 4 * (4 * ceilDiv(numel(shapes.x), 4) + 5 * ceilDiv(numel(shapes.x), 1024))" } ] } }, { "name": "nonzero-parallel-1280-f32-healthy", "preset": "smoke", "inputs": { "x": { "dtype": "float32", "shape": [1280], "dist": "constant", "value": 1 } }, "outputs": { "y": { "dtype": "uint32", "shape": [1, 1280] } }, "bench": { "metrics": [ { "type": "bandwidth", "value": "4 * numel(shapes.x) + 4 * numel(shapes.y) + 4 * (4 * ceilDiv(numel(shapes.x), 4) + 5 * ceilDiv(numel(shapes.x), 1024))" } ] } }, { "name": "nonzero-rank2-f32-1m-dense-100pct", "preset": "smoke", "inputs": { "x": { "dtype": "float32", "shape": [1024, 1024], "dist": "constant", "value": 1 } }, "outputs": { "y": { "dtype": "uint32", "shape": [2, 1048576] } }, "bench": { "metrics": [ { "type": "bandwidth", "value": "4 * numel(shapes.x) + 4 * numel(shapes.y) + 4 * (4 * ceilDiv(numel(shapes.x), 4) + 5 * ceilDiv(numel(shapes.x), 1024))" } ] } }, { "name": "nonzero-rank2-f16-1m-dense", "preset": "smoke", "inputs": { "x": { "dtype": "float16", "shape": [1024, 1024], "dist": "constant", "value": 1 } }, "outputs": { "y": { "dtype": "uint32", "shape": [2, 1048576] } }, "bench": { "metrics": [ { "type": "bandwidth", "value": "2 * numel(shapes.x) + 4 * numel(shapes.y) + 4 * (4 * ceilDiv(numel(shapes.x), 4) + 5 * ceilDiv(numel(shapes.x), 1024))" } ] } }, { "name": "nonzero-dispatch-cliff-16m-f32-sparse", "preset": "smoke", "inputs": { "x": { "dtype": "float32", "shape": [4096, 4096], "data": { "kind": "cycle", "values": [1.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0] } } }, "outputs": { "y": { "dtype": "uint32", "shape": [2, 1048576] } }, "bench": { "metrics": [ { "type": "bandwidth", "value": "4 * numel(shapes.x) + 4 * numel(shapes.y) + 4 * (4 * ceilDiv(numel(shapes.x), 4) + 5 * ceilDiv(numel(shapes.x), 1024))" } ] } }, { "name": "nonzero-2dfold-boundary-16m-int32-exact", "preset": "stress", "inputs": { "x": { "dtype": "int32", "shape": [4096, 4097], "dist": "normal", "seed": 7, "scale": 2 } }, "outputs": { "y": { "dtype": "uint32", "shape": [2, 16781312] } }, "bench": { "metrics": [ { "type": "bandwidth", "value": "4 * numel(shapes.x) + 4 * numel(shapes.y) + 4 * (4 * ceilDiv(numel(shapes.x), 4) + 5 * ceilDiv(numel(shapes.x), 1024))" } ] } }, { "name": "nonzero-rank4-u32-1x32x512x512-exact-activation", "preset": "stress", "inputs": { "x": { "dtype": "uint32", "shape": [1, 32, 512, 512], "dist": "linearMod", "mod": 16 } }, "outputs": { "y": { "dtype": "uint32", "shape": [4, 7864320] } }, "bench": { "metrics": [ { "type": "bandwidth", "value": "4 * numel(shapes.x) + 4 * numel(shapes.y) + 4 * (4 * ceilDiv(numel(shapes.x), 4) + 5 * ceilDiv(numel(shapes.x), 1024))" } ] } }, { "name": "nonzero-rank3-u32-32x512x512-exact-activation", "preset": "stress", "provenance": { "notes": "Rank-3 twin of the rank-4 activation case, same element count and density. The scatter writes one coordinate stream per rank, so it exists to place the items-per-thread threshold on measured rank-2/3/4 behaviour instead of extrapolating from two points." }, "inputs": { "x": { "dtype": "uint32", "shape": [32, 512, 512], "dist": "linearMod", "mod": 16 } }, "outputs": { "y": { "dtype": "uint32", "shape": [3, 7864320] } }, "bench": { "metrics": [ { "type": "bandwidth", "value": "4 * numel(shapes.x) + 4 * numel(shapes.y) + 4 * (4 * ceilDiv(numel(shapes.x), 4) + 5 * ceilDiv(numel(shapes.x), 1024))" } ] } } ] }