Add files using upload-large-folder tool
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/__grp__triton_red_fused_argmax_1.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.cubin +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.llir +206 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.ptx +490 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.source +323 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.ttgir +218 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.ttir +217 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/__grp__triton_red_fused__to_copy_clone_slice_sum_transpose_5.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.cubin +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.llir +204 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ptx +525 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.source +193 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttgir +147 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttir +152 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/__grp__triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.cubin +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.llir +667 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ptx +1534 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.source +299 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ttgir +232 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ttir +231 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/__grp__triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.llir +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ptx +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.source +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ttgir +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ttir +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/__grp__triton_red_fused__to_copy_clone_slice_sum_transpose_5.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.cubin +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.llir +205 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ptx +527 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.source +193 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttgir +147 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttir +152 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/__grp__triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.cubin +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.llir +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ptx +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.source +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ttgir +841 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ttir +799 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/__grp__triton_poi_fused_new_zeros_1.json +1 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/triton_poi_fused_new_zeros_1.cubin +0 -0
- SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/triton_poi_fused_new_zeros_1.json +1 -0
SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/__grp__triton_red_fused_argmax_1.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"child_paths": {"triton_red_fused_argmax_1.source": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.source", "triton_red_fused_argmax_1.ttir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.ttir", "triton_red_fused_argmax_1.ttgir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.ttgir", "triton_red_fused_argmax_1.llir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.llir", "triton_red_fused_argmax_1.ptx": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.ptx", "triton_red_fused_argmax_1.cubin": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.cubin", "triton_red_fused_argmax_1.json": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.json"}}
|
SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.cubin
ADDED
|
Binary file (14.6 kB). View file
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"hash": "d764c4de3a434d91651ee340a32edffa72e41bcb0125e69cee815d3536f2f3ce", "target": {"backend": "cuda", "arch": 90, "warp_size": 32}, "num_warps": 8, "num_ctas": 1, "num_stages": 1, "warp_size": 32, "maxnreg": null, "cluster_dims": [1, 1, 1], "ptx_version": null, "ptx_options": null, "ir_override": null, "enable_fp_fusion": true, "launch_cooperative_grid": false, "launch_pdl": false, "supported_fp8_dtypes": ["fp8e4b15", "fp8e4nv", "fp8e5"], "deprecated_fp8_dot_operand_dtypes": ["fp8e4b15"], "default_dot_input_precision": "tf32", "allowed_dot_input_precisions": ["tf32", "tf32x3", "ieee"], "max_num_imprecise_acc_default": 1073741824, "extern_libs": [["libdevice", "/workspace/specforge/lib/python3.11/site-packages/triton/backends/nvidia/lib/libdevice.10.bc"]], "debug": true, "backend_name": "cuda", "sanitize_overflow": false, "arch": "sm90", "instrumentation_mode": "", "triton_version": "3.5.1", "tensordesc_meta": [], "shared": 256, "tmem_size": 0, "global_scratch_size": 0, "global_scratch_align": 1, "profile_scratch_size": 0, "profile_scratch_align": 1, "name": "triton_red_fused_argmax_1"}
|
SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.llir
ADDED
|
@@ -0,0 +1,206 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
; ModuleID = 'LLVMDialectModule'
|
| 2 |
+
source_filename = "LLVMDialectModule"
|
| 3 |
+
target datalayout = "e-p3:32:32-p4:32:32-p5:32:32-p6:32:32-p7:32:32-i64:64-i128:128-v16:16-v32:32-n16:32:64"
|
| 4 |
+
|
| 5 |
+
@global_smem = external local_unnamed_addr addrspace(3) global [0 x i8], align 16
|
| 6 |
+
|
| 7 |
+
; Function Attrs: nounwind
|
| 8 |
+
define ptx_kernel void @triton_red_fused_argmax_1(ptr addrspace(1) %0, ptr addrspace(1) %1, i64 %2, i64 %3, i32 %4, i32 %5, ptr addrspace(1) readnone captures(none) %6, ptr addrspace(1) readnone captures(none) %7) local_unnamed_addr #0 !dbg !4 {
|
| 9 |
+
%9 = tail call i32 @llvm.nvvm.read.ptx.sreg.ctaid.x(), !dbg !7
|
| 10 |
+
%10 = shl i32 %9, 6, !dbg !8
|
| 11 |
+
%11 = tail call i32 @llvm.nvvm.read.ptx.sreg.tid.x(), !dbg !9
|
| 12 |
+
%12 = and i32 %11, 252, !dbg !9
|
| 13 |
+
%13 = lshr exact i32 %12, 2, !dbg !9
|
| 14 |
+
%14 = or disjoint i32 %13, %10, !dbg !10
|
| 15 |
+
%15 = icmp slt i32 %14, %4, !dbg !11
|
| 16 |
+
%16 = and i32 %11, 3, !dbg !12
|
| 17 |
+
%17 = sext i32 %14 to i64, !dbg !13
|
| 18 |
+
%.frozen = freeze i64 %2, !dbg !14
|
| 19 |
+
%18 = sdiv i64 %17, %.frozen, !dbg !14
|
| 20 |
+
%19 = mul i64 %18, %.frozen, !dbg !13
|
| 21 |
+
%.decomposed = sub i64 %17, %19, !dbg !13
|
| 22 |
+
%20 = mul i64 %18, %3, !dbg !15
|
| 23 |
+
%.idx = mul nsw i64 %.decomposed, 128000
|
| 24 |
+
%21 = getelementptr i8, ptr addrspace(1) %0, i64 %.idx
|
| 25 |
+
%invariant.gep = getelementptr float, ptr addrspace(1) %21, i64 %20, !dbg !16
|
| 26 |
+
%.fr = freeze i1 %15
|
| 27 |
+
%22 = zext nneg i32 %16 to i64, !dbg !16
|
| 28 |
+
br i1 %.fr, label %.split.us, label %.split.preheader
|
| 29 |
+
|
| 30 |
+
.split.preheader: ; preds = %8
|
| 31 |
+
%invariant.gep11 = getelementptr float, ptr addrspace(1) %invariant.gep, i64 %22, !dbg !16
|
| 32 |
+
br label %.split, !dbg !16
|
| 33 |
+
|
| 34 |
+
.split.us: ; preds = %8, %.split.us
|
| 35 |
+
%indvars.iv7 = phi i64 [ %indvars.iv.next8, %.split.us ], [ 0, %8 ]
|
| 36 |
+
%23 = phi i32 [ %44, %.split.us ], [ 2147483647, %8 ]
|
| 37 |
+
%24 = phi float [ %42, %.split.us ], [ 0xFFF0000000000000, %8 ]
|
| 38 |
+
%25 = or disjoint i64 %indvars.iv7, %22, !dbg !17
|
| 39 |
+
%gep.us = getelementptr float, ptr addrspace(1) %invariant.gep, i64 %25, !dbg !18
|
| 40 |
+
%26 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_first.b64 $0, 1.0;", "=l"() #4, !dbg !19
|
| 41 |
+
%27 = tail call i32 asm sideeffect "mov.u32 $0, $1;\0A\09@$4 ld.global.L1::evict_first.L2::cache_hint.b32 { $0 }, [ $2 + 0 ], $3;", "=r,r,l,l,b"(i32 0, ptr addrspace(1) %gep.us, i64 %26, i1 true) #4, !dbg !19
|
| 42 |
+
%28 = bitcast i32 %27 to float, !dbg !19
|
| 43 |
+
%29 = fcmp ogt float %24, %28, !dbg !20
|
| 44 |
+
%30 = fcmp oeq float %24, %28, !dbg !24
|
| 45 |
+
%31 = fcmp uno float %24, 0.000000e+00, !dbg !25
|
| 46 |
+
%32 = fcmp uno float %28, 0.000000e+00, !dbg !26
|
| 47 |
+
%33 = xor i1 %32, true, !dbg !27
|
| 48 |
+
%34 = and i1 %31, %33, !dbg !28
|
| 49 |
+
%35 = or i1 %29, %34, !dbg !29
|
| 50 |
+
%36 = and i1 %31, %32, !dbg !30
|
| 51 |
+
%37 = or i1 %30, %36, !dbg !31
|
| 52 |
+
%38 = sext i32 %23 to i64, !dbg !32
|
| 53 |
+
%39 = icmp sgt i64 %25, %38, !dbg !32
|
| 54 |
+
%40 = and i1 %39, %37, !dbg !33
|
| 55 |
+
%41 = or i1 %35, %40, !dbg !34
|
| 56 |
+
%42 = select i1 %41, float %24, float %28, !dbg !35
|
| 57 |
+
%43 = trunc nuw nsw i64 %25 to i32, !dbg !36
|
| 58 |
+
%44 = select i1 %41, i32 %23, i32 %43, !dbg !36
|
| 59 |
+
%indvars.iv.next8 = add nuw nsw i64 %indvars.iv7, 4, !dbg !16
|
| 60 |
+
%45 = icmp samesign ult i64 %indvars.iv7, 31996, !dbg !16
|
| 61 |
+
br i1 %45, label %.split.us, label %.split3.us, !dbg !16
|
| 62 |
+
|
| 63 |
+
.split: ; preds = %.split.preheader, %.split
|
| 64 |
+
%indvars.iv = phi i64 [ 0, %.split.preheader ], [ %indvars.iv.next, %.split ]
|
| 65 |
+
%gep12 = getelementptr float, ptr addrspace(1) %invariant.gep11, i64 %indvars.iv, !dbg !18
|
| 66 |
+
%46 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_first.b64 $0, 1.0;", "=l"() #4, !dbg !19
|
| 67 |
+
%47 = tail call i32 asm sideeffect "mov.u32 $0, $1;\0A\09@$4 ld.global.L1::evict_first.L2::cache_hint.b32 { $0 }, [ $2 + 0 ], $3;", "=r,r,l,l,b"(i32 0, ptr addrspace(1) %gep12, i64 %46, i1 false) #4, !dbg !19
|
| 68 |
+
%indvars.iv.next = add nuw nsw i64 %indvars.iv, 4, !dbg !16
|
| 69 |
+
%48 = icmp samesign ult i64 %indvars.iv, 31996, !dbg !16
|
| 70 |
+
br i1 %48, label %.split, label %.split3.us, !dbg !16
|
| 71 |
+
|
| 72 |
+
.split3.us: ; preds = %.split, %.split.us
|
| 73 |
+
%.us-phi = phi float [ %42, %.split.us ], [ 0xFFF0000000000000, %.split ], !dbg !9
|
| 74 |
+
%.us-phi4 = phi i32 [ %44, %.split.us ], [ 2147483647, %.split ], !dbg !9
|
| 75 |
+
%49 = and i32 %11, 63, !dbg !9
|
| 76 |
+
%50 = or disjoint i32 %10, %49, !dbg !10
|
| 77 |
+
%51 = icmp slt i32 %50, %4, !dbg !11
|
| 78 |
+
%52 = bitcast float %.us-phi to i32, !dbg !37
|
| 79 |
+
%53 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %52, i32 2, i32 31), !dbg !37
|
| 80 |
+
%54 = bitcast i32 %53 to float, !dbg !37
|
| 81 |
+
%55 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %.us-phi4, i32 2, i32 31), !dbg !37
|
| 82 |
+
%56 = fcmp ogt float %.us-phi, %54, !dbg !39
|
| 83 |
+
%57 = fcmp oeq float %.us-phi, %54, !dbg !40
|
| 84 |
+
%58 = fcmp uno float %.us-phi, 0.000000e+00, !dbg !41
|
| 85 |
+
%59 = fcmp uno float %54, 0.000000e+00, !dbg !42
|
| 86 |
+
%60 = xor i1 %59, true, !dbg !43
|
| 87 |
+
%61 = and i1 %58, %60, !dbg !44
|
| 88 |
+
%62 = or i1 %56, %61, !dbg !45
|
| 89 |
+
%63 = and i1 %58, %59, !dbg !46
|
| 90 |
+
%64 = or i1 %57, %63, !dbg !47
|
| 91 |
+
%65 = icmp slt i32 %.us-phi4, %55, !dbg !48
|
| 92 |
+
%66 = and i1 %65, %64, !dbg !49
|
| 93 |
+
%67 = or i1 %62, %66, !dbg !50
|
| 94 |
+
%68 = select i1 %67, float %.us-phi, float %54, !dbg !51
|
| 95 |
+
%69 = select i1 %67, i32 %.us-phi4, i32 %55, !dbg !52
|
| 96 |
+
%70 = bitcast float %68 to i32, !dbg !37
|
| 97 |
+
%71 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %70, i32 1, i32 31), !dbg !37
|
| 98 |
+
%72 = bitcast i32 %71 to float, !dbg !37
|
| 99 |
+
%73 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %69, i32 1, i32 31), !dbg !37
|
| 100 |
+
%74 = fcmp ogt float %68, %72, !dbg !39
|
| 101 |
+
%75 = fcmp oeq float %68, %72, !dbg !40
|
| 102 |
+
%76 = fcmp uno float %68, 0.000000e+00, !dbg !41
|
| 103 |
+
%77 = fcmp uno float %72, 0.000000e+00, !dbg !42
|
| 104 |
+
%78 = xor i1 %77, true, !dbg !43
|
| 105 |
+
%79 = and i1 %76, %78, !dbg !44
|
| 106 |
+
%80 = or i1 %74, %79, !dbg !45
|
| 107 |
+
%81 = and i1 %77, %76, !dbg !46
|
| 108 |
+
%82 = or i1 %75, %81, !dbg !47
|
| 109 |
+
%83 = icmp slt i32 %69, %73, !dbg !48
|
| 110 |
+
%84 = and i1 %83, %82, !dbg !49
|
| 111 |
+
%85 = or i1 %80, %84, !dbg !50
|
| 112 |
+
%86 = select i1 %85, i32 %69, i32 %73, !dbg !52
|
| 113 |
+
%87 = sext i32 %50 to i64, !dbg !53
|
| 114 |
+
%88 = getelementptr i64, ptr addrspace(1) %1, i64 %87, !dbg !53
|
| 115 |
+
%89 = getelementptr inbounds nuw i8, ptr addrspace(3) @global_smem, i32 %12, !dbg !54
|
| 116 |
+
%90 = insertelement <1 x i32> poison, i32 %86, i64 0, !dbg !54
|
| 117 |
+
store <1 x i32> %90, ptr addrspace(3) %89, align 4, !dbg !54
|
| 118 |
+
tail call void @llvm.nvvm.barrier.cta.sync.aligned.all(i32 0), !dbg !54
|
| 119 |
+
%91 = shl nuw nsw i32 %49, 2, !dbg !54
|
| 120 |
+
%92 = getelementptr inbounds nuw i8, ptr addrspace(3) @global_smem, i32 %91, !dbg !54
|
| 121 |
+
%93 = load i32, ptr addrspace(3) %92, align 4, !dbg !54
|
| 122 |
+
%94 = sext i32 %93 to i64, !dbg !54
|
| 123 |
+
%95 = and i32 %11, 192, !dbg !54
|
| 124 |
+
%96 = icmp eq i32 %95, 0, !dbg !54
|
| 125 |
+
%97 = and i1 %96, %51, !dbg !54
|
| 126 |
+
tail call void asm sideeffect "@$2 st.global.b64 [ $1 + 0 ], { $0 };", "l,l,b"(i64 %94, ptr addrspace(1) %88, i1 %97) #4, !dbg !54
|
| 127 |
+
ret void, !dbg !55
|
| 128 |
+
}
|
| 129 |
+
|
| 130 |
+
; Function Attrs: mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none)
|
| 131 |
+
declare noundef range(i32 0, 2147483647) i32 @llvm.nvvm.read.ptx.sreg.ctaid.x() #1
|
| 132 |
+
|
| 133 |
+
; Function Attrs: mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none)
|
| 134 |
+
declare noundef range(i32 0, 1024) i32 @llvm.nvvm.read.ptx.sreg.tid.x() #1
|
| 135 |
+
|
| 136 |
+
; Function Attrs: convergent nocallback nounwind memory(inaccessiblemem: readwrite)
|
| 137 |
+
declare i32 @llvm.nvvm.shfl.sync.bfly.i32(i32, i32, i32, i32) #2
|
| 138 |
+
|
| 139 |
+
; Function Attrs: convergent nocallback nounwind
|
| 140 |
+
declare void @llvm.nvvm.barrier.cta.sync.aligned.all(i32) #3
|
| 141 |
+
|
| 142 |
+
attributes #0 = { nounwind "nvvm.reqntid"="256" }
|
| 143 |
+
attributes #1 = { mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none) }
|
| 144 |
+
attributes #2 = { convergent nocallback nounwind memory(inaccessiblemem: readwrite) }
|
| 145 |
+
attributes #3 = { convergent nocallback nounwind }
|
| 146 |
+
attributes #4 = { nounwind }
|
| 147 |
+
|
| 148 |
+
!llvm.dbg.cu = !{!0}
|
| 149 |
+
!llvm.module.flags = !{!2, !3}
|
| 150 |
+
|
| 151 |
+
!0 = distinct !DICompileUnit(language: DW_LANG_C, file: !1, producer: "triton", isOptimized: true, runtimeVersion: 0, emissionKind: LineTablesOnly)
|
| 152 |
+
!1 = !DIFile(filename: "ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py", directory: "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu")
|
| 153 |
+
!2 = !{i32 2, !"Debug Info Version", i32 3}
|
| 154 |
+
!3 = !{i32 4, !"nvvm-reflect-ftz", i32 1}
|
| 155 |
+
!4 = distinct !DISubprogram(name: "triton_red_fused_argmax_1", linkageName: "triton_red_fused_argmax_1", scope: !1, file: !1, line: 18, type: !5, scopeLine: 18, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
|
| 156 |
+
!5 = !DISubroutineType(cc: DW_CC_normal, types: !6)
|
| 157 |
+
!6 = !{}
|
| 158 |
+
!7 = !DILocation(line: 22, column: 28, scope: !4)
|
| 159 |
+
!8 = !DILocation(line: 22, column: 33, scope: !4)
|
| 160 |
+
!9 = !DILocation(line: 23, column: 44, scope: !4)
|
| 161 |
+
!10 = !DILocation(line: 23, column: 23, scope: !4)
|
| 162 |
+
!11 = !DILocation(line: 24, column: 21, scope: !4)
|
| 163 |
+
!12 = !DILocation(line: 25, column: 37, scope: !4)
|
| 164 |
+
!13 = !DILocation(line: 27, column: 19, scope: !4)
|
| 165 |
+
!14 = !DILocation(line: 28, column: 19, scope: !4)
|
| 166 |
+
!15 = !DILocation(line: 38, column: 56, scope: !4)
|
| 167 |
+
!16 = !DILocation(line: 32, column: 40, scope: !4)
|
| 168 |
+
!17 = !DILocation(line: 33, column: 31, scope: !4)
|
| 169 |
+
!18 = !DILocation(line: 38, column: 34, scope: !4)
|
| 170 |
+
!19 = !DILocation(line: 38, column: 61, scope: !4)
|
| 171 |
+
!20 = !DILocation(line: 144, column: 21, scope: !21, inlinedAt: !23)
|
| 172 |
+
!21 = distinct !DILexicalBlockFile(scope: !4, file: !22, discriminator: 0)
|
| 173 |
+
!22 = !DIFile(filename: "triton_helpers.py", directory: "/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime")
|
| 174 |
+
!23 = !DILocation(line: 41, column: 38, scope: !4)
|
| 175 |
+
!24 = !DILocation(line: 145, column: 23, scope: !21, inlinedAt: !23)
|
| 176 |
+
!25 = !DILocation(line: 147, column: 29, scope: !21, inlinedAt: !23)
|
| 177 |
+
!26 = !DILocation(line: 148, column: 29, scope: !21, inlinedAt: !23)
|
| 178 |
+
!27 = !DILocation(line: 149, column: 31, scope: !21, inlinedAt: !23)
|
| 179 |
+
!28 = !DILocation(line: 149, column: 27, scope: !21, inlinedAt: !23)
|
| 180 |
+
!29 = !DILocation(line: 149, column: 16, scope: !21, inlinedAt: !23)
|
| 181 |
+
!30 = !DILocation(line: 151, column: 27, scope: !21, inlinedAt: !23)
|
| 182 |
+
!31 = !DILocation(line: 151, column: 17, scope: !21, inlinedAt: !23)
|
| 183 |
+
!32 = !DILocation(line: 154, column: 31, scope: !21, inlinedAt: !23)
|
| 184 |
+
!33 = !DILocation(line: 154, column: 21, scope: !21, inlinedAt: !23)
|
| 185 |
+
!34 = !DILocation(line: 154, column: 12, scope: !21, inlinedAt: !23)
|
| 186 |
+
!35 = !DILocation(line: 155, column: 35, scope: !21, inlinedAt: !23)
|
| 187 |
+
!36 = !DILocation(line: 155, column: 69, scope: !21, inlinedAt: !23)
|
| 188 |
+
!37 = !DILocation(line: 165, column: 42, scope: !21, inlinedAt: !38)
|
| 189 |
+
!38 = !DILocation(line: 45, column: 75, scope: !4)
|
| 190 |
+
!39 = !DILocation(line: 144, column: 21, scope: !21, inlinedAt: !38)
|
| 191 |
+
!40 = !DILocation(line: 145, column: 23, scope: !21, inlinedAt: !38)
|
| 192 |
+
!41 = !DILocation(line: 147, column: 29, scope: !21, inlinedAt: !38)
|
| 193 |
+
!42 = !DILocation(line: 148, column: 29, scope: !21, inlinedAt: !38)
|
| 194 |
+
!43 = !DILocation(line: 149, column: 31, scope: !21, inlinedAt: !38)
|
| 195 |
+
!44 = !DILocation(line: 149, column: 27, scope: !21, inlinedAt: !38)
|
| 196 |
+
!45 = !DILocation(line: 149, column: 16, scope: !21, inlinedAt: !38)
|
| 197 |
+
!46 = !DILocation(line: 151, column: 27, scope: !21, inlinedAt: !38)
|
| 198 |
+
!47 = !DILocation(line: 151, column: 17, scope: !21, inlinedAt: !38)
|
| 199 |
+
!48 = !DILocation(line: 154, column: 31, scope: !21, inlinedAt: !38)
|
| 200 |
+
!49 = !DILocation(line: 154, column: 21, scope: !21, inlinedAt: !38)
|
| 201 |
+
!50 = !DILocation(line: 154, column: 12, scope: !21, inlinedAt: !38)
|
| 202 |
+
!51 = !DILocation(line: 155, column: 35, scope: !21, inlinedAt: !38)
|
| 203 |
+
!52 = !DILocation(line: 155, column: 69, scope: !21, inlinedAt: !38)
|
| 204 |
+
!53 = !DILocation(line: 47, column: 25, scope: !4)
|
| 205 |
+
!54 = !DILocation(line: 47, column: 36, scope: !4)
|
| 206 |
+
!55 = !DILocation(line: 47, column: 4, scope: !4)
|
SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.ptx
ADDED
|
@@ -0,0 +1,490 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
//
|
| 2 |
+
// Generated by LLVM NVPTX Back-End
|
| 3 |
+
//
|
| 4 |
+
|
| 5 |
+
.version 8.7
|
| 6 |
+
.target sm_90a
|
| 7 |
+
.address_size 64
|
| 8 |
+
|
| 9 |
+
// .globl triton_red_fused_argmax_1 // -- Begin function triton_red_fused_argmax_1
|
| 10 |
+
.extern .shared .align 16 .b8 global_smem[];
|
| 11 |
+
// @triton_red_fused_argmax_1
|
| 12 |
+
.visible .entry triton_red_fused_argmax_1(
|
| 13 |
+
.param .u64 .ptr .global .align 1 triton_red_fused_argmax_1_param_0,
|
| 14 |
+
.param .u64 .ptr .global .align 1 triton_red_fused_argmax_1_param_1,
|
| 15 |
+
.param .u64 triton_red_fused_argmax_1_param_2,
|
| 16 |
+
.param .u64 triton_red_fused_argmax_1_param_3,
|
| 17 |
+
.param .u32 triton_red_fused_argmax_1_param_4,
|
| 18 |
+
.param .u32 triton_red_fused_argmax_1_param_5,
|
| 19 |
+
.param .u64 .ptr .global .align 1 triton_red_fused_argmax_1_param_6,
|
| 20 |
+
.param .u64 .ptr .global .align 1 triton_red_fused_argmax_1_param_7
|
| 21 |
+
)
|
| 22 |
+
.reqntid 256
|
| 23 |
+
{
|
| 24 |
+
.reg .pred %p<39>;
|
| 25 |
+
.reg .b32 %r<55>;
|
| 26 |
+
.reg .b64 %rd<54>;
|
| 27 |
+
.loc 1 18 0 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:18:0
|
| 28 |
+
$L__func_begin0:
|
| 29 |
+
.loc 1 18 0 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:18:0
|
| 30 |
+
|
| 31 |
+
// %bb.0:
|
| 32 |
+
ld.param.b32 %r12, [triton_red_fused_argmax_1_param_4];
|
| 33 |
+
$L__tmp0:
|
| 34 |
+
.loc 1 22 28 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:22:28
|
| 35 |
+
mov.u32 %r13, %ctaid.x;
|
| 36 |
+
.loc 1 22 33 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:22:33
|
| 37 |
+
shl.b32 %r1, %r13, 6;
|
| 38 |
+
ld.param.b64 %rd20, [triton_red_fused_argmax_1_param_2];
|
| 39 |
+
.loc 1 23 44 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:23:44
|
| 40 |
+
mov.u32 %r2, %tid.x;
|
| 41 |
+
bfe.u32 %r4, %r2, 2, 6;
|
| 42 |
+
.loc 1 23 23 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:23:23
|
| 43 |
+
or.b32 %r14, %r4, %r1;
|
| 44 |
+
.loc 1 25 37 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:25:37
|
| 45 |
+
and.b32 %r5, %r2, 3;
|
| 46 |
+
.loc 1 27 19 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:27:19
|
| 47 |
+
cvt.s64.s32 %rd1, %r14;
|
| 48 |
+
.loc 1 28 19 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:28:19
|
| 49 |
+
or.b64 %rd21, %rd1, %rd20;
|
| 50 |
+
and.b64 %rd22, %rd21, -4294967296;
|
| 51 |
+
setp.ne.b64 %p1, %rd22, 0;
|
| 52 |
+
cvt.u32.u64 %r50, %rd1;
|
| 53 |
+
@%p1 bra $L__BB0_2;
|
| 54 |
+
bra.uni $L__BB0_1;
|
| 55 |
+
$L__BB0_2:
|
| 56 |
+
div.s64 %rd49, %rd1, %rd20;
|
| 57 |
+
bra.uni $L__BB0_3;
|
| 58 |
+
$L__BB0_1:
|
| 59 |
+
cvt.u32.u64 %r15, %rd20;
|
| 60 |
+
div.u32 %r17, %r50, %r15;
|
| 61 |
+
cvt.u64.u32 %rd49, %r17;
|
| 62 |
+
$L__BB0_3:
|
| 63 |
+
.loc 1 0 19 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:0:19
|
| 64 |
+
ld.param.b64 %rd19, [triton_red_fused_argmax_1_param_3];
|
| 65 |
+
ld.param.b64 %rd18, [triton_red_fused_argmax_1_param_1];
|
| 66 |
+
ld.param.b64 %rd17, [triton_red_fused_argmax_1_param_0];
|
| 67 |
+
and.b32 %r3, %r2, 252;
|
| 68 |
+
.loc 1 32 40 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:32:40
|
| 69 |
+
cvt.u64.u32 %rd6, %r5;
|
| 70 |
+
setp.ge.s32 %p2, %r50, %r12;
|
| 71 |
+
@%p2 bra $L__BB0_6;
|
| 72 |
+
// %bb.4: // %.split.us.preheader
|
| 73 |
+
shl.b64 %rd35, %rd19, 2;
|
| 74 |
+
mul.lo.s64 %rd36, %rd20, 128000;
|
| 75 |
+
sub.s64 %rd37, %rd35, %rd36;
|
| 76 |
+
mul.lo.s64 %rd38, %rd49, %rd37;
|
| 77 |
+
add.s32 %r26, %r1, %r4;
|
| 78 |
+
mad.wide.s32 %rd39, %r26, 128000, %rd38;
|
| 79 |
+
shl.b64 %rd40, %rd6, 2;
|
| 80 |
+
add.s64 %rd41, %rd39, %rd40;
|
| 81 |
+
add.s64 %rd50, %rd17, %rd41;
|
| 82 |
+
mov.b32 %r53, 0fFF800000;
|
| 83 |
+
mov.b32 %r54, 2147483647;
|
| 84 |
+
mov.b64 %rd51, 0;
|
| 85 |
+
$L__BB0_5: // %.split.us
|
| 86 |
+
// =>This Inner Loop Header: Depth=1
|
| 87 |
+
.loc 1 38 34 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:38:34
|
| 88 |
+
add.s64 %rd45, %rd6, %rd51;
|
| 89 |
+
.loc 1 38 61 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:38:61
|
| 90 |
+
// begin inline asm
|
| 91 |
+
mov.u64 %rd42, 0x0;
|
| 92 |
+
createpolicy.fractional.L2::evict_first.b64 %rd42, 1.0;
|
| 93 |
+
// end inline asm
|
| 94 |
+
mov.b32 %r28, 0;
|
| 95 |
+
mov.pred %p5, -1;
|
| 96 |
+
// begin inline asm
|
| 97 |
+
mov.u32 %r27, %r28;
|
| 98 |
+
@%p5 ld.global.L1::evict_first.L2::cache_hint.b32 { %r27 }, [ %rd50 + 0 ], %rd42;
|
| 99 |
+
// end inline asm
|
| 100 |
+
$L__tmp1:
|
| 101 |
+
.loc 2 144 21 // triton_helpers.py:144:21 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 102 |
+
setp.gt.f32 %p6, %r53, %r27;
|
| 103 |
+
.loc 2 145 23 // triton_helpers.py:145:23 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 104 |
+
setp.eq.f32 %p7, %r53, %r27;
|
| 105 |
+
.loc 2 147 29 // triton_helpers.py:147:29 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 106 |
+
setp.nan.f32 %p8, %r53, %r53;
|
| 107 |
+
.loc 2 148 29 // triton_helpers.py:148:29 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 108 |
+
setp.nan.f32 %p9, %r27, %r27;
|
| 109 |
+
setp.num.f32 %p10, %r27, %r27;
|
| 110 |
+
.loc 2 149 27 // triton_helpers.py:149:27 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 111 |
+
and.pred %p11, %p8, %p10;
|
| 112 |
+
.loc 2 149 16 // triton_helpers.py:149:16 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 113 |
+
or.pred %p12, %p6, %p11;
|
| 114 |
+
.loc 2 151 27 // triton_helpers.py:151:27 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 115 |
+
and.pred %p13, %p8, %p9;
|
| 116 |
+
.loc 2 151 17 // triton_helpers.py:151:17 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 117 |
+
or.pred %p14, %p7, %p13;
|
| 118 |
+
.loc 2 154 31 // triton_helpers.py:154:31 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 119 |
+
cvt.s64.s32 %rd46, %r54;
|
| 120 |
+
setp.gt.s64 %p15, %rd45, %rd46;
|
| 121 |
+
.loc 2 154 21 // triton_helpers.py:154:21 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 122 |
+
and.pred %p16, %p15, %p14;
|
| 123 |
+
.loc 2 154 12 // triton_helpers.py:154:12 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 124 |
+
or.pred %p17, %p12, %p16;
|
| 125 |
+
.loc 2 155 35 // triton_helpers.py:155:35 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 126 |
+
selp.f32 %r53, %r53, %r27, %p17;
|
| 127 |
+
cvt.u32.u64 %r29, %rd45;
|
| 128 |
+
.loc 2 155 69 // triton_helpers.py:155:69 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:41:38 ]
|
| 129 |
+
selp.b32 %r54, %r54, %r29, %p17;
|
| 130 |
+
$L__tmp2:
|
| 131 |
+
.loc 1 32 40 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:32:40
|
| 132 |
+
add.s64 %rd11, %rd51, 4;
|
| 133 |
+
add.s64 %rd50, %rd50, 16;
|
| 134 |
+
setp.lt.u64 %p18, %rd51, 31996;
|
| 135 |
+
mov.b64 %rd51, %rd11;
|
| 136 |
+
@%p18 bra $L__BB0_5;
|
| 137 |
+
bra.uni $L__BB0_8;
|
| 138 |
+
$L__BB0_6: // %.split.preheader
|
| 139 |
+
shl.b64 %rd24, %rd19, 2;
|
| 140 |
+
mul.lo.s64 %rd25, %rd20, 128000;
|
| 141 |
+
sub.s64 %rd26, %rd24, %rd25;
|
| 142 |
+
mul.lo.s64 %rd27, %rd49, %rd26;
|
| 143 |
+
add.s32 %r19, %r1, %r4;
|
| 144 |
+
mad.wide.s32 %rd28, %r19, 128000, %rd27;
|
| 145 |
+
shl.b64 %rd29, %rd6, 2;
|
| 146 |
+
add.s64 %rd30, %rd28, %rd29;
|
| 147 |
+
add.s64 %rd52, %rd17, %rd30;
|
| 148 |
+
mov.b64 %rd53, -4;
|
| 149 |
+
$L__BB0_7: // %.split
|
| 150 |
+
// =>This Inner Loop Header: Depth=1
|
| 151 |
+
.loc 1 38 61 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:38:61
|
| 152 |
+
// begin inline asm
|
| 153 |
+
mov.u64 %rd31, 0x0;
|
| 154 |
+
createpolicy.fractional.L2::evict_first.b64 %rd31, 1.0;
|
| 155 |
+
// end inline asm
|
| 156 |
+
mov.b32 %r21, 0;
|
| 157 |
+
mov.pred %p3, 0;
|
| 158 |
+
// begin inline asm
|
| 159 |
+
mov.u32 %r20, %r21;
|
| 160 |
+
@%p3 ld.global.L1::evict_first.L2::cache_hint.b32 { %r20 }, [ %rd52 + 0 ], %rd31;
|
| 161 |
+
// end inline asm
|
| 162 |
+
.loc 1 32 40 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:32:40
|
| 163 |
+
add.s64 %rd53, %rd53, 4;
|
| 164 |
+
add.s64 %rd52, %rd52, 16;
|
| 165 |
+
setp.lt.u64 %p4, %rd53, 31996;
|
| 166 |
+
mov.b32 %r54, 2147483647;
|
| 167 |
+
mov.b32 %r53, 0fFF800000;
|
| 168 |
+
@%p4 bra $L__BB0_7;
|
| 169 |
+
$L__BB0_8: // %.split3.us
|
| 170 |
+
.loc 1 23 44 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:23:44
|
| 171 |
+
and.b32 %r30, %r2, 63;
|
| 172 |
+
.loc 1 23 23 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:23:23
|
| 173 |
+
or.b32 %r31, %r1, %r30;
|
| 174 |
+
.loc 1 24 21 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:24:21
|
| 175 |
+
setp.lt.s32 %p20, %r31, %r12;
|
| 176 |
+
$L__tmp3:
|
| 177 |
+
.loc 2 165 42 // triton_helpers.py:165:42 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 178 |
+
shfl.sync.bfly.b32 %r32, %r53, 2, 31, -1;
|
| 179 |
+
shfl.sync.bfly.b32 %r33, %r54, 2, 31, -1;
|
| 180 |
+
.loc 2 144 21 // triton_helpers.py:144:21 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 181 |
+
setp.gt.f32 %p21, %r53, %r32;
|
| 182 |
+
.loc 2 145 23 // triton_helpers.py:145:23 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 183 |
+
setp.eq.f32 %p22, %r53, %r32;
|
| 184 |
+
.loc 2 147 29 // triton_helpers.py:147:29 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 185 |
+
setp.nan.f32 %p23, %r53, %r53;
|
| 186 |
+
.loc 2 148 29 // triton_helpers.py:148:29 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 187 |
+
setp.nan.f32 %p24, %r32, %r32;
|
| 188 |
+
setp.num.f32 %p25, %r32, %r32;
|
| 189 |
+
.loc 2 149 27 // triton_helpers.py:149:27 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 190 |
+
and.pred %p26, %p23, %p25;
|
| 191 |
+
.loc 2 149 16 // triton_helpers.py:149:16 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 192 |
+
or.pred %p27, %p21, %p26;
|
| 193 |
+
.loc 2 151 27 // triton_helpers.py:151:27 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 194 |
+
and.pred %p28, %p23, %p24;
|
| 195 |
+
.loc 2 151 17 // triton_helpers.py:151:17 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 196 |
+
or.pred %p29, %p22, %p28;
|
| 197 |
+
.loc 2 154 31 // triton_helpers.py:154:31 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 198 |
+
setp.lt.s32 %p30, %r54, %r33;
|
| 199 |
+
.loc 2 154 21 // triton_helpers.py:154:21 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 200 |
+
and.pred %p31, %p30, %p29;
|
| 201 |
+
.loc 2 154 12 // triton_helpers.py:154:12 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 202 |
+
or.pred %p32, %p27, %p31;
|
| 203 |
+
.loc 2 155 35 // triton_helpers.py:155:35 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 204 |
+
selp.f32 %r34, %r53, %r32, %p32;
|
| 205 |
+
.loc 2 155 69 // triton_helpers.py:155:69 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 206 |
+
selp.b32 %r35, %r54, %r33, %p32;
|
| 207 |
+
.loc 2 165 42 // triton_helpers.py:165:42 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 208 |
+
shfl.sync.bfly.b32 %r36, %r34, 1, 31, -1;
|
| 209 |
+
shfl.sync.bfly.b32 %r37, %r35, 1, 31, -1;
|
| 210 |
+
.loc 2 144 21 // triton_helpers.py:144:21 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 211 |
+
setp.gt.f32 %p33, %r34, %r36;
|
| 212 |
+
.loc 2 145 23 // triton_helpers.py:145:23 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 213 |
+
setp.eq.f32 %p34, %r34, %r36;
|
| 214 |
+
.loc 2 147 29 // triton_helpers.py:147:29 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 215 |
+
setp.nan.f32 %p35, %r34, %r34;
|
| 216 |
+
.loc 2 148 29 // triton_helpers.py:148:29 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 217 |
+
setp.nan.f32 %p36, %r36, %r36;
|
| 218 |
+
.loc 2 154 31 // triton_helpers.py:154:31 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 219 |
+
setp.lt.s32 %p37, %r35, %r37;
|
| 220 |
+
.loc 2 155 69 // triton_helpers.py:155:69 @[ ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:45:75 ]
|
| 221 |
+
selp.b32 %r38, %r35, %r37, %p35;
|
| 222 |
+
selp.b32 %r39, %r38, %r37, %p36;
|
| 223 |
+
selp.b32 %r40, %r35, %r39, %p34;
|
| 224 |
+
selp.b32 %r41, %r40, %r37, %p37;
|
| 225 |
+
selp.b32 %r42, %r41, %r35, %p36;
|
| 226 |
+
selp.b32 %r43, %r42, %r41, %p35;
|
| 227 |
+
selp.b32 %r44, %r35, %r43, %p33;
|
| 228 |
+
$L__tmp4:
|
| 229 |
+
.loc 1 47 25 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:47:25
|
| 230 |
+
mad.wide.s32 %rd48, %r31, 8, %rd18;
|
| 231 |
+
.loc 1 47 36 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:47:36
|
| 232 |
+
mov.b32 %r45, global_smem;
|
| 233 |
+
add.s32 %r46, %r45, %r3;
|
| 234 |
+
st.shared.b32 [%r46], %r44;
|
| 235 |
+
bar.sync 0;
|
| 236 |
+
shl.b32 %r47, %r30, 2;
|
| 237 |
+
add.s32 %r48, %r45, %r47;
|
| 238 |
+
ld.shared.s32 %rd47, [%r48];
|
| 239 |
+
and.b32 %r49, %r2, 192;
|
| 240 |
+
setp.eq.b32 %p38, %r49, 0;
|
| 241 |
+
and.pred %p19, %p38, %p20;
|
| 242 |
+
// begin inline asm
|
| 243 |
+
@%p19 st.global.b64 [ %rd48 + 0 ], { %rd47 };
|
| 244 |
+
// end inline asm
|
| 245 |
+
.loc 1 47 4 // ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py:47:4
|
| 246 |
+
ret;
|
| 247 |
+
$L__tmp5:
|
| 248 |
+
$L__func_end0:
|
| 249 |
+
// -- End function
|
| 250 |
+
}
|
| 251 |
+
.file 1 "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py"
|
| 252 |
+
.file 2 "/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py"
|
| 253 |
+
.section .debug_abbrev
|
| 254 |
+
{
|
| 255 |
+
.b8 1 // Abbreviation Code
|
| 256 |
+
.b8 17 // DW_TAG_compile_unit
|
| 257 |
+
.b8 1 // DW_CHILDREN_yes
|
| 258 |
+
.b8 37 // DW_AT_producer
|
| 259 |
+
.b8 8 // DW_FORM_string
|
| 260 |
+
.b8 19 // DW_AT_language
|
| 261 |
+
.b8 5 // DW_FORM_data2
|
| 262 |
+
.b8 3 // DW_AT_name
|
| 263 |
+
.b8 8 // DW_FORM_string
|
| 264 |
+
.b8 16 // DW_AT_stmt_list
|
| 265 |
+
.b8 6 // DW_FORM_data4
|
| 266 |
+
.b8 27 // DW_AT_comp_dir
|
| 267 |
+
.b8 8 // DW_FORM_string
|
| 268 |
+
.b8 0 // EOM(1)
|
| 269 |
+
.b8 0 // EOM(2)
|
| 270 |
+
.b8 2 // Abbreviation Code
|
| 271 |
+
.b8 46 // DW_TAG_subprogram
|
| 272 |
+
.b8 0 // DW_CHILDREN_no
|
| 273 |
+
.b8 3 // DW_AT_name
|
| 274 |
+
.b8 8 // DW_FORM_string
|
| 275 |
+
.b8 32 // DW_AT_inline
|
| 276 |
+
.b8 11 // DW_FORM_data1
|
| 277 |
+
.b8 0 // EOM(1)
|
| 278 |
+
.b8 0 // EOM(2)
|
| 279 |
+
.b8 3 // Abbreviation Code
|
| 280 |
+
.b8 46 // DW_TAG_subprogram
|
| 281 |
+
.b8 1 // DW_CHILDREN_yes
|
| 282 |
+
.b8 17 // DW_AT_low_pc
|
| 283 |
+
.b8 1 // DW_FORM_addr
|
| 284 |
+
.b8 18 // DW_AT_high_pc
|
| 285 |
+
.b8 1 // DW_FORM_addr
|
| 286 |
+
.b8 49 // DW_AT_abstract_origin
|
| 287 |
+
.b8 19 // DW_FORM_ref4
|
| 288 |
+
.b8 0 // EOM(1)
|
| 289 |
+
.b8 0 // EOM(2)
|
| 290 |
+
.b8 4 // Abbreviation Code
|
| 291 |
+
.b8 29 // DW_TAG_inlined_subroutine
|
| 292 |
+
.b8 0 // DW_CHILDREN_no
|
| 293 |
+
.b8 49 // DW_AT_abstract_origin
|
| 294 |
+
.b8 19 // DW_FORM_ref4
|
| 295 |
+
.b8 17 // DW_AT_low_pc
|
| 296 |
+
.b8 1 // DW_FORM_addr
|
| 297 |
+
.b8 18 // DW_AT_high_pc
|
| 298 |
+
.b8 1 // DW_FORM_addr
|
| 299 |
+
.b8 88 // DW_AT_call_file
|
| 300 |
+
.b8 11 // DW_FORM_data1
|
| 301 |
+
.b8 89 // DW_AT_call_line
|
| 302 |
+
.b8 11 // DW_FORM_data1
|
| 303 |
+
.b8 87 // DW_AT_call_column
|
| 304 |
+
.b8 11 // DW_FORM_data1
|
| 305 |
+
.b8 0 // EOM(1)
|
| 306 |
+
.b8 0 // EOM(2)
|
| 307 |
+
.b8 0 // EOM(3)
|
| 308 |
+
}
|
| 309 |
+
.section .debug_info
|
| 310 |
+
{
|
| 311 |
+
.b32 234 // Length of Unit
|
| 312 |
+
.b8 2 // DWARF version number
|
| 313 |
+
.b8 0
|
| 314 |
+
.b32 .debug_abbrev // Offset Into Abbrev. Section
|
| 315 |
+
.b8 8 // Address Size (in bytes)
|
| 316 |
+
.b8 1 // Abbrev [1] 0xb:0xe3 DW_TAG_compile_unit
|
| 317 |
+
.b8 116 // DW_AT_producer
|
| 318 |
+
.b8 114
|
| 319 |
+
.b8 105
|
| 320 |
+
.b8 116
|
| 321 |
+
.b8 111
|
| 322 |
+
.b8 110
|
| 323 |
+
.b8 0
|
| 324 |
+
.b8 2 // DW_AT_language
|
| 325 |
+
.b8 0
|
| 326 |
+
.b8 99 // DW_AT_name
|
| 327 |
+
.b8 101
|
| 328 |
+
.b8 117
|
| 329 |
+
.b8 105
|
| 330 |
+
.b8 54
|
| 331 |
+
.b8 113
|
| 332 |
+
.b8 114
|
| 333 |
+
.b8 98
|
| 334 |
+
.b8 50
|
| 335 |
+
.b8 116
|
| 336 |
+
.b8 51
|
| 337 |
+
.b8 108
|
| 338 |
+
.b8 109
|
| 339 |
+
.b8 122
|
| 340 |
+
.b8 115
|
| 341 |
+
.b8 51
|
| 342 |
+
.b8 108
|
| 343 |
+
.b8 106
|
| 344 |
+
.b8 114
|
| 345 |
+
.b8 113
|
| 346 |
+
.b8 116
|
| 347 |
+
.b8 99
|
| 348 |
+
.b8 111
|
| 349 |
+
.b8 109
|
| 350 |
+
.b8 116
|
| 351 |
+
.b8 52
|
| 352 |
+
.b8 98
|
| 353 |
+
.b8 50
|
| 354 |
+
.b8 113
|
| 355 |
+
.b8 54
|
| 356 |
+
.b8 115
|
| 357 |
+
.b8 118
|
| 358 |
+
.b8 122
|
| 359 |
+
.b8 111
|
| 360 |
+
.b8 50
|
| 361 |
+
.b8 52
|
| 362 |
+
.b8 106
|
| 363 |
+
.b8 54
|
| 364 |
+
.b8 109
|
| 365 |
+
.b8 109
|
| 366 |
+
.b8 114
|
| 367 |
+
.b8 121
|
| 368 |
+
.b8 97
|
| 369 |
+
.b8 105
|
| 370 |
+
.b8 111
|
| 371 |
+
.b8 118
|
| 372 |
+
.b8 114
|
| 373 |
+
.b8 54
|
| 374 |
+
.b8 107
|
| 375 |
+
.b8 112
|
| 376 |
+
.b8 55
|
| 377 |
+
.b8 121
|
| 378 |
+
.b8 46
|
| 379 |
+
.b8 112
|
| 380 |
+
.b8 121
|
| 381 |
+
.b8 0
|
| 382 |
+
.b32 .debug_line // DW_AT_stmt_list
|
| 383 |
+
.b8 47 // DW_AT_comp_dir
|
| 384 |
+
.b8 119
|
| 385 |
+
.b8 111
|
| 386 |
+
.b8 114
|
| 387 |
+
.b8 107
|
| 388 |
+
.b8 115
|
| 389 |
+
.b8 112
|
| 390 |
+
.b8 97
|
| 391 |
+
.b8 99
|
| 392 |
+
.b8 101
|
| 393 |
+
.b8 47
|
| 394 |
+
.b8 104
|
| 395 |
+
.b8 97
|
| 396 |
+
.b8 110
|
| 397 |
+
.b8 114
|
| 398 |
+
.b8 117
|
| 399 |
+
.b8 105
|
| 400 |
+
.b8 47
|
| 401 |
+
.b8 83
|
| 402 |
+
.b8 112
|
| 403 |
+
.b8 101
|
| 404 |
+
.b8 99
|
| 405 |
+
.b8 70
|
| 406 |
+
.b8 111
|
| 407 |
+
.b8 114
|
| 408 |
+
.b8 103
|
| 409 |
+
.b8 101
|
| 410 |
+
.b8 45
|
| 411 |
+
.b8 101
|
| 412 |
+
.b8 120
|
| 413 |
+
.b8 116
|
| 414 |
+
.b8 47
|
| 415 |
+
.b8 99
|
| 416 |
+
.b8 97
|
| 417 |
+
.b8 99
|
| 418 |
+
.b8 104
|
| 419 |
+
.b8 101
|
| 420 |
+
.b8 47
|
| 421 |
+
.b8 99
|
| 422 |
+
.b8 111
|
| 423 |
+
.b8 109
|
| 424 |
+
.b8 112
|
| 425 |
+
.b8 105
|
| 426 |
+
.b8 108
|
| 427 |
+
.b8 101
|
| 428 |
+
.b8 100
|
| 429 |
+
.b8 95
|
| 430 |
+
.b8 107
|
| 431 |
+
.b8 101
|
| 432 |
+
.b8 114
|
| 433 |
+
.b8 110
|
| 434 |
+
.b8 101
|
| 435 |
+
.b8 108
|
| 436 |
+
.b8 115
|
| 437 |
+
.b8 47
|
| 438 |
+
.b8 101
|
| 439 |
+
.b8 117
|
| 440 |
+
.b8 0
|
| 441 |
+
.b8 2 // Abbrev [2] 0x8b:0x1c DW_TAG_subprogram
|
| 442 |
+
.b8 116 // DW_AT_name
|
| 443 |
+
.b8 114
|
| 444 |
+
.b8 105
|
| 445 |
+
.b8 116
|
| 446 |
+
.b8 111
|
| 447 |
+
.b8 110
|
| 448 |
+
.b8 95
|
| 449 |
+
.b8 114
|
| 450 |
+
.b8 101
|
| 451 |
+
.b8 100
|
| 452 |
+
.b8 95
|
| 453 |
+
.b8 102
|
| 454 |
+
.b8 117
|
| 455 |
+
.b8 115
|
| 456 |
+
.b8 101
|
| 457 |
+
.b8 100
|
| 458 |
+
.b8 95
|
| 459 |
+
.b8 97
|
| 460 |
+
.b8 114
|
| 461 |
+
.b8 103
|
| 462 |
+
.b8 109
|
| 463 |
+
.b8 97
|
| 464 |
+
.b8 120
|
| 465 |
+
.b8 95
|
| 466 |
+
.b8 49
|
| 467 |
+
.b8 0
|
| 468 |
+
.b8 1 // DW_AT_inline
|
| 469 |
+
.b8 3 // Abbrev [3] 0xa7:0x46 DW_TAG_subprogram
|
| 470 |
+
.b64 $L__func_begin0 // DW_AT_low_pc
|
| 471 |
+
.b64 $L__func_end0 // DW_AT_high_pc
|
| 472 |
+
.b32 139 // DW_AT_abstract_origin
|
| 473 |
+
.b8 4 // Abbrev [4] 0xbc:0x18 DW_TAG_inlined_subroutine
|
| 474 |
+
.b32 139 // DW_AT_abstract_origin
|
| 475 |
+
.b64 $L__tmp1 // DW_AT_low_pc
|
| 476 |
+
.b64 $L__tmp2 // DW_AT_high_pc
|
| 477 |
+
.b8 1 // DW_AT_call_file
|
| 478 |
+
.b8 41 // DW_AT_call_line
|
| 479 |
+
.b8 38 // DW_AT_call_column
|
| 480 |
+
.b8 4 // Abbrev [4] 0xd4:0x18 DW_TAG_inlined_subroutine
|
| 481 |
+
.b32 139 // DW_AT_abstract_origin
|
| 482 |
+
.b64 $L__tmp3 // DW_AT_low_pc
|
| 483 |
+
.b64 $L__tmp4 // DW_AT_high_pc
|
| 484 |
+
.b8 1 // DW_AT_call_file
|
| 485 |
+
.b8 45 // DW_AT_call_line
|
| 486 |
+
.b8 75 // DW_AT_call_column
|
| 487 |
+
.b8 0 // End Of Children Mark
|
| 488 |
+
.b8 0 // End Of Children Mark
|
| 489 |
+
}
|
| 490 |
+
.section .debug_macinfo { }
|
SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.source
ADDED
|
@@ -0,0 +1,323 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":18:0)
|
| 2 |
+
#loc35 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":143:0)
|
| 3 |
+
#loc47 = loc(unknown)
|
| 4 |
+
#loc55 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":86:0)
|
| 5 |
+
#loc59 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":63:0)
|
| 6 |
+
#loc68 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":164:0)
|
| 7 |
+
#loc72 = loc("in_ptr0"(#loc))
|
| 8 |
+
#loc73 = loc("out_ptr0"(#loc))
|
| 9 |
+
#loc74 = loc("ks0"(#loc))
|
| 10 |
+
#loc75 = loc("ks1"(#loc))
|
| 11 |
+
#loc76 = loc("xnumel"(#loc))
|
| 12 |
+
#loc77 = loc("r0_numel"(#loc))
|
| 13 |
+
#loc106 = loc("a_value"(#loc35))
|
| 14 |
+
#loc107 = loc("a_index"(#loc35))
|
| 15 |
+
#loc108 = loc("b_value"(#loc35))
|
| 16 |
+
#loc109 = loc("b_index"(#loc35))
|
| 17 |
+
#loc122 = loc("x"(#loc55))
|
| 18 |
+
#loc123 = loc("x"(#loc59))
|
| 19 |
+
#loc124 = loc("value"(#loc68))
|
| 20 |
+
#loc125 = loc("index"(#loc68))
|
| 21 |
+
module {
|
| 22 |
+
tt.func public @triton_red_fused_argmax_1(%in_ptr0: !tt.ptr<f32> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %out_ptr0: !tt.ptr<i64> {tt.divisibility = 16 : i32} loc("out_ptr0"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %xnumel: i32 loc("xnumel"(#loc)), %r0_numel: i32 {tt.divisibility = 16 : i32} loc("r0_numel"(#loc))) attributes {noinline = false} {
|
| 23 |
+
%r0_numel_0 = arith.constant 32000 : i32 loc(#loc78)
|
| 24 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc79)
|
| 25 |
+
%xoffset_1 = arith.constant 64 : i32 loc(#loc80)
|
| 26 |
+
%xoffset_2 = arith.constant 64 : i32 loc(#loc80)
|
| 27 |
+
%xoffset_3 = arith.muli %xoffset, %xoffset_2 : i32 loc(#loc80)
|
| 28 |
+
%xindex = tt.make_range {end = 64 : i32, start = 0 : i32} : tensor<64xi32> loc(#loc81)
|
| 29 |
+
%xindex_4 = tt.expand_dims %xindex {axis = 1 : i32} : tensor<64xi32> -> tensor<64x1xi32> loc(#loc82)
|
| 30 |
+
%xindex_5 = tt.splat %xoffset_3 : i32 -> tensor<64x1xi32> loc(#loc83)
|
| 31 |
+
%xindex_6 = arith.addi %xindex_5, %xindex_4 : tensor<64x1xi32> loc(#loc83)
|
| 32 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<64x1xi32> loc(#loc84)
|
| 33 |
+
%xmask_7 = arith.cmpi slt, %xindex_6, %xmask : tensor<64x1xi32> loc(#loc84)
|
| 34 |
+
%r0_base = tt.make_range {end = 4 : i32, start = 0 : i32} : tensor<4xi32> loc(#loc85)
|
| 35 |
+
%r0_base_8 = tt.expand_dims %r0_base {axis = 0 : i32} : tensor<4xi32> -> tensor<1x4xi32> loc(#loc86)
|
| 36 |
+
%x0 = arith.extsi %xindex_6 : tensor<64x1xi32> to tensor<64x1xi64> loc(#loc87)
|
| 37 |
+
%x0_9 = tt.splat %ks0 : i64 -> tensor<64x1xi64> loc(#loc87)
|
| 38 |
+
%x0_10 = arith.remsi %x0, %x0_9 : tensor<64x1xi64> loc(#loc87)
|
| 39 |
+
%x1 = arith.extsi %xindex_6 : tensor<64x1xi32> to tensor<64x1xi64> loc(#loc88)
|
| 40 |
+
%x1_11 = tt.splat %ks0 : i64 -> tensor<64x1xi64> loc(#loc88)
|
| 41 |
+
%x1_12 = arith.divsi %x1, %x1_11 : tensor<64x1xi64> loc(#loc88)
|
| 42 |
+
%_tmp2 = arith.constant 0xFF800000 : f32 loc(#loc89)
|
| 43 |
+
%_tmp2_13 = arith.constant dense<0xFF800000> : tensor<64x4xf32> loc(#loc89)
|
| 44 |
+
%_tmp2_index = arith.constant 2147483647 : i32 loc(#loc90)
|
| 45 |
+
%_tmp2_index_14 = arith.constant dense<2147483647> : tensor<64x4xi32> loc(#loc90)
|
| 46 |
+
%c0_i32 = arith.constant 0 : i32 loc(#loc14)
|
| 47 |
+
%c4_i32 = arith.constant 4 : i32 loc(#loc14)
|
| 48 |
+
%0 = arith.bitcast %c0_i32 : i32 to i32 loc(#loc14)
|
| 49 |
+
%1 = arith.bitcast %r0_numel_0 : i32 to i32 loc(#loc14)
|
| 50 |
+
%2 = arith.bitcast %c4_i32 : i32 to i32 loc(#loc14)
|
| 51 |
+
%3 = ub.poison : i32 loc(#loc14)
|
| 52 |
+
%_tmp2_index_15:2 = scf.for %r0_offset = %0 to %1 step %2 iter_args(%_tmp2_16 = %_tmp2_13, %_tmp2_index_17 = %_tmp2_index_14) -> (tensor<64x4xf32>, tensor<64x4xi32>) : i32 {
|
| 53 |
+
%r0_index = tt.splat %r0_offset : i32 -> tensor<1x4xi32> loc(#loc92)
|
| 54 |
+
%r0_index_18 = arith.addi %r0_index, %r0_base_8 : tensor<1x4xi32> loc(#loc92)
|
| 55 |
+
%r0_mask = arith.constant dense<32000> : tensor<1x4xi32> loc(#loc93)
|
| 56 |
+
%r0_mask_19 = arith.cmpi slt, %r0_index_18, %r0_mask : tensor<1x4xi32> loc(#loc93)
|
| 57 |
+
%tmp0 = arith.constant 32000 : i32 loc(#loc94)
|
| 58 |
+
%tmp0_20 = arith.constant 32000 : i64 loc(#loc94)
|
| 59 |
+
%tmp0_21 = arith.constant dense<32000> : tensor<64x1xi64> loc(#loc94)
|
| 60 |
+
%tmp0_22 = arith.muli %tmp0_21, %x0_10 : tensor<64x1xi64> loc(#loc94)
|
| 61 |
+
%tmp0_23 = arith.extsi %r0_index_18 : tensor<1x4xi32> to tensor<1x4xi64> loc(#loc95)
|
| 62 |
+
%tmp0_24 = tt.broadcast %tmp0_23 : tensor<1x4xi64> -> tensor<64x4xi64> loc(#loc95)
|
| 63 |
+
%tmp0_25 = tt.broadcast %tmp0_22 : tensor<64x1xi64> -> tensor<64x4xi64> loc(#loc95)
|
| 64 |
+
%tmp0_26 = arith.addi %tmp0_24, %tmp0_25 : tensor<64x4xi64> loc(#loc95)
|
| 65 |
+
%tmp0_27 = tt.splat %ks1 : i64 -> tensor<64x1xi64> loc(#loc96)
|
| 66 |
+
%tmp0_28 = arith.muli %tmp0_27, %x1_12 : tensor<64x1xi64> loc(#loc96)
|
| 67 |
+
%tmp0_29 = tt.broadcast %tmp0_28 : tensor<64x1xi64> -> tensor<64x4xi64> loc(#loc97)
|
| 68 |
+
%tmp0_30 = arith.addi %tmp0_26, %tmp0_29 : tensor<64x4xi64> loc(#loc97)
|
| 69 |
+
%tmp0_31 = tt.splat %in_ptr0 : !tt.ptr<f32> -> tensor<64x4x!tt.ptr<f32>> loc(#loc98)
|
| 70 |
+
%tmp0_32 = tt.addptr %tmp0_31, %tmp0_30 : tensor<64x4x!tt.ptr<f32>>, tensor<64x4xi64> loc(#loc98)
|
| 71 |
+
%tmp0_33 = tt.broadcast %r0_mask_19 : tensor<1x4xi1> -> tensor<64x4xi1> loc(#loc99)
|
| 72 |
+
%tmp0_34 = tt.broadcast %xmask_7 : tensor<64x1xi1> -> tensor<64x4xi1> loc(#loc99)
|
| 73 |
+
%tmp0_35 = arith.andi %tmp0_33, %tmp0_34 : tensor<64x4xi1> loc(#loc99)
|
| 74 |
+
%tmp0_36 = arith.constant 0.000000e+00 : f32 loc(#loc100)
|
| 75 |
+
%tmp0_37 = arith.constant dense<0.000000e+00> : tensor<64x4xf32> loc(#loc100)
|
| 76 |
+
%tmp0_38 = tt.load %tmp0_32, %tmp0_35, %tmp0_37 evictionPolicy = evict_first : tensor<64x4x!tt.ptr<f32>> loc(#loc100)
|
| 77 |
+
%8:2 = tt.call @torch._inductor.runtime.triton_helpers.maximum_with_index__fp32S64_4S_i32S64_4S_fp32S64_4S_i32S1_4S__(%_tmp2_16, %_tmp2_index_17, %tmp0_38, %r0_index_18) : (tensor<64x4xf32>, tensor<64x4xi32>, tensor<64x4xf32>, tensor<1x4xi32>) -> (tensor<64x4xf32>, tensor<64x4xi32>) loc(#loc24)
|
| 78 |
+
%_tmp2_39 = tt.broadcast %r0_mask_19 : tensor<1x4xi1> -> tensor<64x4xi1> loc(#loc101)
|
| 79 |
+
%_tmp2_40 = tt.broadcast %xmask_7 : tensor<64x1xi1> -> tensor<64x4xi1> loc(#loc101)
|
| 80 |
+
%_tmp2_41 = arith.andi %_tmp2_39, %_tmp2_40 : tensor<64x4xi1> loc(#loc101)
|
| 81 |
+
%_tmp2_42 = arith.select %_tmp2_41, %8#0, %_tmp2_16 : tensor<64x4xi1>, tensor<64x4xf32> loc(#loc102)
|
| 82 |
+
%_tmp2_index_43 = tt.broadcast %r0_mask_19 : tensor<1x4xi1> -> tensor<64x4xi1> loc(#loc103)
|
| 83 |
+
%_tmp2_index_44 = tt.broadcast %xmask_7 : tensor<64x1xi1> -> tensor<64x4xi1> loc(#loc103)
|
| 84 |
+
%_tmp2_index_45 = arith.andi %_tmp2_index_43, %_tmp2_index_44 : tensor<64x4xi1> loc(#loc103)
|
| 85 |
+
%_tmp2_index_46 = arith.select %_tmp2_index_45, %8#1, %_tmp2_index_17 : tensor<64x4xi1>, tensor<64x4xi32> loc(#loc104)
|
| 86 |
+
scf.yield %_tmp2_42, %_tmp2_index_46 : tensor<64x4xf32>, tensor<64x4xi32> loc(#loc29)
|
| 87 |
+
} loc(#loc126)
|
| 88 |
+
%4:2 = tt.call @"torch._inductor.runtime.triton_helpers.max_with_index__fp32S64_4S_i32S64_4S__(2,)cconstexpr_1_"(%_tmp2_index_15#0, %_tmp2_index_15#1) : (tensor<64x4xf32>, tensor<64x4xi32>) -> (tensor<64xf32>, tensor<64xi32>) loc(#loc30)
|
| 89 |
+
%tmp2 = tt.expand_dims %4#1 {axis = 1 : i32} : tensor<64xi32> -> tensor<64x1xi32> loc(#loc105)
|
| 90 |
+
%5 = tt.splat %out_ptr0 : !tt.ptr<i64> -> tensor<64x1x!tt.ptr<i64>> loc(#loc32)
|
| 91 |
+
%6 = tt.addptr %5, %xindex_6 : tensor<64x1x!tt.ptr<i64>>, tensor<64x1xi32> loc(#loc32)
|
| 92 |
+
%7 = arith.extsi %tmp2 : tensor<64x1xi32> to tensor<64x1xi64> loc(#loc33)
|
| 93 |
+
tt.store %6, %7, %xmask_7 : tensor<64x1x!tt.ptr<i64>> loc(#loc33)
|
| 94 |
+
tt.return loc(#loc34)
|
| 95 |
+
} loc(#loc)
|
| 96 |
+
tt.func private @torch._inductor.runtime.triton_helpers.maximum_with_index__fp32S64_4S_i32S64_4S_fp32S64_4S_i32S1_4S__(%a_value: tensor<64x4xf32> loc("a_value"(#loc35)), %a_index: tensor<64x4xi32> loc("a_index"(#loc35)), %b_value: tensor<64x4xf32> loc("b_value"(#loc35)), %b_index: tensor<1x4xi32> loc("b_index"(#loc35))) -> (tensor<64x4xf32>, tensor<64x4xi32>) attributes {noinline = false} {
|
| 97 |
+
%mask = arith.cmpf ogt, %a_value, %b_value : tensor<64x4xf32> loc(#loc127)
|
| 98 |
+
%equal = arith.cmpf oeq, %a_value, %b_value : tensor<64x4xf32> loc(#loc128)
|
| 99 |
+
%0 = tt.call @torch._inductor.runtime.triton_helpers.is_floating__fp32S64_4S__(%a_value) : (tensor<64x4xf32>) -> i1 loc(#loc38)
|
| 100 |
+
%1:2 = scf.if %0 -> (tensor<64x4xi1>, tensor<64x4xi1>) {
|
| 101 |
+
%a_isnan = arith.cmpf une, %a_value, %a_value : tensor<64x4xf32> loc(#loc112)
|
| 102 |
+
%b_isnan = arith.cmpf une, %b_value, %b_value : tensor<64x4xf32> loc(#loc113)
|
| 103 |
+
%mask_4 = arith.constant true loc(#loc114)
|
| 104 |
+
%mask_5 = arith.constant dense<true> : tensor<64x4xi1> loc(#loc114)
|
| 105 |
+
%mask_6 = arith.xori %b_isnan, %mask_5 : tensor<64x4xi1> loc(#loc114)
|
| 106 |
+
%mask_7 = arith.andi %a_isnan, %mask_6 : tensor<64x4xi1> loc(#loc115)
|
| 107 |
+
%mask_8 = arith.ori %mask, %mask_7 : tensor<64x4xi1> loc(#loc129)
|
| 108 |
+
%equal_9 = arith.andi %a_isnan, %b_isnan : tensor<64x4xi1> loc(#loc117)
|
| 109 |
+
%equal_10 = arith.ori %equal, %equal_9 : tensor<64x4xi1> loc(#loc130)
|
| 110 |
+
scf.yield %mask_8, %equal_10 : tensor<64x4xi1>, tensor<64x4xi1> loc(#loc130)
|
| 111 |
+
} else {
|
| 112 |
+
scf.yield %mask, %equal : tensor<64x4xi1>, tensor<64x4xi1> loc(#loc47)
|
| 113 |
+
} loc(#loc39)
|
| 114 |
+
%mask_0 = tt.broadcast %b_index : tensor<1x4xi32> -> tensor<64x4xi32> loc(#loc119)
|
| 115 |
+
%mask_1 = arith.cmpi slt, %a_index, %mask_0 : tensor<64x4xi32> loc(#loc119)
|
| 116 |
+
%mask_2 = arith.andi %1#1, %mask_1 : tensor<64x4xi1> loc(#loc120)
|
| 117 |
+
%mask_3 = arith.ori %1#0, %mask_2 : tensor<64x4xi1> loc(#loc121)
|
| 118 |
+
%2 = arith.select %mask_3, %a_value, %b_value : tensor<64x4xi1>, tensor<64x4xf32> loc(#loc51)
|
| 119 |
+
%3 = tt.broadcast %b_index : tensor<1x4xi32> -> tensor<64x4xi32> loc(#loc52)
|
| 120 |
+
%4 = arith.select %mask_3, %a_index, %3 : tensor<64x4xi1>, tensor<64x4xi32> loc(#loc52)
|
| 121 |
+
tt.return %2, %4 : tensor<64x4xf32>, tensor<64x4xi32> loc(#loc53)
|
| 122 |
+
^bb1: // no predecessors
|
| 123 |
+
%5 = ub.poison : tensor<64x4xf32> loc(#loc54)
|
| 124 |
+
%6 = ub.poison : tensor<64x4xi32> loc(#loc54)
|
| 125 |
+
tt.return %5, %6 : tensor<64x4xf32>, tensor<64x4xi32> loc(#loc54)
|
| 126 |
+
} loc(#loc35)
|
| 127 |
+
tt.func private @torch._inductor.runtime.triton_helpers.is_floating__fp32S64_4S__(%x: tensor<64x4xf32> loc("x"(#loc55))) -> i1 attributes {noinline = false} {
|
| 128 |
+
%0 = tt.call @torch._inductor.runtime.triton_helpers.promote_to_tensor__fp32S64_4S__(%x) : (tensor<64x4xf32>) -> tensor<64x4xf32> loc(#loc56)
|
| 129 |
+
%true = arith.constant true loc(#loc57)
|
| 130 |
+
tt.return %true : i1 loc(#loc57)
|
| 131 |
+
^bb1: // no predecessors
|
| 132 |
+
%1 = ub.poison : i1 loc(#loc58)
|
| 133 |
+
tt.return %1 : i1 loc(#loc58)
|
| 134 |
+
} loc(#loc55)
|
| 135 |
+
tt.func private @torch._inductor.runtime.triton_helpers.promote_to_tensor__fp32S64_4S__(%x: tensor<64x4xf32> loc("x"(#loc59))) -> tensor<64x4xf32> attributes {noinline = false} {
|
| 136 |
+
%0 = tt.call @"triton.language.standard.zeros____(0, 0)cconstexpr_1__(1,)cconstexpr_int1_"() : () -> tensor<1xi1> loc(#loc60)
|
| 137 |
+
%1 = arith.uitofp %0 : tensor<1xi1> to tensor<1xf32> loc(#loc61)
|
| 138 |
+
%2 = tt.expand_dims %1 {axis = 0 : i32} : tensor<1xf32> -> tensor<1x1xf32> loc(#loc61)
|
| 139 |
+
%3 = tt.broadcast %2 : tensor<1x1xf32> -> tensor<64x4xf32> loc(#loc61)
|
| 140 |
+
%4 = arith.addf %x, %3 : tensor<64x4xf32> loc(#loc61)
|
| 141 |
+
tt.return %4 : tensor<64x4xf32> loc(#loc62)
|
| 142 |
+
^bb1: // no predecessors
|
| 143 |
+
%5 = ub.poison : tensor<64x4xf32> loc(#loc63)
|
| 144 |
+
tt.return %5 : tensor<64x4xf32> loc(#loc63)
|
| 145 |
+
} loc(#loc59)
|
| 146 |
+
tt.func private @"triton.language.standard.zeros____(0, 0)cconstexpr_1__(1,)cconstexpr_int1_"() -> tensor<1xi1> attributes {noinline = false} {
|
| 147 |
+
%false = arith.constant false loc(#loc65)
|
| 148 |
+
%cst = arith.constant dense<false> : tensor<1xi1> loc(#loc65)
|
| 149 |
+
tt.return %cst : tensor<1xi1> loc(#loc66)
|
| 150 |
+
^bb1: // no predecessors
|
| 151 |
+
%0 = ub.poison : tensor<1xi1> loc(#loc67)
|
| 152 |
+
tt.return %0 : tensor<1xi1> loc(#loc67)
|
| 153 |
+
} loc(#loc64)
|
| 154 |
+
tt.func private @"torch._inductor.runtime.triton_helpers.max_with_index__fp32S64_4S_i32S64_4S__(2,)cconstexpr_1_"(%value: tensor<64x4xf32> loc("value"(#loc68)), %index: tensor<64x4xi32> loc("index"(#loc68))) -> (tensor<64xf32>, tensor<64xi32>) attributes {noinline = false} {
|
| 155 |
+
%0:2 = "tt.reduce"(%value, %index) <{axis = 1 : i32}> ({
|
| 156 |
+
^bb0(%arg2: f32 loc(unknown), %arg3: i32 loc(unknown), %arg4: f32 loc(unknown), %arg5: i32 loc(unknown)):
|
| 157 |
+
%3:2 = tt.call @torch._inductor.runtime.triton_helpers.maximum_with_index__fp32_i32_fp32_i32__(%arg2, %arg3, %arg4, %arg5) : (f32, i32, f32, i32) -> (f32, i32) loc(#loc69)
|
| 158 |
+
tt.reduce.return %3#0, %3#1 : f32, i32 loc(#loc69)
|
| 159 |
+
}) : (tensor<64x4xf32>, tensor<64x4xi32>) -> (tensor<64xf32>, tensor<64xi32>) loc(#loc69)
|
| 160 |
+
tt.return %0#0, %0#1 : tensor<64xf32>, tensor<64xi32> loc(#loc70)
|
| 161 |
+
^bb1: // no predecessors
|
| 162 |
+
%1 = ub.poison : tensor<64xf32> loc(#loc71)
|
| 163 |
+
%2 = ub.poison : tensor<64xi32> loc(#loc71)
|
| 164 |
+
tt.return %1, %2 : tensor<64xf32>, tensor<64xi32> loc(#loc71)
|
| 165 |
+
} loc(#loc68)
|
| 166 |
+
tt.func private @torch._inductor.runtime.triton_helpers.maximum_with_index__fp32_i32_fp32_i32__(%a_value: f32 loc("a_value"(#loc35)), %a_index: i32 loc("a_index"(#loc35)), %b_value: f32 loc("b_value"(#loc35)), %b_index: i32 loc("b_index"(#loc35))) -> (f32, i32) attributes {noinline = false} {
|
| 167 |
+
%mask = arith.cmpf ogt, %a_value, %b_value : f32 loc(#loc127)
|
| 168 |
+
%equal = arith.cmpf oeq, %a_value, %b_value : f32 loc(#loc128)
|
| 169 |
+
%0 = tt.call @torch._inductor.runtime.triton_helpers.is_floating__fp32__(%a_value) : (f32) -> i1 loc(#loc38)
|
| 170 |
+
%1:2 = scf.if %0 -> (i1, i1) {
|
| 171 |
+
%a_isnan = arith.cmpf une, %a_value, %a_value : f32 loc(#loc112)
|
| 172 |
+
%b_isnan = arith.cmpf une, %b_value, %b_value : f32 loc(#loc113)
|
| 173 |
+
%mask_3 = arith.constant true loc(#loc114)
|
| 174 |
+
%mask_4 = arith.xori %b_isnan, %mask_3 : i1 loc(#loc114)
|
| 175 |
+
%mask_5 = arith.andi %a_isnan, %mask_4 : i1 loc(#loc115)
|
| 176 |
+
%mask_6 = arith.ori %mask, %mask_5 : i1 loc(#loc129)
|
| 177 |
+
%equal_7 = arith.andi %a_isnan, %b_isnan : i1 loc(#loc117)
|
| 178 |
+
%equal_8 = arith.ori %equal, %equal_7 : i1 loc(#loc130)
|
| 179 |
+
scf.yield %mask_6, %equal_8 : i1, i1 loc(#loc130)
|
| 180 |
+
} else {
|
| 181 |
+
scf.yield %mask, %equal : i1, i1 loc(#loc47)
|
| 182 |
+
} loc(#loc39)
|
| 183 |
+
%mask_0 = arith.cmpi slt, %a_index, %b_index : i32 loc(#loc119)
|
| 184 |
+
%mask_1 = arith.andi %1#1, %mask_0 : i1 loc(#loc120)
|
| 185 |
+
%mask_2 = arith.ori %1#0, %mask_1 : i1 loc(#loc121)
|
| 186 |
+
%2 = arith.select %mask_2, %a_value, %b_value : f32 loc(#loc51)
|
| 187 |
+
%3 = arith.select %mask_2, %a_index, %b_index : i32 loc(#loc52)
|
| 188 |
+
tt.return %2, %3 : f32, i32 loc(#loc53)
|
| 189 |
+
^bb1: // no predecessors
|
| 190 |
+
%4 = ub.poison : f32 loc(#loc54)
|
| 191 |
+
%5 = ub.poison : i32 loc(#loc54)
|
| 192 |
+
tt.return %4, %5 : f32, i32 loc(#loc54)
|
| 193 |
+
} loc(#loc35)
|
| 194 |
+
tt.func private @torch._inductor.runtime.triton_helpers.is_floating__fp32__(%x: f32 loc("x"(#loc55))) -> i1 attributes {noinline = false} {
|
| 195 |
+
%0 = tt.call @torch._inductor.runtime.triton_helpers.promote_to_tensor__fp32__(%x) : (f32) -> tensor<1xf32> loc(#loc56)
|
| 196 |
+
%true = arith.constant true loc(#loc57)
|
| 197 |
+
tt.return %true : i1 loc(#loc57)
|
| 198 |
+
^bb1: // no predecessors
|
| 199 |
+
%1 = ub.poison : i1 loc(#loc58)
|
| 200 |
+
tt.return %1 : i1 loc(#loc58)
|
| 201 |
+
} loc(#loc55)
|
| 202 |
+
tt.func private @torch._inductor.runtime.triton_helpers.promote_to_tensor__fp32__(%x: f32 loc("x"(#loc59))) -> tensor<1xf32> attributes {noinline = false} {
|
| 203 |
+
%0 = tt.call @"triton.language.standard.zeros____(0, 0)cconstexpr_1__(1,)cconstexpr_int1_"() : () -> tensor<1xi1> loc(#loc60)
|
| 204 |
+
%1 = arith.uitofp %0 : tensor<1xi1> to tensor<1xf32> loc(#loc61)
|
| 205 |
+
%2 = tt.splat %x : f32 -> tensor<1xf32> loc(#loc61)
|
| 206 |
+
%3 = arith.addf %2, %1 : tensor<1xf32> loc(#loc61)
|
| 207 |
+
tt.return %3 : tensor<1xf32> loc(#loc62)
|
| 208 |
+
^bb1: // no predecessors
|
| 209 |
+
%4 = ub.poison : tensor<1xf32> loc(#loc63)
|
| 210 |
+
tt.return %4 : tensor<1xf32> loc(#loc63)
|
| 211 |
+
} loc(#loc59)
|
| 212 |
+
} loc(#loc)
|
| 213 |
+
#loc1 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":19:15)
|
| 214 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":22:28)
|
| 215 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":22:33)
|
| 216 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":23:36)
|
| 217 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":23:44)
|
| 218 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":23:23)
|
| 219 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":24:21)
|
| 220 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":25:27)
|
| 221 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":25:37)
|
| 222 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":27:19)
|
| 223 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":28:19)
|
| 224 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":29:55)
|
| 225 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":30:58)
|
| 226 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":32:40)
|
| 227 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":33:31)
|
| 228 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":34:29)
|
| 229 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:47)
|
| 230 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:41)
|
| 231 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:56)
|
| 232 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:52)
|
| 233 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:34)
|
| 234 |
+
#loc22 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:71)
|
| 235 |
+
#loc23 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:61)
|
| 236 |
+
#loc24 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":41:38)
|
| 237 |
+
#loc25 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":43:35)
|
| 238 |
+
#loc26 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":43:54)
|
| 239 |
+
#loc27 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":44:41)
|
| 240 |
+
#loc28 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":44:66)
|
| 241 |
+
#loc29 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":44:8)
|
| 242 |
+
#loc30 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":45:75)
|
| 243 |
+
#loc31 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":46:20)
|
| 244 |
+
#loc32 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":47:25)
|
| 245 |
+
#loc33 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":47:36)
|
| 246 |
+
#loc34 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":47:4)
|
| 247 |
+
#loc36 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":144:21)
|
| 248 |
+
#loc37 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":145:23)
|
| 249 |
+
#loc38 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":146:19)
|
| 250 |
+
#loc39 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":146:7)
|
| 251 |
+
#loc40 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":147:29)
|
| 252 |
+
#loc41 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":148:29)
|
| 253 |
+
#loc42 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":149:31)
|
| 254 |
+
#loc43 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":149:27)
|
| 255 |
+
#loc44 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":149:16)
|
| 256 |
+
#loc45 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":151:27)
|
| 257 |
+
#loc46 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":151:17)
|
| 258 |
+
#loc48 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":154:31)
|
| 259 |
+
#loc49 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":154:21)
|
| 260 |
+
#loc50 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":154:12)
|
| 261 |
+
#loc51 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":155:35)
|
| 262 |
+
#loc52 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":155:69)
|
| 263 |
+
#loc53 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":155:11)
|
| 264 |
+
#loc54 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":155:4)
|
| 265 |
+
#loc56 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":87:29)
|
| 266 |
+
#loc57 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":87:11)
|
| 267 |
+
#loc58 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":87:4)
|
| 268 |
+
#loc60 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":65:30)
|
| 269 |
+
#loc61 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":65:15)
|
| 270 |
+
#loc62 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":65:11)
|
| 271 |
+
#loc63 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":65:4)
|
| 272 |
+
#loc64 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":118:0)
|
| 273 |
+
#loc65 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":127:31)
|
| 274 |
+
#loc66 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":127:11)
|
| 275 |
+
#loc67 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":127:4)
|
| 276 |
+
#loc69 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":165:42)
|
| 277 |
+
#loc70 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":165:11)
|
| 278 |
+
#loc71 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":165:4)
|
| 279 |
+
#loc78 = loc("r0_numel"(#loc1))
|
| 280 |
+
#loc79 = loc("xoffset"(#loc2))
|
| 281 |
+
#loc80 = loc("xoffset"(#loc3))
|
| 282 |
+
#loc81 = loc("xindex"(#loc4))
|
| 283 |
+
#loc82 = loc("xindex"(#loc5))
|
| 284 |
+
#loc83 = loc("xindex"(#loc6))
|
| 285 |
+
#loc84 = loc("xmask"(#loc7))
|
| 286 |
+
#loc85 = loc("r0_base"(#loc8))
|
| 287 |
+
#loc86 = loc("r0_base"(#loc9))
|
| 288 |
+
#loc87 = loc("x0"(#loc10))
|
| 289 |
+
#loc88 = loc("x1"(#loc11))
|
| 290 |
+
#loc89 = loc("_tmp2"(#loc12))
|
| 291 |
+
#loc90 = loc("_tmp2_index"(#loc13))
|
| 292 |
+
#loc91 = loc("_tmp2"(#loc14))
|
| 293 |
+
#loc92 = loc("r0_index"(#loc15))
|
| 294 |
+
#loc93 = loc("r0_mask"(#loc16))
|
| 295 |
+
#loc94 = loc("tmp0"(#loc17))
|
| 296 |
+
#loc95 = loc("tmp0"(#loc18))
|
| 297 |
+
#loc96 = loc("tmp0"(#loc19))
|
| 298 |
+
#loc97 = loc("tmp0"(#loc20))
|
| 299 |
+
#loc98 = loc("tmp0"(#loc21))
|
| 300 |
+
#loc99 = loc("tmp0"(#loc22))
|
| 301 |
+
#loc100 = loc("tmp0"(#loc23))
|
| 302 |
+
#loc101 = loc("_tmp2"(#loc25))
|
| 303 |
+
#loc102 = loc("_tmp2"(#loc26))
|
| 304 |
+
#loc103 = loc("_tmp2_index"(#loc27))
|
| 305 |
+
#loc104 = loc("_tmp2_index"(#loc28))
|
| 306 |
+
#loc105 = loc("tmp2"(#loc31))
|
| 307 |
+
#loc110 = loc("mask"(#loc36))
|
| 308 |
+
#loc111 = loc("equal"(#loc37))
|
| 309 |
+
#loc112 = loc("a_isnan"(#loc40))
|
| 310 |
+
#loc113 = loc("b_isnan"(#loc41))
|
| 311 |
+
#loc114 = loc("mask"(#loc42))
|
| 312 |
+
#loc115 = loc("mask"(#loc43))
|
| 313 |
+
#loc116 = loc("mask"(#loc44))
|
| 314 |
+
#loc117 = loc("equal"(#loc45))
|
| 315 |
+
#loc118 = loc("equal"(#loc46))
|
| 316 |
+
#loc119 = loc("mask"(#loc48))
|
| 317 |
+
#loc120 = loc("mask"(#loc49))
|
| 318 |
+
#loc121 = loc("mask"(#loc50))
|
| 319 |
+
#loc126 = loc("_tmp2_index"(#loc91))
|
| 320 |
+
#loc127 = loc("mask"(#loc110))
|
| 321 |
+
#loc128 = loc("equal"(#loc111))
|
| 322 |
+
#loc129 = loc("mask"(#loc116))
|
| 323 |
+
#loc130 = loc("equal"(#loc118))
|
SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.ttgir
ADDED
|
@@ -0,0 +1,218 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#blocked = #ttg.blocked<{sizePerThread = [1, 1], threadsPerWarp = [8, 4], warpsPerCTA = [8, 1], order = [1, 0]}>
|
| 2 |
+
#blocked1 = #ttg.blocked<{sizePerThread = [1, 1], threadsPerWarp = [32, 1], warpsPerCTA = [2, 4], order = [0, 1]}>
|
| 3 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":18:0)
|
| 4 |
+
#loc1 = loc(unknown)
|
| 5 |
+
#loc39 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":45:75)
|
| 6 |
+
#loc44 = loc("in_ptr0"(#loc))
|
| 7 |
+
#loc45 = loc("out_ptr0"(#loc))
|
| 8 |
+
#loc46 = loc("ks0"(#loc))
|
| 9 |
+
#loc47 = loc("ks1"(#loc))
|
| 10 |
+
#loc48 = loc("xnumel"(#loc))
|
| 11 |
+
#loc49 = loc("r0_numel"(#loc))
|
| 12 |
+
#loc85 = loc(callsite(#loc1 at #loc39))
|
| 13 |
+
module attributes {"ttg.num-ctas" = 1 : i32, "ttg.num-warps" = 8 : i32, ttg.target = "cuda:90", "ttg.threads-per-warp" = 32 : i32} {
|
| 14 |
+
tt.func public @triton_red_fused_argmax_1(%in_ptr0: !tt.ptr<f32> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %out_ptr0: !tt.ptr<i64> {tt.divisibility = 16 : i32} loc("out_ptr0"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %xnumel: i32 loc("xnumel"(#loc)), %r0_numel: i32 {tt.divisibility = 16 : i32} loc("r0_numel"(#loc))) attributes {noinline = false} {
|
| 15 |
+
%cst = arith.constant dense<32000> : tensor<64x1xi64, #blocked> loc(#loc1)
|
| 16 |
+
%cst_0 = arith.constant dense<0.000000e+00> : tensor<64x4xf32, #blocked> loc(#loc1)
|
| 17 |
+
%c0_i32 = arith.constant 0 : i32 loc(#loc1)
|
| 18 |
+
%c32000_i32 = arith.constant 32000 : i32 loc(#loc1)
|
| 19 |
+
%c4_i32 = arith.constant 4 : i32 loc(#loc1)
|
| 20 |
+
%cst_1 = arith.constant dense<true> : tensor<64x4xi1, #blocked> loc(#loc1)
|
| 21 |
+
%true = arith.constant true loc(#loc1)
|
| 22 |
+
%cst_2 = arith.constant dense<32000> : tensor<1x4xi32, #blocked> loc(#loc1)
|
| 23 |
+
%cst_3 = arith.constant dense<2147483647> : tensor<64x4xi32, #blocked> loc(#loc1)
|
| 24 |
+
%cst_4 = arith.constant dense<0xFF800000> : tensor<64x4xf32, #blocked> loc(#loc1)
|
| 25 |
+
%c64_i32 = arith.constant 64 : i32 loc(#loc1)
|
| 26 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc50)
|
| 27 |
+
%xoffset_5 = arith.muli %xoffset, %c64_i32 : i32 loc(#loc51)
|
| 28 |
+
%xindex = tt.make_range {end = 64 : i32, start = 0 : i32} : tensor<64xi32, #ttg.slice<{dim = 1, parent = #blocked}>> loc(#loc52)
|
| 29 |
+
%xindex_6 = tt.make_range {end = 64 : i32, start = 0 : i32} : tensor<64xi32, #ttg.slice<{dim = 1, parent = #blocked1}>> loc(#loc52)
|
| 30 |
+
%xindex_7 = tt.expand_dims %xindex {axis = 1 : i32} : tensor<64xi32, #ttg.slice<{dim = 1, parent = #blocked}>> -> tensor<64x1xi32, #blocked> loc(#loc52)
|
| 31 |
+
%xindex_8 = tt.expand_dims %xindex_6 {axis = 1 : i32} : tensor<64xi32, #ttg.slice<{dim = 1, parent = #blocked1}>> -> tensor<64x1xi32, #blocked1> loc(#loc52)
|
| 32 |
+
%xindex_9 = tt.splat %xoffset_5 : i32 -> tensor<64x1xi32, #blocked> loc(#loc53)
|
| 33 |
+
%xindex_10 = tt.splat %xoffset_5 : i32 -> tensor<64x1xi32, #blocked1> loc(#loc53)
|
| 34 |
+
%xindex_11 = arith.addi %xindex_9, %xindex_7 : tensor<64x1xi32, #blocked> loc(#loc53)
|
| 35 |
+
%xindex_12 = arith.addi %xindex_10, %xindex_8 : tensor<64x1xi32, #blocked1> loc(#loc53)
|
| 36 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<64x1xi32, #blocked> loc(#loc54)
|
| 37 |
+
%xmask_13 = tt.splat %xnumel : i32 -> tensor<64x1xi32, #blocked1> loc(#loc54)
|
| 38 |
+
%xmask_14 = arith.cmpi slt, %xindex_11, %xmask : tensor<64x1xi32, #blocked> loc(#loc54)
|
| 39 |
+
%xmask_15 = arith.cmpi slt, %xindex_12, %xmask_13 : tensor<64x1xi32, #blocked1> loc(#loc54)
|
| 40 |
+
%r0_base = tt.make_range {end = 4 : i32, start = 0 : i32} : tensor<4xi32, #ttg.slice<{dim = 0, parent = #blocked}>> loc(#loc55)
|
| 41 |
+
%r0_base_16 = tt.expand_dims %r0_base {axis = 0 : i32} : tensor<4xi32, #ttg.slice<{dim = 0, parent = #blocked}>> -> tensor<1x4xi32, #blocked> loc(#loc55)
|
| 42 |
+
%x0 = arith.extsi %xindex_11 : tensor<64x1xi32, #blocked> to tensor<64x1xi64, #blocked> loc(#loc56)
|
| 43 |
+
%x0_17 = tt.splat %ks0 : i64 -> tensor<64x1xi64, #blocked> loc(#loc56)
|
| 44 |
+
%x0_18 = arith.remsi %x0, %x0_17 : tensor<64x1xi64, #blocked> loc(#loc56)
|
| 45 |
+
%x1 = arith.divsi %x0, %x0_17 : tensor<64x1xi64, #blocked> loc(#loc57)
|
| 46 |
+
%tmp0 = arith.muli %x0_18, %cst : tensor<64x1xi64, #blocked> loc(#loc58)
|
| 47 |
+
%tmp0_19 = tt.broadcast %tmp0 : tensor<64x1xi64, #blocked> -> tensor<64x4xi64, #blocked> loc(#loc59)
|
| 48 |
+
%tmp0_20 = tt.splat %ks1 : i64 -> tensor<64x1xi64, #blocked> loc(#loc60)
|
| 49 |
+
%tmp0_21 = arith.muli %tmp0_20, %x1 : tensor<64x1xi64, #blocked> loc(#loc60)
|
| 50 |
+
%tmp0_22 = tt.broadcast %tmp0_21 : tensor<64x1xi64, #blocked> -> tensor<64x4xi64, #blocked> loc(#loc61)
|
| 51 |
+
%tmp0_23 = tt.splat %in_ptr0 : !tt.ptr<f32> -> tensor<64x4x!tt.ptr<f32>, #blocked> loc(#loc62)
|
| 52 |
+
%tmp0_24 = tt.broadcast %xmask_14 : tensor<64x1xi1, #blocked> -> tensor<64x4xi1, #blocked> loc(#loc63)
|
| 53 |
+
%_tmp2_index:2 = scf.for %r0_offset = %c0_i32 to %c32000_i32 step %c4_i32 iter_args(%_tmp2 = %cst_4, %_tmp2_index_25 = %cst_3) -> (tensor<64x4xf32, #blocked>, tensor<64x4xi32, #blocked>) : i32 {
|
| 54 |
+
%r0_index = tt.splat %r0_offset : i32 -> tensor<1x4xi32, #blocked> loc(#loc65)
|
| 55 |
+
%r0_index_26 = arith.addi %r0_index, %r0_base_16 : tensor<1x4xi32, #blocked> loc(#loc65)
|
| 56 |
+
%r0_mask = arith.cmpi slt, %r0_index_26, %cst_2 : tensor<1x4xi32, #blocked> loc(#loc66)
|
| 57 |
+
%tmp0_27 = arith.extsi %r0_index_26 : tensor<1x4xi32, #blocked> to tensor<1x4xi64, #blocked> loc(#loc59)
|
| 58 |
+
%tmp0_28 = tt.broadcast %tmp0_27 : tensor<1x4xi64, #blocked> -> tensor<64x4xi64, #blocked> loc(#loc59)
|
| 59 |
+
%tmp0_29 = arith.addi %tmp0_28, %tmp0_19 : tensor<64x4xi64, #blocked> loc(#loc59)
|
| 60 |
+
%tmp0_30 = arith.addi %tmp0_29, %tmp0_22 : tensor<64x4xi64, #blocked> loc(#loc61)
|
| 61 |
+
%tmp0_31 = tt.addptr %tmp0_23, %tmp0_30 : tensor<64x4x!tt.ptr<f32>, #blocked>, tensor<64x4xi64, #blocked> loc(#loc62)
|
| 62 |
+
%tmp0_32 = tt.broadcast %r0_mask : tensor<1x4xi1, #blocked> -> tensor<64x4xi1, #blocked> loc(#loc63)
|
| 63 |
+
%tmp0_33 = arith.andi %tmp0_32, %tmp0_24 : tensor<64x4xi1, #blocked> loc(#loc63)
|
| 64 |
+
%tmp0_34 = tt.load %tmp0_31, %tmp0_33, %cst_0 evictionPolicy = evict_first : tensor<64x4x!tt.ptr<f32>, #blocked> loc(#loc67)
|
| 65 |
+
%mask = arith.cmpf ogt, %_tmp2, %tmp0_34 : tensor<64x4xf32, #blocked> loc(#loc110)
|
| 66 |
+
%equal = arith.cmpf oeq, %_tmp2, %tmp0_34 : tensor<64x4xf32, #blocked> loc(#loc111)
|
| 67 |
+
%a_isnan = arith.cmpf une, %_tmp2, %_tmp2 : tensor<64x4xf32, #blocked> loc(#loc90)
|
| 68 |
+
%b_isnan = arith.cmpf une, %tmp0_34, %tmp0_34 : tensor<64x4xf32, #blocked> loc(#loc91)
|
| 69 |
+
%mask_35 = arith.xori %b_isnan, %cst_1 : tensor<64x4xi1, #blocked> loc(#loc92)
|
| 70 |
+
%mask_36 = arith.andi %a_isnan, %mask_35 : tensor<64x4xi1, #blocked> loc(#loc93)
|
| 71 |
+
%mask_37 = arith.ori %mask, %mask_36 : tensor<64x4xi1, #blocked> loc(#loc112)
|
| 72 |
+
%equal_38 = arith.andi %a_isnan, %b_isnan : tensor<64x4xi1, #blocked> loc(#loc95)
|
| 73 |
+
%equal_39 = arith.ori %equal, %equal_38 : tensor<64x4xi1, #blocked> loc(#loc113)
|
| 74 |
+
%mask_40 = tt.broadcast %r0_index_26 : tensor<1x4xi32, #blocked> -> tensor<64x4xi32, #blocked> loc(#loc97)
|
| 75 |
+
%mask_41 = arith.cmpi slt, %_tmp2_index_25, %mask_40 : tensor<64x4xi32, #blocked> loc(#loc97)
|
| 76 |
+
%mask_42 = arith.andi %equal_39, %mask_41 : tensor<64x4xi1, #blocked> loc(#loc98)
|
| 77 |
+
%mask_43 = arith.ori %mask_37, %mask_42 : tensor<64x4xi1, #blocked> loc(#loc99)
|
| 78 |
+
%5 = arith.select %mask_43, %_tmp2, %tmp0_34 : tensor<64x4xi1, #blocked>, tensor<64x4xf32, #blocked> loc(#loc80)
|
| 79 |
+
%6 = arith.select %mask_43, %_tmp2_index_25, %mask_40 : tensor<64x4xi1, #blocked>, tensor<64x4xi32, #blocked> loc(#loc81)
|
| 80 |
+
%_tmp2_44 = arith.select %tmp0_33, %5, %_tmp2 : tensor<64x4xi1, #blocked>, tensor<64x4xf32, #blocked> loc(#loc82)
|
| 81 |
+
%_tmp2_index_45 = arith.select %tmp0_33, %6, %_tmp2_index_25 : tensor<64x4xi1, #blocked>, tensor<64x4xi32, #blocked> loc(#loc83)
|
| 82 |
+
scf.yield %_tmp2_44, %_tmp2_index_45 : tensor<64x4xf32, #blocked>, tensor<64x4xi32, #blocked> loc(#loc37)
|
| 83 |
+
} loc(#loc87)
|
| 84 |
+
%0:2 = "tt.reduce"(%_tmp2_index#0, %_tmp2_index#1) <{axis = 1 : i32}> ({
|
| 85 |
+
^bb0(%arg6: f32 loc(callsite(#loc1 at #loc39)), %arg7: i32 loc(callsite(#loc1 at #loc39)), %arg8: f32 loc(callsite(#loc1 at #loc39)), %arg9: i32 loc(callsite(#loc1 at #loc39))):
|
| 86 |
+
%mask = arith.cmpf ogt, %arg6, %arg8 : f32 loc(#loc114)
|
| 87 |
+
%equal = arith.cmpf oeq, %arg6, %arg8 : f32 loc(#loc115)
|
| 88 |
+
%a_isnan = arith.cmpf une, %arg6, %arg6 : f32 loc(#loc100)
|
| 89 |
+
%b_isnan = arith.cmpf une, %arg8, %arg8 : f32 loc(#loc101)
|
| 90 |
+
%mask_25 = arith.xori %b_isnan, %true : i1 loc(#loc102)
|
| 91 |
+
%mask_26 = arith.andi %a_isnan, %mask_25 : i1 loc(#loc103)
|
| 92 |
+
%mask_27 = arith.ori %mask, %mask_26 : i1 loc(#loc116)
|
| 93 |
+
%equal_28 = arith.andi %a_isnan, %b_isnan : i1 loc(#loc104)
|
| 94 |
+
%equal_29 = arith.ori %equal, %equal_28 : i1 loc(#loc117)
|
| 95 |
+
%mask_30 = arith.cmpi slt, %arg7, %arg9 : i32 loc(#loc105)
|
| 96 |
+
%mask_31 = arith.andi %equal_29, %mask_30 : i1 loc(#loc106)
|
| 97 |
+
%mask_32 = arith.ori %mask_27, %mask_31 : i1 loc(#loc107)
|
| 98 |
+
%5 = arith.select %mask_32, %arg6, %arg8 : f32 loc(#loc108)
|
| 99 |
+
%6 = arith.select %mask_32, %arg7, %arg9 : i32 loc(#loc109)
|
| 100 |
+
tt.reduce.return %5, %6 : f32, i32 loc(#loc84)
|
| 101 |
+
}) : (tensor<64x4xf32, #blocked>, tensor<64x4xi32, #blocked>) -> (tensor<64xf32, #ttg.slice<{dim = 1, parent = #blocked}>>, tensor<64xi32, #ttg.slice<{dim = 1, parent = #blocked}>>) loc(#loc84)
|
| 102 |
+
%tmp2 = tt.expand_dims %0#1 {axis = 1 : i32} : tensor<64xi32, #ttg.slice<{dim = 1, parent = #blocked}>> -> tensor<64x1xi32, #blocked> loc(#loc86)
|
| 103 |
+
%1 = tt.splat %out_ptr0 : !tt.ptr<i64> -> tensor<64x1x!tt.ptr<i64>, #blocked1> loc(#loc41)
|
| 104 |
+
%2 = tt.addptr %1, %xindex_12 : tensor<64x1x!tt.ptr<i64>, #blocked1>, tensor<64x1xi32, #blocked1> loc(#loc41)
|
| 105 |
+
%3 = ttg.convert_layout %tmp2 : tensor<64x1xi32, #blocked> -> tensor<64x1xi32, #blocked1> loc(#loc42)
|
| 106 |
+
%4 = arith.extsi %3 : tensor<64x1xi32, #blocked1> to tensor<64x1xi64, #blocked1> loc(#loc42)
|
| 107 |
+
tt.store %2, %4, %xmask_15 : tensor<64x1x!tt.ptr<i64>, #blocked1> loc(#loc42)
|
| 108 |
+
tt.return loc(#loc43)
|
| 109 |
+
} loc(#loc)
|
| 110 |
+
} loc(#loc)
|
| 111 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":22:28)
|
| 112 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":22:33)
|
| 113 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":23:44)
|
| 114 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":23:23)
|
| 115 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":24:21)
|
| 116 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":25:37)
|
| 117 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":27:19)
|
| 118 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":28:19)
|
| 119 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:47)
|
| 120 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:41)
|
| 121 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:56)
|
| 122 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:52)
|
| 123 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:34)
|
| 124 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:71)
|
| 125 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":32:40)
|
| 126 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":33:31)
|
| 127 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":34:29)
|
| 128 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:61)
|
| 129 |
+
#loc20 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":144:21)
|
| 130 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":41:38)
|
| 131 |
+
#loc22 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":145:23)
|
| 132 |
+
#loc23 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":147:29)
|
| 133 |
+
#loc24 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":148:29)
|
| 134 |
+
#loc25 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":149:31)
|
| 135 |
+
#loc26 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":149:27)
|
| 136 |
+
#loc27 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":149:16)
|
| 137 |
+
#loc28 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":151:27)
|
| 138 |
+
#loc29 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":151:17)
|
| 139 |
+
#loc30 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":154:31)
|
| 140 |
+
#loc31 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":154:21)
|
| 141 |
+
#loc32 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":154:12)
|
| 142 |
+
#loc33 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":155:35)
|
| 143 |
+
#loc34 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":155:69)
|
| 144 |
+
#loc35 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":43:54)
|
| 145 |
+
#loc36 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":44:66)
|
| 146 |
+
#loc37 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":44:8)
|
| 147 |
+
#loc38 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":165:42)
|
| 148 |
+
#loc40 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":46:20)
|
| 149 |
+
#loc41 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":47:25)
|
| 150 |
+
#loc42 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":47:36)
|
| 151 |
+
#loc43 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":47:4)
|
| 152 |
+
#loc50 = loc("xoffset"(#loc2))
|
| 153 |
+
#loc51 = loc("xoffset"(#loc3))
|
| 154 |
+
#loc52 = loc("xindex"(#loc4))
|
| 155 |
+
#loc53 = loc("xindex"(#loc5))
|
| 156 |
+
#loc54 = loc("xmask"(#loc6))
|
| 157 |
+
#loc55 = loc("r0_base"(#loc7))
|
| 158 |
+
#loc56 = loc("x0"(#loc8))
|
| 159 |
+
#loc57 = loc("x1"(#loc9))
|
| 160 |
+
#loc58 = loc("tmp0"(#loc10))
|
| 161 |
+
#loc59 = loc("tmp0"(#loc11))
|
| 162 |
+
#loc60 = loc("tmp0"(#loc12))
|
| 163 |
+
#loc61 = loc("tmp0"(#loc13))
|
| 164 |
+
#loc62 = loc("tmp0"(#loc14))
|
| 165 |
+
#loc63 = loc("tmp0"(#loc15))
|
| 166 |
+
#loc64 = loc("_tmp2"(#loc16))
|
| 167 |
+
#loc65 = loc("r0_index"(#loc17))
|
| 168 |
+
#loc66 = loc("r0_mask"(#loc18))
|
| 169 |
+
#loc67 = loc("tmp0"(#loc19))
|
| 170 |
+
#loc68 = loc("mask"(#loc20))
|
| 171 |
+
#loc69 = loc("equal"(#loc22))
|
| 172 |
+
#loc70 = loc("a_isnan"(#loc23))
|
| 173 |
+
#loc71 = loc("b_isnan"(#loc24))
|
| 174 |
+
#loc72 = loc("mask"(#loc25))
|
| 175 |
+
#loc73 = loc("mask"(#loc26))
|
| 176 |
+
#loc74 = loc("mask"(#loc27))
|
| 177 |
+
#loc75 = loc("equal"(#loc28))
|
| 178 |
+
#loc76 = loc("equal"(#loc29))
|
| 179 |
+
#loc77 = loc("mask"(#loc30))
|
| 180 |
+
#loc78 = loc("mask"(#loc31))
|
| 181 |
+
#loc79 = loc("mask"(#loc32))
|
| 182 |
+
#loc80 = loc(callsite(#loc33 at #loc21))
|
| 183 |
+
#loc81 = loc(callsite(#loc34 at #loc21))
|
| 184 |
+
#loc82 = loc("_tmp2"(#loc35))
|
| 185 |
+
#loc83 = loc("_tmp2_index"(#loc36))
|
| 186 |
+
#loc84 = loc(callsite(#loc38 at #loc39))
|
| 187 |
+
#loc86 = loc("tmp2"(#loc40))
|
| 188 |
+
#loc87 = loc("_tmp2_index"(#loc64))
|
| 189 |
+
#loc88 = loc("mask"(#loc68))
|
| 190 |
+
#loc89 = loc("equal"(#loc69))
|
| 191 |
+
#loc90 = loc(callsite(#loc70 at #loc21))
|
| 192 |
+
#loc91 = loc(callsite(#loc71 at #loc21))
|
| 193 |
+
#loc92 = loc(callsite(#loc72 at #loc21))
|
| 194 |
+
#loc93 = loc(callsite(#loc73 at #loc21))
|
| 195 |
+
#loc94 = loc("mask"(#loc74))
|
| 196 |
+
#loc95 = loc(callsite(#loc75 at #loc21))
|
| 197 |
+
#loc96 = loc("equal"(#loc76))
|
| 198 |
+
#loc97 = loc(callsite(#loc77 at #loc21))
|
| 199 |
+
#loc98 = loc(callsite(#loc78 at #loc21))
|
| 200 |
+
#loc99 = loc(callsite(#loc79 at #loc21))
|
| 201 |
+
#loc100 = loc(callsite(#loc70 at #loc84))
|
| 202 |
+
#loc101 = loc(callsite(#loc71 at #loc84))
|
| 203 |
+
#loc102 = loc(callsite(#loc72 at #loc84))
|
| 204 |
+
#loc103 = loc(callsite(#loc73 at #loc84))
|
| 205 |
+
#loc104 = loc(callsite(#loc75 at #loc84))
|
| 206 |
+
#loc105 = loc(callsite(#loc77 at #loc84))
|
| 207 |
+
#loc106 = loc(callsite(#loc78 at #loc84))
|
| 208 |
+
#loc107 = loc(callsite(#loc79 at #loc84))
|
| 209 |
+
#loc108 = loc(callsite(#loc33 at #loc84))
|
| 210 |
+
#loc109 = loc(callsite(#loc34 at #loc84))
|
| 211 |
+
#loc110 = loc(callsite(#loc88 at #loc21))
|
| 212 |
+
#loc111 = loc(callsite(#loc89 at #loc21))
|
| 213 |
+
#loc112 = loc(callsite(#loc94 at #loc21))
|
| 214 |
+
#loc113 = loc(callsite(#loc96 at #loc21))
|
| 215 |
+
#loc114 = loc(callsite(#loc88 at #loc84))
|
| 216 |
+
#loc115 = loc(callsite(#loc89 at #loc84))
|
| 217 |
+
#loc116 = loc(callsite(#loc94 at #loc84))
|
| 218 |
+
#loc117 = loc(callsite(#loc96 at #loc84))
|
SpecForge-ext/cache/compiled_kernels/triton/0/25SMJXR2INGZCZI64NAKGLW77JZOIG6LAES6NHHOQFOTKNXS6PHA/triton_red_fused_argmax_1.ttir
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":18:0)
|
| 2 |
+
#loc1 = loc(unknown)
|
| 3 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":45:75)
|
| 4 |
+
#loc48 = loc("in_ptr0"(#loc))
|
| 5 |
+
#loc49 = loc("out_ptr0"(#loc))
|
| 6 |
+
#loc50 = loc("ks0"(#loc))
|
| 7 |
+
#loc51 = loc("ks1"(#loc))
|
| 8 |
+
#loc52 = loc("xnumel"(#loc))
|
| 9 |
+
#loc53 = loc("r0_numel"(#loc))
|
| 10 |
+
#loc54 = loc(callsite(#loc1 at #loc2))
|
| 11 |
+
module {
|
| 12 |
+
tt.func public @triton_red_fused_argmax_1(%in_ptr0: !tt.ptr<f32> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %out_ptr0: !tt.ptr<i64> {tt.divisibility = 16 : i32} loc("out_ptr0"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %xnumel: i32 loc("xnumel"(#loc)), %r0_numel: i32 {tt.divisibility = 16 : i32} loc("r0_numel"(#loc))) attributes {noinline = false} {
|
| 13 |
+
%true = arith.constant true loc(#loc54)
|
| 14 |
+
%cst = arith.constant dense<true> : tensor<64x4xi1> loc(#loc1)
|
| 15 |
+
%c4_i32 = arith.constant 4 : i32 loc(#loc3)
|
| 16 |
+
%c32000_i32 = arith.constant 32000 : i32 loc(#loc3)
|
| 17 |
+
%c0_i32 = arith.constant 0 : i32 loc(#loc3)
|
| 18 |
+
%cst_0 = arith.constant dense<0.000000e+00> : tensor<64x4xf32> loc(#loc1)
|
| 19 |
+
%cst_1 = arith.constant dense<32000> : tensor<64x1xi64> loc(#loc1)
|
| 20 |
+
%cst_2 = arith.constant dense<32000> : tensor<1x4xi32> loc(#loc1)
|
| 21 |
+
%_tmp2_index = arith.constant dense<2147483647> : tensor<64x4xi32> loc(#loc55)
|
| 22 |
+
%_tmp2 = arith.constant dense<0xFF800000> : tensor<64x4xf32> loc(#loc56)
|
| 23 |
+
%c64_i32 = arith.constant 64 : i32 loc(#loc1)
|
| 24 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc57)
|
| 25 |
+
%xoffset_3 = arith.muli %xoffset, %c64_i32 : i32 loc(#loc58)
|
| 26 |
+
%xindex = tt.make_range {end = 64 : i32, start = 0 : i32} : tensor<64xi32> loc(#loc59)
|
| 27 |
+
%xindex_4 = tt.expand_dims %xindex {axis = 1 : i32} : tensor<64xi32> -> tensor<64x1xi32> loc(#loc60)
|
| 28 |
+
%xindex_5 = tt.splat %xoffset_3 : i32 -> tensor<64x1xi32> loc(#loc61)
|
| 29 |
+
%xindex_6 = arith.addi %xindex_5, %xindex_4 : tensor<64x1xi32> loc(#loc61)
|
| 30 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<64x1xi32> loc(#loc62)
|
| 31 |
+
%xmask_7 = arith.cmpi slt, %xindex_6, %xmask : tensor<64x1xi32> loc(#loc62)
|
| 32 |
+
%r0_base = tt.make_range {end = 4 : i32, start = 0 : i32} : tensor<4xi32> loc(#loc63)
|
| 33 |
+
%r0_base_8 = tt.expand_dims %r0_base {axis = 0 : i32} : tensor<4xi32> -> tensor<1x4xi32> loc(#loc64)
|
| 34 |
+
%x0 = arith.extsi %xindex_6 : tensor<64x1xi32> to tensor<64x1xi64> loc(#loc65)
|
| 35 |
+
%x0_9 = tt.splat %ks0 : i64 -> tensor<64x1xi64> loc(#loc65)
|
| 36 |
+
%x0_10 = arith.remsi %x0, %x0_9 : tensor<64x1xi64> loc(#loc65)
|
| 37 |
+
%x1 = arith.divsi %x0, %x0_9 : tensor<64x1xi64> loc(#loc66)
|
| 38 |
+
%_tmp2_index_11:2 = scf.for %r0_offset = %c0_i32 to %c32000_i32 step %c4_i32 iter_args(%_tmp2_12 = %_tmp2, %_tmp2_index_13 = %_tmp2_index) -> (tensor<64x4xf32>, tensor<64x4xi32>) : i32 {
|
| 39 |
+
%r0_index = tt.splat %r0_offset : i32 -> tensor<1x4xi32> loc(#loc68)
|
| 40 |
+
%r0_index_14 = arith.addi %r0_index, %r0_base_8 : tensor<1x4xi32> loc(#loc68)
|
| 41 |
+
%r0_mask = arith.cmpi slt, %r0_index_14, %cst_2 : tensor<1x4xi32> loc(#loc69)
|
| 42 |
+
%tmp0 = arith.muli %x0_10, %cst_1 : tensor<64x1xi64> loc(#loc70)
|
| 43 |
+
%tmp0_15 = arith.extsi %r0_index_14 : tensor<1x4xi32> to tensor<1x4xi64> loc(#loc71)
|
| 44 |
+
%tmp0_16 = tt.broadcast %tmp0_15 : tensor<1x4xi64> -> tensor<64x4xi64> loc(#loc71)
|
| 45 |
+
%tmp0_17 = tt.broadcast %tmp0 : tensor<64x1xi64> -> tensor<64x4xi64> loc(#loc71)
|
| 46 |
+
%tmp0_18 = arith.addi %tmp0_16, %tmp0_17 : tensor<64x4xi64> loc(#loc71)
|
| 47 |
+
%tmp0_19 = tt.splat %ks1 : i64 -> tensor<64x1xi64> loc(#loc72)
|
| 48 |
+
%tmp0_20 = arith.muli %tmp0_19, %x1 : tensor<64x1xi64> loc(#loc72)
|
| 49 |
+
%tmp0_21 = tt.broadcast %tmp0_20 : tensor<64x1xi64> -> tensor<64x4xi64> loc(#loc73)
|
| 50 |
+
%tmp0_22 = arith.addi %tmp0_18, %tmp0_21 : tensor<64x4xi64> loc(#loc73)
|
| 51 |
+
%tmp0_23 = tt.splat %in_ptr0 : !tt.ptr<f32> -> tensor<64x4x!tt.ptr<f32>> loc(#loc74)
|
| 52 |
+
%tmp0_24 = tt.addptr %tmp0_23, %tmp0_22 : tensor<64x4x!tt.ptr<f32>>, tensor<64x4xi64> loc(#loc74)
|
| 53 |
+
%tmp0_25 = tt.broadcast %r0_mask : tensor<1x4xi1> -> tensor<64x4xi1> loc(#loc75)
|
| 54 |
+
%tmp0_26 = tt.broadcast %xmask_7 : tensor<64x1xi1> -> tensor<64x4xi1> loc(#loc75)
|
| 55 |
+
%tmp0_27 = arith.andi %tmp0_25, %tmp0_26 : tensor<64x4xi1> loc(#loc75)
|
| 56 |
+
%tmp0_28 = tt.load %tmp0_24, %tmp0_27, %cst_0 evictionPolicy = evict_first : tensor<64x4x!tt.ptr<f32>> loc(#loc76)
|
| 57 |
+
%mask = arith.cmpf ogt, %_tmp2_12, %tmp0_28 : tensor<64x4xf32> loc(#loc118)
|
| 58 |
+
%equal = arith.cmpf oeq, %_tmp2_12, %tmp0_28 : tensor<64x4xf32> loc(#loc119)
|
| 59 |
+
%a_isnan = arith.cmpf une, %_tmp2_12, %_tmp2_12 : tensor<64x4xf32> loc(#loc98)
|
| 60 |
+
%b_isnan = arith.cmpf une, %tmp0_28, %tmp0_28 : tensor<64x4xf32> loc(#loc99)
|
| 61 |
+
%mask_29 = arith.xori %b_isnan, %cst : tensor<64x4xi1> loc(#loc100)
|
| 62 |
+
%mask_30 = arith.andi %a_isnan, %mask_29 : tensor<64x4xi1> loc(#loc101)
|
| 63 |
+
%mask_31 = arith.ori %mask, %mask_30 : tensor<64x4xi1> loc(#loc120)
|
| 64 |
+
%equal_32 = arith.andi %a_isnan, %b_isnan : tensor<64x4xi1> loc(#loc103)
|
| 65 |
+
%equal_33 = arith.ori %equal, %equal_32 : tensor<64x4xi1> loc(#loc121)
|
| 66 |
+
%mask_34 = tt.broadcast %r0_index_14 : tensor<1x4xi32> -> tensor<64x4xi32> loc(#loc105)
|
| 67 |
+
%mask_35 = arith.cmpi slt, %_tmp2_index_13, %mask_34 : tensor<64x4xi32> loc(#loc105)
|
| 68 |
+
%mask_36 = arith.andi %equal_33, %mask_35 : tensor<64x4xi1> loc(#loc106)
|
| 69 |
+
%mask_37 = arith.ori %mask_31, %mask_36 : tensor<64x4xi1> loc(#loc107)
|
| 70 |
+
%4 = arith.select %mask_37, %_tmp2_12, %tmp0_28 : tensor<64x4xi1>, tensor<64x4xf32> loc(#loc89)
|
| 71 |
+
%5 = arith.select %mask_37, %_tmp2_index_13, %mask_34 : tensor<64x4xi1>, tensor<64x4xi32> loc(#loc90)
|
| 72 |
+
%_tmp2_38 = arith.select %tmp0_27, %4, %_tmp2_12 : tensor<64x4xi1>, tensor<64x4xf32> loc(#loc91)
|
| 73 |
+
%_tmp2_index_39 = arith.select %tmp0_27, %5, %_tmp2_index_13 : tensor<64x4xi1>, tensor<64x4xi32> loc(#loc92)
|
| 74 |
+
scf.yield %_tmp2_38, %_tmp2_index_39 : tensor<64x4xf32>, tensor<64x4xi32> loc(#loc42)
|
| 75 |
+
} loc(#loc95)
|
| 76 |
+
%0:2 = "tt.reduce"(%_tmp2_index_11#0, %_tmp2_index_11#1) <{axis = 1 : i32}> ({
|
| 77 |
+
^bb0(%arg6: f32 loc(callsite(#loc1 at #loc2)), %arg7: i32 loc(callsite(#loc1 at #loc2)), %arg8: f32 loc(callsite(#loc1 at #loc2)), %arg9: i32 loc(callsite(#loc1 at #loc2))):
|
| 78 |
+
%mask = arith.cmpf ogt, %arg6, %arg8 : f32 loc(#loc122)
|
| 79 |
+
%equal = arith.cmpf oeq, %arg6, %arg8 : f32 loc(#loc123)
|
| 80 |
+
%a_isnan = arith.cmpf une, %arg6, %arg6 : f32 loc(#loc108)
|
| 81 |
+
%b_isnan = arith.cmpf une, %arg8, %arg8 : f32 loc(#loc109)
|
| 82 |
+
%mask_12 = arith.xori %b_isnan, %true : i1 loc(#loc110)
|
| 83 |
+
%mask_13 = arith.andi %a_isnan, %mask_12 : i1 loc(#loc111)
|
| 84 |
+
%mask_14 = arith.ori %mask, %mask_13 : i1 loc(#loc124)
|
| 85 |
+
%equal_15 = arith.andi %a_isnan, %b_isnan : i1 loc(#loc112)
|
| 86 |
+
%equal_16 = arith.ori %equal, %equal_15 : i1 loc(#loc125)
|
| 87 |
+
%mask_17 = arith.cmpi slt, %arg7, %arg9 : i32 loc(#loc113)
|
| 88 |
+
%mask_18 = arith.andi %equal_16, %mask_17 : i1 loc(#loc114)
|
| 89 |
+
%mask_19 = arith.ori %mask_14, %mask_18 : i1 loc(#loc115)
|
| 90 |
+
%4 = arith.select %mask_19, %arg6, %arg8 : f32 loc(#loc116)
|
| 91 |
+
%5 = arith.select %mask_19, %arg7, %arg9 : i32 loc(#loc117)
|
| 92 |
+
tt.reduce.return %4, %5 : f32, i32 loc(#loc93)
|
| 93 |
+
}) : (tensor<64x4xf32>, tensor<64x4xi32>) -> (tensor<64xf32>, tensor<64xi32>) loc(#loc93)
|
| 94 |
+
%tmp2 = tt.expand_dims %0#1 {axis = 1 : i32} : tensor<64xi32> -> tensor<64x1xi32> loc(#loc94)
|
| 95 |
+
%1 = tt.splat %out_ptr0 : !tt.ptr<i64> -> tensor<64x1x!tt.ptr<i64>> loc(#loc45)
|
| 96 |
+
%2 = tt.addptr %1, %xindex_6 : tensor<64x1x!tt.ptr<i64>>, tensor<64x1xi32> loc(#loc45)
|
| 97 |
+
%3 = arith.extsi %tmp2 : tensor<64x1xi32> to tensor<64x1xi64> loc(#loc46)
|
| 98 |
+
tt.store %2, %3, %xmask_7 : tensor<64x1x!tt.ptr<i64>> loc(#loc46)
|
| 99 |
+
tt.return loc(#loc47)
|
| 100 |
+
} loc(#loc)
|
| 101 |
+
} loc(#loc)
|
| 102 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":32:40)
|
| 103 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":30:58)
|
| 104 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":29:55)
|
| 105 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":22:28)
|
| 106 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":22:33)
|
| 107 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":23:36)
|
| 108 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":23:44)
|
| 109 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":23:23)
|
| 110 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":24:21)
|
| 111 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":25:27)
|
| 112 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":25:37)
|
| 113 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":27:19)
|
| 114 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":28:19)
|
| 115 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":33:31)
|
| 116 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":34:29)
|
| 117 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:47)
|
| 118 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:41)
|
| 119 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:56)
|
| 120 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:52)
|
| 121 |
+
#loc22 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:34)
|
| 122 |
+
#loc23 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:71)
|
| 123 |
+
#loc24 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":38:61)
|
| 124 |
+
#loc25 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":144:21)
|
| 125 |
+
#loc26 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":41:38)
|
| 126 |
+
#loc27 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":145:23)
|
| 127 |
+
#loc28 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":147:29)
|
| 128 |
+
#loc29 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":148:29)
|
| 129 |
+
#loc30 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":149:31)
|
| 130 |
+
#loc31 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":149:27)
|
| 131 |
+
#loc32 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":149:16)
|
| 132 |
+
#loc33 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":151:27)
|
| 133 |
+
#loc34 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":151:17)
|
| 134 |
+
#loc35 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":154:31)
|
| 135 |
+
#loc36 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":154:21)
|
| 136 |
+
#loc37 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":154:12)
|
| 137 |
+
#loc38 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":155:35)
|
| 138 |
+
#loc39 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":155:69)
|
| 139 |
+
#loc40 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":43:54)
|
| 140 |
+
#loc41 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":44:66)
|
| 141 |
+
#loc42 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":44:8)
|
| 142 |
+
#loc43 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":165:42)
|
| 143 |
+
#loc44 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":46:20)
|
| 144 |
+
#loc45 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":47:25)
|
| 145 |
+
#loc46 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":47:36)
|
| 146 |
+
#loc47 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/eu/ceui6qrb2t3lmzs3ljrqtcomt4b2q6svzo24j6mmryaiovr6kp7y.py":47:4)
|
| 147 |
+
#loc55 = loc("_tmp2_index"(#loc4))
|
| 148 |
+
#loc56 = loc("_tmp2"(#loc5))
|
| 149 |
+
#loc57 = loc("xoffset"(#loc6))
|
| 150 |
+
#loc58 = loc("xoffset"(#loc7))
|
| 151 |
+
#loc59 = loc("xindex"(#loc8))
|
| 152 |
+
#loc60 = loc("xindex"(#loc9))
|
| 153 |
+
#loc61 = loc("xindex"(#loc10))
|
| 154 |
+
#loc62 = loc("xmask"(#loc11))
|
| 155 |
+
#loc63 = loc("r0_base"(#loc12))
|
| 156 |
+
#loc64 = loc("r0_base"(#loc13))
|
| 157 |
+
#loc65 = loc("x0"(#loc14))
|
| 158 |
+
#loc66 = loc("x1"(#loc15))
|
| 159 |
+
#loc67 = loc("_tmp2"(#loc3))
|
| 160 |
+
#loc68 = loc("r0_index"(#loc16))
|
| 161 |
+
#loc69 = loc("r0_mask"(#loc17))
|
| 162 |
+
#loc70 = loc("tmp0"(#loc18))
|
| 163 |
+
#loc71 = loc("tmp0"(#loc19))
|
| 164 |
+
#loc72 = loc("tmp0"(#loc20))
|
| 165 |
+
#loc73 = loc("tmp0"(#loc21))
|
| 166 |
+
#loc74 = loc("tmp0"(#loc22))
|
| 167 |
+
#loc75 = loc("tmp0"(#loc23))
|
| 168 |
+
#loc76 = loc("tmp0"(#loc24))
|
| 169 |
+
#loc77 = loc("mask"(#loc25))
|
| 170 |
+
#loc78 = loc("equal"(#loc27))
|
| 171 |
+
#loc79 = loc("a_isnan"(#loc28))
|
| 172 |
+
#loc80 = loc("b_isnan"(#loc29))
|
| 173 |
+
#loc81 = loc("mask"(#loc30))
|
| 174 |
+
#loc82 = loc("mask"(#loc31))
|
| 175 |
+
#loc83 = loc("mask"(#loc32))
|
| 176 |
+
#loc84 = loc("equal"(#loc33))
|
| 177 |
+
#loc85 = loc("equal"(#loc34))
|
| 178 |
+
#loc86 = loc("mask"(#loc35))
|
| 179 |
+
#loc87 = loc("mask"(#loc36))
|
| 180 |
+
#loc88 = loc("mask"(#loc37))
|
| 181 |
+
#loc89 = loc(callsite(#loc38 at #loc26))
|
| 182 |
+
#loc90 = loc(callsite(#loc39 at #loc26))
|
| 183 |
+
#loc91 = loc("_tmp2"(#loc40))
|
| 184 |
+
#loc92 = loc("_tmp2_index"(#loc41))
|
| 185 |
+
#loc93 = loc(callsite(#loc43 at #loc2))
|
| 186 |
+
#loc94 = loc("tmp2"(#loc44))
|
| 187 |
+
#loc95 = loc("_tmp2_index"(#loc67))
|
| 188 |
+
#loc96 = loc("mask"(#loc77))
|
| 189 |
+
#loc97 = loc("equal"(#loc78))
|
| 190 |
+
#loc98 = loc(callsite(#loc79 at #loc26))
|
| 191 |
+
#loc99 = loc(callsite(#loc80 at #loc26))
|
| 192 |
+
#loc100 = loc(callsite(#loc81 at #loc26))
|
| 193 |
+
#loc101 = loc(callsite(#loc82 at #loc26))
|
| 194 |
+
#loc102 = loc("mask"(#loc83))
|
| 195 |
+
#loc103 = loc(callsite(#loc84 at #loc26))
|
| 196 |
+
#loc104 = loc("equal"(#loc85))
|
| 197 |
+
#loc105 = loc(callsite(#loc86 at #loc26))
|
| 198 |
+
#loc106 = loc(callsite(#loc87 at #loc26))
|
| 199 |
+
#loc107 = loc(callsite(#loc88 at #loc26))
|
| 200 |
+
#loc108 = loc(callsite(#loc79 at #loc93))
|
| 201 |
+
#loc109 = loc(callsite(#loc80 at #loc93))
|
| 202 |
+
#loc110 = loc(callsite(#loc81 at #loc93))
|
| 203 |
+
#loc111 = loc(callsite(#loc82 at #loc93))
|
| 204 |
+
#loc112 = loc(callsite(#loc84 at #loc93))
|
| 205 |
+
#loc113 = loc(callsite(#loc86 at #loc93))
|
| 206 |
+
#loc114 = loc(callsite(#loc87 at #loc93))
|
| 207 |
+
#loc115 = loc(callsite(#loc88 at #loc93))
|
| 208 |
+
#loc116 = loc(callsite(#loc38 at #loc93))
|
| 209 |
+
#loc117 = loc(callsite(#loc39 at #loc93))
|
| 210 |
+
#loc118 = loc(callsite(#loc96 at #loc26))
|
| 211 |
+
#loc119 = loc(callsite(#loc97 at #loc26))
|
| 212 |
+
#loc120 = loc(callsite(#loc102 at #loc26))
|
| 213 |
+
#loc121 = loc(callsite(#loc104 at #loc26))
|
| 214 |
+
#loc122 = loc(callsite(#loc96 at #loc93))
|
| 215 |
+
#loc123 = loc(callsite(#loc97 at #loc93))
|
| 216 |
+
#loc124 = loc(callsite(#loc102 at #loc93))
|
| 217 |
+
#loc125 = loc(callsite(#loc104 at #loc93))
|
SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/__grp__triton_red_fused__to_copy_clone_slice_sum_transpose_5.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"child_paths": {"triton_red_fused__to_copy_clone_slice_sum_transpose_5.source": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.source", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttir", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttgir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttgir", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.llir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.llir", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.ptx": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ptx", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.cubin": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.cubin", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.json": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.json"}}
|
SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.cubin
ADDED
|
Binary file (17.2 kB). View file
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"hash": "d1c2e6527ce27b628a96c1a250025b39aad19d679fce295e820390aa7ae64b66", "target": {"backend": "cuda", "arch": 90, "warp_size": 32}, "num_warps": 4, "num_ctas": 1, "num_stages": 1, "warp_size": 32, "maxnreg": null, "cluster_dims": [1, 1, 1], "ptx_version": null, "ptx_options": null, "ir_override": null, "enable_fp_fusion": true, "launch_cooperative_grid": false, "launch_pdl": false, "supported_fp8_dtypes": ["fp8e4b15", "fp8e4nv", "fp8e5"], "deprecated_fp8_dot_operand_dtypes": ["fp8e4b15"], "default_dot_input_precision": "tf32", "allowed_dot_input_precisions": ["tf32", "tf32x3", "ieee"], "max_num_imprecise_acc_default": 1073741824, "extern_libs": [["libdevice", "/workspace/specforge/lib/python3.11/site-packages/triton/backends/nvidia/lib/libdevice.10.bc"]], "debug": true, "backend_name": "cuda", "sanitize_overflow": false, "arch": "sm90", "instrumentation_mode": "", "triton_version": "3.5.1", "tensordesc_meta": [], "shared": 1024, "tmem_size": 0, "global_scratch_size": 0, "global_scratch_align": 1, "profile_scratch_size": 0, "profile_scratch_align": 1, "name": "triton_red_fused__to_copy_clone_slice_sum_transpose_5"}
|
SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.llir
ADDED
|
@@ -0,0 +1,204 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
; ModuleID = 'LLVMDialectModule'
|
| 2 |
+
source_filename = "LLVMDialectModule"
|
| 3 |
+
target datalayout = "e-p3:32:32-p4:32:32-p5:32:32-p6:32:32-p7:32:32-i64:64-i128:128-v16:16-v32:32-n16:32:64"
|
| 4 |
+
|
| 5 |
+
@global_smem = external addrspace(3) global [0 x i8], align 16
|
| 6 |
+
|
| 7 |
+
; Function Attrs: nounwind
|
| 8 |
+
define ptx_kernel void @triton_red_fused__to_copy_clone_slice_sum_transpose_5(ptr addrspace(1) %0, ptr addrspace(1) %1, i64 %2, i64 %3, i32 %4, i32 %5, ptr addrspace(1) readnone captures(none) %6, ptr addrspace(1) readnone captures(none) %7) local_unnamed_addr #0 !dbg !4 {
|
| 9 |
+
%9 = tail call i32 @llvm.nvvm.read.ptx.sreg.ctaid.x(), !dbg !7
|
| 10 |
+
%10 = shl i32 %9, 5, !dbg !8
|
| 11 |
+
%11 = tail call i32 @llvm.nvvm.read.ptx.sreg.tid.x(), !dbg !9
|
| 12 |
+
%12 = and i32 %11, 31, !dbg !9
|
| 13 |
+
%13 = or disjoint i32 %10, %12, !dbg !10
|
| 14 |
+
%14 = icmp slt i32 %13, %4, !dbg !11
|
| 15 |
+
%15 = lshr i32 %11, 5, !dbg !12
|
| 16 |
+
%16 = and i32 %15, 3, !dbg !12
|
| 17 |
+
%17 = sext i32 %13 to i64, !dbg !13
|
| 18 |
+
%.frozen = freeze i64 %2, !dbg !14
|
| 19 |
+
%18 = sdiv i64 %17, %.frozen, !dbg !14
|
| 20 |
+
%19 = mul i64 %18, %.frozen, !dbg !13
|
| 21 |
+
%.decomposed = sub i64 %17, %19, !dbg !13
|
| 22 |
+
%20 = icmp sgt i32 %5, 0, !dbg !15
|
| 23 |
+
br i1 %20, label %.lr.ph, label %._crit_edge, !dbg !15
|
| 24 |
+
|
| 25 |
+
.lr.ph: ; preds = %8
|
| 26 |
+
%21 = mul i64 %3, %2, !dbg !16
|
| 27 |
+
%22 = mul i64 %21, %18, !dbg !17
|
| 28 |
+
%23 = getelementptr i32, ptr addrspace(1) %0, i64 %.decomposed
|
| 29 |
+
%invariant.gep = getelementptr i32, ptr addrspace(1) %23, i64 %22, !dbg !15
|
| 30 |
+
%24 = insertelement <4 x i1> poison, i1 %14, i64 0, !dbg !18
|
| 31 |
+
%25 = shufflevector <4 x i1> %24, <4 x i1> poison, <4 x i32> zeroinitializer, !dbg !18
|
| 32 |
+
%26 = insertelement <4 x i32> poison, i32 %5, i64 0, !dbg !19
|
| 33 |
+
%27 = shufflevector <4 x i32> %26, <4 x i32> poison, <4 x i32> zeroinitializer, !dbg !19
|
| 34 |
+
br label %28, !dbg !15
|
| 35 |
+
|
| 36 |
+
28: ; preds = %.lr.ph, %28
|
| 37 |
+
%29 = phi i32 [ 0, %.lr.ph ], [ %68, %28 ]
|
| 38 |
+
%30 = phi <4 x i64> [ zeroinitializer, %.lr.ph ], [ %67, %28 ]
|
| 39 |
+
%31 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #5, !dbg !20
|
| 40 |
+
%32 = or disjoint i32 %29, %16, !dbg !21
|
| 41 |
+
%33 = or disjoint i32 %32, 4, !dbg !21
|
| 42 |
+
%34 = or disjoint i32 %32, 8, !dbg !21
|
| 43 |
+
%35 = or disjoint i32 %32, 12, !dbg !21
|
| 44 |
+
%36 = insertelement <4 x i32> poison, i32 %32, i64 0, !dbg !19
|
| 45 |
+
%37 = insertelement <4 x i32> %36, i32 %33, i64 1, !dbg !19
|
| 46 |
+
%38 = insertelement <4 x i32> %37, i32 %34, i64 2, !dbg !19
|
| 47 |
+
%39 = insertelement <4 x i32> %38, i32 %35, i64 3, !dbg !19
|
| 48 |
+
%40 = icmp slt <4 x i32> %39, %27, !dbg !19
|
| 49 |
+
%41 = sext i32 %32 to i64, !dbg !22
|
| 50 |
+
%42 = sext i32 %33 to i64, !dbg !22
|
| 51 |
+
%43 = sext i32 %34 to i64, !dbg !22
|
| 52 |
+
%44 = sext i32 %35 to i64, !dbg !22
|
| 53 |
+
%45 = mul i64 %2, %41, !dbg !22
|
| 54 |
+
%46 = mul i64 %2, %42, !dbg !22
|
| 55 |
+
%47 = mul i64 %2, %43, !dbg !22
|
| 56 |
+
%48 = mul i64 %2, %44, !dbg !22
|
| 57 |
+
%gep = getelementptr i32, ptr addrspace(1) %invariant.gep, i64 %45, !dbg !23
|
| 58 |
+
%gep4 = getelementptr i32, ptr addrspace(1) %invariant.gep, i64 %46, !dbg !23
|
| 59 |
+
%gep6 = getelementptr i32, ptr addrspace(1) %invariant.gep, i64 %47, !dbg !23
|
| 60 |
+
%gep8 = getelementptr i32, ptr addrspace(1) %invariant.gep, i64 %48, !dbg !23
|
| 61 |
+
%49 = and <4 x i1> %25, %40, !dbg !18
|
| 62 |
+
%50 = extractelement <4 x i1> %49, i64 0, !dbg !20
|
| 63 |
+
%51 = tail call i32 asm sideeffect "mov.u32 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b32 { $0 }, [ $1 + 0 ], $2;", "=r,l,l,b"(ptr addrspace(1) %gep, i64 %31, i1 %50) #5, !dbg !20
|
| 64 |
+
%52 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #5, !dbg !20
|
| 65 |
+
%53 = extractelement <4 x i1> %49, i64 1, !dbg !20
|
| 66 |
+
%54 = tail call i32 asm sideeffect "mov.u32 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b32 { $0 }, [ $1 + 0 ], $2;", "=r,l,l,b"(ptr addrspace(1) %gep4, i64 %52, i1 %53) #5, !dbg !20
|
| 67 |
+
%55 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #5, !dbg !20
|
| 68 |
+
%56 = extractelement <4 x i1> %49, i64 2, !dbg !20
|
| 69 |
+
%57 = tail call i32 asm sideeffect "mov.u32 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b32 { $0 }, [ $1 + 0 ], $2;", "=r,l,l,b"(ptr addrspace(1) %gep6, i64 %55, i1 %56) #5, !dbg !20
|
| 70 |
+
%58 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #5, !dbg !20
|
| 71 |
+
%59 = extractelement <4 x i1> %49, i64 3, !dbg !20
|
| 72 |
+
%60 = tail call i32 asm sideeffect "mov.u32 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b32 { $0 }, [ $1 + 0 ], $2;", "=r,l,l,b"(ptr addrspace(1) %gep8, i64 %58, i1 %59) #5, !dbg !20
|
| 73 |
+
%61 = insertelement <4 x i32> poison, i32 %51, i64 0, !dbg !24
|
| 74 |
+
%62 = insertelement <4 x i32> %61, i32 %54, i64 1, !dbg !24
|
| 75 |
+
%63 = insertelement <4 x i32> %62, i32 %57, i64 2, !dbg !24
|
| 76 |
+
%64 = insertelement <4 x i32> %63, i32 %60, i64 3, !dbg !24
|
| 77 |
+
%65 = sext <4 x i32> %64 to <4 x i64>, !dbg !24
|
| 78 |
+
%66 = select <4 x i1> %49, <4 x i64> %65, <4 x i64> zeroinitializer, !dbg !25
|
| 79 |
+
%67 = add <4 x i64> %66, %30, !dbg !25
|
| 80 |
+
%68 = add i32 %29, 16, !dbg !15
|
| 81 |
+
%69 = icmp slt i32 %68, %5, !dbg !15
|
| 82 |
+
br i1 %69, label %28, label %._crit_edge.loopexit, !dbg !15
|
| 83 |
+
|
| 84 |
+
._crit_edge.loopexit: ; preds = %28
|
| 85 |
+
%70 = tail call i64 @llvm.vector.reduce.add.v4i64(<4 x i64> %67), !dbg !26
|
| 86 |
+
br label %._crit_edge, !dbg !26
|
| 87 |
+
|
| 88 |
+
._crit_edge: ; preds = %._crit_edge.loopexit, %8
|
| 89 |
+
%71 = phi i64 [ 0, %8 ], [ %70, %._crit_edge.loopexit ], !dbg !26
|
| 90 |
+
%.idx = shl nuw nsw i32 %12, 5, !dbg !30
|
| 91 |
+
%72 = getelementptr i8, ptr addrspace(3) @global_smem, i32 %.idx, !dbg !30
|
| 92 |
+
%73 = getelementptr i64, ptr addrspace(3) %72, i32 %16, !dbg !30
|
| 93 |
+
%74 = insertelement <1 x i64> poison, i64 %71, i64 0, !dbg !30
|
| 94 |
+
tail call void asm sideeffect "@$2 st.shared.b64 [ $0 + 0 ], $1;", "r,l,b"(ptr addrspace(3) %73, <1 x i64> %74, i1 true) #5, !dbg !30
|
| 95 |
+
tail call void @llvm.nvvm.barrier.cta.sync.aligned.all(i32 0), !dbg !30
|
| 96 |
+
%75 = icmp samesign ult i32 %11, 128, !dbg !30
|
| 97 |
+
%76 = getelementptr i64, ptr addrspace(3) @global_smem, i32 %11, !dbg !30
|
| 98 |
+
%77 = tail call i64 asm sideeffect "@$2 ld.shared.b64 $0, [ $1 + 0 ];", "=l,r,b"(ptr addrspace(3) %76, i1 %75) #5, !dbg !30
|
| 99 |
+
%extelt.offset = lshr i64 %77, 32, !dbg !30
|
| 100 |
+
%78 = trunc nuw i64 %extelt.offset to i32, !dbg !30
|
| 101 |
+
%79 = trunc i64 %77 to i32, !dbg !30
|
| 102 |
+
%80 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %79, i32 2, i32 31), !dbg !30
|
| 103 |
+
%81 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %78, i32 2, i32 31), !dbg !30
|
| 104 |
+
%82 = insertelement <2 x i32> poison, i32 %80, i64 0, !dbg !30
|
| 105 |
+
%83 = insertelement <2 x i32> %82, i32 %81, i64 1, !dbg !30
|
| 106 |
+
%84 = bitcast <2 x i32> %83 to i64, !dbg !30
|
| 107 |
+
%85 = add i64 %77, %84, !dbg !26
|
| 108 |
+
%extelt.offset2 = lshr i64 %85, 32, !dbg !30
|
| 109 |
+
%86 = trunc nuw i64 %extelt.offset2 to i32, !dbg !30
|
| 110 |
+
%87 = trunc i64 %85 to i32, !dbg !30
|
| 111 |
+
%88 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %87, i32 1, i32 31), !dbg !30
|
| 112 |
+
%89 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %86, i32 1, i32 31), !dbg !30
|
| 113 |
+
%90 = insertelement <2 x i32> poison, i32 %88, i64 0, !dbg !30
|
| 114 |
+
%91 = insertelement <2 x i32> %90, i32 %89, i64 1, !dbg !30
|
| 115 |
+
%92 = bitcast <2 x i32> %91 to i64, !dbg !30
|
| 116 |
+
%93 = add i64 %85, %92, !dbg !26
|
| 117 |
+
%94 = and i32 %11, 899, !dbg !30
|
| 118 |
+
%95 = icmp eq i32 %94, 0, !dbg !30
|
| 119 |
+
%96 = insertelement <1 x i64> poison, i64 %93, i64 0, !dbg !30
|
| 120 |
+
tail call void asm sideeffect "@$2 st.shared.b64 [ $0 + 0 ], $1;", "r,l,b"(ptr addrspace(3) %76, <1 x i64> %96, i1 %95) #5, !dbg !30
|
| 121 |
+
tail call void @llvm.nvvm.barrier.cta.sync.aligned.all(i32 0), !dbg !30
|
| 122 |
+
%97 = load i64, ptr addrspace(3) %72, align 16, !dbg !30
|
| 123 |
+
%98 = trunc i64 %97 to i32, !dbg !31
|
| 124 |
+
%99 = icmp slt i64 %2, 2, !dbg !32
|
| 125 |
+
%100 = icmp sgt i64 %2, 1, !dbg !33
|
| 126 |
+
%101 = select i1 %100, i64 %2, i64 0, !dbg !34
|
| 127 |
+
%102 = zext i1 %99 to i64, !dbg !35
|
| 128 |
+
%103 = add i64 %101, %102, !dbg !36
|
| 129 |
+
%104 = mul i64 %18, %103, !dbg !37
|
| 130 |
+
%105 = getelementptr i32, ptr addrspace(1) %1, i64 %.decomposed, !dbg !38
|
| 131 |
+
%106 = getelementptr i32, ptr addrspace(1) %105, i64 %104, !dbg !38
|
| 132 |
+
%107 = and i32 %11, 96, !dbg !39
|
| 133 |
+
%108 = icmp eq i32 %107, 0, !dbg !39
|
| 134 |
+
%109 = and i1 %108, %14, !dbg !39
|
| 135 |
+
tail call void asm sideeffect "@$2 st.global.b32 [ $1 + 0 ], { $0 };", "r,l,b"(i32 %98, ptr addrspace(1) %106, i1 %109) #5, !dbg !39
|
| 136 |
+
ret void, !dbg !40
|
| 137 |
+
}
|
| 138 |
+
|
| 139 |
+
; Function Attrs: mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none)
|
| 140 |
+
declare noundef range(i32 0, 2147483647) i32 @llvm.nvvm.read.ptx.sreg.ctaid.x() #1
|
| 141 |
+
|
| 142 |
+
; Function Attrs: mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none)
|
| 143 |
+
declare noundef range(i32 0, 1024) i32 @llvm.nvvm.read.ptx.sreg.tid.x() #1
|
| 144 |
+
|
| 145 |
+
; Function Attrs: convergent nocallback nounwind
|
| 146 |
+
declare void @llvm.nvvm.barrier.cta.sync.aligned.all(i32) #2
|
| 147 |
+
|
| 148 |
+
; Function Attrs: convergent nocallback nounwind memory(inaccessiblemem: readwrite)
|
| 149 |
+
declare i32 @llvm.nvvm.shfl.sync.bfly.i32(i32, i32, i32, i32) #3
|
| 150 |
+
|
| 151 |
+
; Function Attrs: nocallback nofree nosync nounwind speculatable willreturn memory(none)
|
| 152 |
+
declare i64 @llvm.vector.reduce.add.v4i64(<4 x i64>) #4
|
| 153 |
+
|
| 154 |
+
attributes #0 = { nounwind "nvvm.reqntid"="128" }
|
| 155 |
+
attributes #1 = { mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none) }
|
| 156 |
+
attributes #2 = { convergent nocallback nounwind }
|
| 157 |
+
attributes #3 = { convergent nocallback nounwind memory(inaccessiblemem: readwrite) }
|
| 158 |
+
attributes #4 = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
|
| 159 |
+
attributes #5 = { nounwind }
|
| 160 |
+
|
| 161 |
+
!llvm.dbg.cu = !{!0}
|
| 162 |
+
!llvm.module.flags = !{!2, !3}
|
| 163 |
+
|
| 164 |
+
!0 = distinct !DICompileUnit(language: DW_LANG_C, file: !1, producer: "triton", isOptimized: true, runtimeVersion: 0, emissionKind: LineTablesOnly)
|
| 165 |
+
!1 = !DIFile(filename: "cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py", directory: "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc")
|
| 166 |
+
!2 = !{i32 2, !"Debug Info Version", i32 3}
|
| 167 |
+
!3 = !{i32 4, !"nvvm-reflect-ftz", i32 1}
|
| 168 |
+
!4 = distinct !DISubprogram(name: "triton_red_fused__to_copy_clone_slice_sum_transpose_5", linkageName: "triton_red_fused__to_copy_clone_slice_sum_transpose_5", scope: !1, file: !1, line: 18, type: !5, scopeLine: 18, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
|
| 169 |
+
!5 = !DISubroutineType(cc: DW_CC_normal, types: !6)
|
| 170 |
+
!6 = !{}
|
| 171 |
+
!7 = !DILocation(line: 21, column: 28, scope: !4)
|
| 172 |
+
!8 = !DILocation(line: 21, column: 33, scope: !4)
|
| 173 |
+
!9 = !DILocation(line: 22, column: 44, scope: !4)
|
| 174 |
+
!10 = !DILocation(line: 22, column: 23, scope: !4)
|
| 175 |
+
!11 = !DILocation(line: 23, column: 21, scope: !4)
|
| 176 |
+
!12 = !DILocation(line: 24, column: 37, scope: !4)
|
| 177 |
+
!13 = !DILocation(line: 26, column: 19, scope: !4)
|
| 178 |
+
!14 = !DILocation(line: 27, column: 19, scope: !4)
|
| 179 |
+
!15 = !DILocation(line: 30, column: 40, scope: !4)
|
| 180 |
+
!16 = !DILocation(line: 36, column: 54, scope: !4)
|
| 181 |
+
!17 = !DILocation(line: 36, column: 58, scope: !4)
|
| 182 |
+
!18 = !DILocation(line: 36, column: 73, scope: !4)
|
| 183 |
+
!19 = !DILocation(line: 32, column: 29, scope: !4)
|
| 184 |
+
!20 = !DILocation(line: 36, column: 63, scope: !4)
|
| 185 |
+
!21 = !DILocation(line: 31, column: 31, scope: !4)
|
| 186 |
+
!22 = !DILocation(line: 36, column: 43, scope: !4)
|
| 187 |
+
!23 = !DILocation(line: 36, column: 34, scope: !4)
|
| 188 |
+
!24 = !DILocation(line: 37, column: 23, scope: !4)
|
| 189 |
+
!25 = !DILocation(line: 40, column: 48, scope: !4)
|
| 190 |
+
!26 = !DILocation(line: 261, column: 15, scope: !27, inlinedAt: !29)
|
| 191 |
+
!27 = distinct !DILexicalBlockFile(scope: !4, file: !28, discriminator: 0)
|
| 192 |
+
!28 = !DIFile(filename: "standard.py", directory: "/workspace/specforge/lib/python3.11/site-packages/triton/language")
|
| 193 |
+
!29 = !DILocation(line: 41, column: 25, scope: !4)
|
| 194 |
+
!30 = !DILocation(line: 291, column: 36, scope: !27, inlinedAt: !29)
|
| 195 |
+
!31 = !DILocation(line: 42, column: 19, scope: !4)
|
| 196 |
+
!32 = !DILocation(line: 43, column: 49, scope: !4)
|
| 197 |
+
!33 = !DILocation(line: 43, column: 75, scope: !4)
|
| 198 |
+
!34 = !DILocation(line: 43, column: 66, scope: !4)
|
| 199 |
+
!35 = !DILocation(line: 43, scope: !4)
|
| 200 |
+
!36 = !DILocation(line: 43, column: 57, scope: !4)
|
| 201 |
+
!37 = !DILocation(line: 43, column: 34, scope: !4)
|
| 202 |
+
!38 = !DILocation(line: 43, column: 25, scope: !4)
|
| 203 |
+
!39 = !DILocation(line: 43, column: 88, scope: !4)
|
| 204 |
+
!40 = !DILocation(line: 43, column: 4, scope: !4)
|
SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ptx
ADDED
|
@@ -0,0 +1,525 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
//
|
| 2 |
+
// Generated by LLVM NVPTX Back-End
|
| 3 |
+
//
|
| 4 |
+
|
| 5 |
+
.version 8.7
|
| 6 |
+
.target sm_90a
|
| 7 |
+
.address_size 64
|
| 8 |
+
|
| 9 |
+
// .globl triton_red_fused__to_copy_clone_slice_sum_transpose_5 // -- Begin function triton_red_fused__to_copy_clone_slice_sum_transpose_5
|
| 10 |
+
.extern .shared .align 16 .b8 global_smem[];
|
| 11 |
+
// @triton_red_fused__to_copy_clone_slice_sum_transpose_5
|
| 12 |
+
.visible .entry triton_red_fused__to_copy_clone_slice_sum_transpose_5(
|
| 13 |
+
.param .u64 .ptr .global .align 1 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_0,
|
| 14 |
+
.param .u64 .ptr .global .align 1 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_1,
|
| 15 |
+
.param .u64 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_2,
|
| 16 |
+
.param .u64 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_3,
|
| 17 |
+
.param .u32 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_4,
|
| 18 |
+
.param .u32 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_5,
|
| 19 |
+
.param .u64 .ptr .global .align 1 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_6,
|
| 20 |
+
.param .u64 .ptr .global .align 1 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_7
|
| 21 |
+
)
|
| 22 |
+
.reqntid 128
|
| 23 |
+
{
|
| 24 |
+
.reg .pred %p<24>;
|
| 25 |
+
.reg .b32 %r<51>;
|
| 26 |
+
.reg .b64 %rd<97>;
|
| 27 |
+
.loc 1 18 0 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:18:0
|
| 28 |
+
$L__func_begin0:
|
| 29 |
+
.loc 1 18 0 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:18:0
|
| 30 |
+
|
| 31 |
+
// %bb.0:
|
| 32 |
+
ld.param.b32 %r11, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_5];
|
| 33 |
+
ld.param.b64 %rd20, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_2];
|
| 34 |
+
$L__tmp0:
|
| 35 |
+
.loc 1 21 28 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:21:28
|
| 36 |
+
mov.u32 %r12, %ctaid.x;
|
| 37 |
+
.loc 1 21 33 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:21:33
|
| 38 |
+
shl.b32 %r13, %r12, 5;
|
| 39 |
+
.loc 1 22 44 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:22:44
|
| 40 |
+
mov.u32 %r1, %tid.x;
|
| 41 |
+
and.b32 %r2, %r1, 31;
|
| 42 |
+
.loc 1 22 23 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:22:23
|
| 43 |
+
or.b32 %r14, %r13, %r2;
|
| 44 |
+
.loc 1 26 19 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:26:19
|
| 45 |
+
cvt.s64.s32 %rd1, %r14;
|
| 46 |
+
.loc 1 27 19 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:27:19
|
| 47 |
+
or.b64 %rd23, %rd1, %rd20;
|
| 48 |
+
and.b64 %rd24, %rd23, -4294967296;
|
| 49 |
+
setp.ne.b64 %p5, %rd24, 0;
|
| 50 |
+
cvt.u32.u64 %r49, %rd1;
|
| 51 |
+
@%p5 bra $L__BB0_2;
|
| 52 |
+
bra.uni $L__BB0_1;
|
| 53 |
+
$L__BB0_2:
|
| 54 |
+
div.s64 %rd91, %rd1, %rd20;
|
| 55 |
+
bra.uni $L__BB0_3;
|
| 56 |
+
$L__BB0_1:
|
| 57 |
+
cvt.u32.u64 %r15, %rd20;
|
| 58 |
+
div.u32 %r17, %r49, %r15;
|
| 59 |
+
cvt.u64.u32 %rd91, %r17;
|
| 60 |
+
$L__BB0_3:
|
| 61 |
+
.loc 1 0 19 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:0:19
|
| 62 |
+
ld.param.b32 %r10, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_4];
|
| 63 |
+
ld.param.b64 %rd19, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_1];
|
| 64 |
+
bfe.u32 %r3, %r1, 5, 2;
|
| 65 |
+
.loc 1 26 19 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:26:19
|
| 66 |
+
mul.lo.s64 %rd26, %rd91, %rd20;
|
| 67 |
+
sub.s64 %rd6, %rd1, %rd26;
|
| 68 |
+
.loc 1 30 40 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:30:40
|
| 69 |
+
setp.lt.s32 %p6, %r11, 1;
|
| 70 |
+
mov.b64 %rd96, 0;
|
| 71 |
+
shl.b64 %rd90, %rd6, 2;
|
| 72 |
+
@%p6 bra $L__BB0_7;
|
| 73 |
+
// %bb.4: // %.lr.ph
|
| 74 |
+
.loc 1 0 40 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:0:40
|
| 75 |
+
ld.param.b64 %rd21, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_3];
|
| 76 |
+
ld.param.b64 %rd18, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_0];
|
| 77 |
+
.loc 1 23 21 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:23:21
|
| 78 |
+
setp.lt.s32 %p1, %r49, %r10;
|
| 79 |
+
.loc 1 36 54 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:36:54
|
| 80 |
+
mul.lo.s64 %rd31, %rd21, %rd20;
|
| 81 |
+
.loc 1 36 58 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:36:58
|
| 82 |
+
mul.lo.s64 %rd32, %rd31, %rd91;
|
| 83 |
+
add.s64 %rd34, %rd18, %rd90;
|
| 84 |
+
.loc 1 30 40 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:30:40
|
| 85 |
+
shl.b64 %rd35, %rd32, 2;
|
| 86 |
+
add.s64 %rd7, %rd34, %rd35;
|
| 87 |
+
mov.b64 %rd92, 0;
|
| 88 |
+
mov.b32 %r50, 0;
|
| 89 |
+
mov.b64 %rd93, %rd92;
|
| 90 |
+
mov.b64 %rd94, %rd92;
|
| 91 |
+
mov.b64 %rd95, %rd92;
|
| 92 |
+
$L__BB0_5: // =>This Inner Loop Header: Depth=1
|
| 93 |
+
.loc 1 36 63 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:36:63
|
| 94 |
+
// begin inline asm
|
| 95 |
+
mov.u64 %rd36, 0x0;
|
| 96 |
+
createpolicy.fractional.L2::evict_last.b64 %rd36, 1.0;
|
| 97 |
+
// end inline asm
|
| 98 |
+
.loc 1 31 31 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:31:31
|
| 99 |
+
add.s32 %r24, %r3, %r50;
|
| 100 |
+
add.s32 %r25, %r24, 4;
|
| 101 |
+
add.s32 %r26, %r24, 8;
|
| 102 |
+
.loc 1 32 29 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:32:29
|
| 103 |
+
add.s32 %r27, %r24, 12;
|
| 104 |
+
setp.lt.s32 %p11, %r24, %r11;
|
| 105 |
+
setp.lt.s32 %p12, %r25, %r11;
|
| 106 |
+
setp.lt.s32 %p13, %r26, %r11;
|
| 107 |
+
setp.lt.s32 %p14, %r27, %r11;
|
| 108 |
+
.loc 1 36 43 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:36:43
|
| 109 |
+
cvt.s64.s32 %rd48, %r24;
|
| 110 |
+
cvt.s64.s32 %rd49, %r25;
|
| 111 |
+
cvt.s64.s32 %rd50, %r26;
|
| 112 |
+
cvt.s64.s32 %rd51, %r27;
|
| 113 |
+
mul.lo.s64 %rd52, %rd20, %rd48;
|
| 114 |
+
mul.lo.s64 %rd53, %rd20, %rd49;
|
| 115 |
+
mul.lo.s64 %rd54, %rd20, %rd50;
|
| 116 |
+
mul.lo.s64 %rd55, %rd20, %rd51;
|
| 117 |
+
.loc 1 36 34 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:36:34
|
| 118 |
+
shl.b64 %rd56, %rd52, 2;
|
| 119 |
+
add.s64 %rd37, %rd7, %rd56;
|
| 120 |
+
shl.b64 %rd57, %rd53, 2;
|
| 121 |
+
add.s64 %rd40, %rd7, %rd57;
|
| 122 |
+
shl.b64 %rd58, %rd54, 2;
|
| 123 |
+
add.s64 %rd43, %rd7, %rd58;
|
| 124 |
+
shl.b64 %rd59, %rd55, 2;
|
| 125 |
+
add.s64 %rd46, %rd7, %rd59;
|
| 126 |
+
.loc 1 36 73 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:36:73
|
| 127 |
+
and.pred %p10, %p1, %p14;
|
| 128 |
+
and.pred %p9, %p1, %p13;
|
| 129 |
+
and.pred %p8, %p1, %p12;
|
| 130 |
+
and.pred %p7, %p1, %p11;
|
| 131 |
+
.loc 1 36 63 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:36:63
|
| 132 |
+
// begin inline asm
|
| 133 |
+
mov.u32 %r20, 0x0;
|
| 134 |
+
@%p7 ld.global.L1::evict_last.L2::cache_hint.b32 { %r20 }, [ %rd37 + 0 ], %rd36;
|
| 135 |
+
// end inline asm
|
| 136 |
+
// begin inline asm
|
| 137 |
+
mov.u64 %rd39, 0x0;
|
| 138 |
+
createpolicy.fractional.L2::evict_last.b64 %rd39, 1.0;
|
| 139 |
+
// end inline asm
|
| 140 |
+
// begin inline asm
|
| 141 |
+
mov.u32 %r21, 0x0;
|
| 142 |
+
@%p8 ld.global.L1::evict_last.L2::cache_hint.b32 { %r21 }, [ %rd40 + 0 ], %rd39;
|
| 143 |
+
// end inline asm
|
| 144 |
+
// begin inline asm
|
| 145 |
+
mov.u64 %rd42, 0x0;
|
| 146 |
+
createpolicy.fractional.L2::evict_last.b64 %rd42, 1.0;
|
| 147 |
+
// end inline asm
|
| 148 |
+
// begin inline asm
|
| 149 |
+
mov.u32 %r22, 0x0;
|
| 150 |
+
@%p9 ld.global.L1::evict_last.L2::cache_hint.b32 { %r22 }, [ %rd43 + 0 ], %rd42;
|
| 151 |
+
// end inline asm
|
| 152 |
+
// begin inline asm
|
| 153 |
+
mov.u64 %rd45, 0x0;
|
| 154 |
+
createpolicy.fractional.L2::evict_last.b64 %rd45, 1.0;
|
| 155 |
+
// end inline asm
|
| 156 |
+
// begin inline asm
|
| 157 |
+
mov.u32 %r23, 0x0;
|
| 158 |
+
@%p10 ld.global.L1::evict_last.L2::cache_hint.b32 { %r23 }, [ %rd46 + 0 ], %rd45;
|
| 159 |
+
// end inline asm
|
| 160 |
+
.loc 1 37 23 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:37:23
|
| 161 |
+
cvt.s64.s32 %rd60, %r20;
|
| 162 |
+
cvt.s64.s32 %rd61, %r21;
|
| 163 |
+
cvt.s64.s32 %rd62, %r22;
|
| 164 |
+
cvt.s64.s32 %rd63, %r23;
|
| 165 |
+
.loc 1 40 48 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:40:48
|
| 166 |
+
selp.b64 %rd64, %rd63, 0, %p10;
|
| 167 |
+
selp.b64 %rd65, %rd62, 0, %p9;
|
| 168 |
+
selp.b64 %rd66, %rd61, 0, %p8;
|
| 169 |
+
selp.b64 %rd67, %rd60, 0, %p7;
|
| 170 |
+
add.s64 %rd92, %rd67, %rd92;
|
| 171 |
+
add.s64 %rd93, %rd66, %rd93;
|
| 172 |
+
add.s64 %rd94, %rd65, %rd94;
|
| 173 |
+
add.s64 %rd95, %rd64, %rd95;
|
| 174 |
+
.loc 1 30 40 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:30:40
|
| 175 |
+
add.s32 %r50, %r50, 16;
|
| 176 |
+
setp.lt.s32 %p15, %r50, %r11;
|
| 177 |
+
@%p15 bra $L__BB0_5;
|
| 178 |
+
// %bb.6: // %._crit_edge.loopexit
|
| 179 |
+
$L__tmp1:
|
| 180 |
+
.loc 2 261 15 // standard.py:261:15 @[ cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:41:25 ]
|
| 181 |
+
add.s64 %rd68, %rd92, %rd94;
|
| 182 |
+
add.s64 %rd69, %rd93, %rd95;
|
| 183 |
+
add.s64 %rd96, %rd68, %rd69;
|
| 184 |
+
$L__tmp2:
|
| 185 |
+
$L__BB0_7: // %._crit_edge
|
| 186 |
+
.loc 1 23 21 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:23:21
|
| 187 |
+
setp.lt.s32 %p20, %r49, %r10;
|
| 188 |
+
$L__tmp3:
|
| 189 |
+
.loc 2 291 36 // standard.py:291:36 @[ cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:41:25 ]
|
| 190 |
+
shl.b32 %r33, %r2, 5;
|
| 191 |
+
mov.b32 %r34, global_smem;
|
| 192 |
+
add.s32 %r35, %r34, %r33;
|
| 193 |
+
shl.b32 %r36, %r3, 3;
|
| 194 |
+
add.s32 %r28, %r35, %r36;
|
| 195 |
+
mov.pred %p16, -1;
|
| 196 |
+
// begin inline asm
|
| 197 |
+
@%p16 st.shared.b64 [ %r28 + 0 ], %rd96;
|
| 198 |
+
// end inline asm
|
| 199 |
+
bar.sync 0;
|
| 200 |
+
setp.lt.u32 %p17, %r1, 128;
|
| 201 |
+
shl.b32 %r37, %r1, 3;
|
| 202 |
+
add.s32 %r29, %r34, %r37;
|
| 203 |
+
// begin inline asm
|
| 204 |
+
@%p17 ld.shared.b64 %rd71, [ %r29 + 0 ];
|
| 205 |
+
// end inline asm
|
| 206 |
+
mov.b64 {_, %r38}, %rd71;
|
| 207 |
+
cvt.u32.u64 %r39, %rd71;
|
| 208 |
+
shfl.sync.bfly.b32 %r40, %r39, 2, 31, -1;
|
| 209 |
+
shfl.sync.bfly.b32 %r41, %r38, 2, 31, -1;
|
| 210 |
+
cvt.u64.u32 %rd74, %r40;
|
| 211 |
+
cvt.u64.u32 %rd75, %r41;
|
| 212 |
+
shl.b64 %rd76, %rd75, 32;
|
| 213 |
+
or.b64 %rd77, %rd74, %rd76;
|
| 214 |
+
.loc 2 261 15 // standard.py:261:15 @[ cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:41:25 ]
|
| 215 |
+
add.s64 %rd78, %rd71, %rd77;
|
| 216 |
+
.loc 2 291 36 // standard.py:291:36 @[ cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:41:25 ]
|
| 217 |
+
mov.b64 {_, %r42}, %rd78;
|
| 218 |
+
cvt.u32.u64 %r43, %rd78;
|
| 219 |
+
shfl.sync.bfly.b32 %r44, %r43, 1, 31, -1;
|
| 220 |
+
shfl.sync.bfly.b32 %r45, %r42, 1, 31, -1;
|
| 221 |
+
cvt.u64.u32 %rd79, %r44;
|
| 222 |
+
cvt.u64.u32 %rd80, %r45;
|
| 223 |
+
shl.b64 %rd81, %rd80, 32;
|
| 224 |
+
or.b64 %rd82, %rd79, %rd81;
|
| 225 |
+
.loc 2 261 15 // standard.py:261:15 @[ cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:41:25 ]
|
| 226 |
+
add.s64 %rd72, %rd78, %rd82;
|
| 227 |
+
.loc 2 291 36 // standard.py:291:36 @[ cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:41:25 ]
|
| 228 |
+
and.b32 %r46, %r1, 899;
|
| 229 |
+
setp.eq.b32 %p18, %r46, 0;
|
| 230 |
+
// begin inline asm
|
| 231 |
+
@%p18 st.shared.b64 [ %r29 + 0 ], %rd72;
|
| 232 |
+
// end inline asm
|
| 233 |
+
bar.sync 0;
|
| 234 |
+
ld.shared.b32 %r31, [%r35];
|
| 235 |
+
$L__tmp4:
|
| 236 |
+
.loc 1 43 49 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:43:49
|
| 237 |
+
setp.lt.s64 %p21, %rd20, 2;
|
| 238 |
+
.loc 1 43 75 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:43:75
|
| 239 |
+
setp.gt.s64 %p22, %rd20, 1;
|
| 240 |
+
.loc 1 43 66 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:43:66
|
| 241 |
+
selp.b64 %rd83, %rd20, 0, %p22;
|
| 242 |
+
.loc 1 43 0 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:43
|
| 243 |
+
selp.b64 %rd84, 1, 0, %p21;
|
| 244 |
+
.loc 1 43 57 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:43:57
|
| 245 |
+
add.s64 %rd85, %rd83, %rd84;
|
| 246 |
+
.loc 1 43 34 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:43:34
|
| 247 |
+
mul.lo.s64 %rd86, %rd91, %rd85;
|
| 248 |
+
.loc 1 43 25 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:43:25
|
| 249 |
+
add.s64 %rd88, %rd19, %rd90;
|
| 250 |
+
shl.b64 %rd89, %rd86, 2;
|
| 251 |
+
add.s64 %rd73, %rd88, %rd89;
|
| 252 |
+
.loc 1 43 88 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:43:88
|
| 253 |
+
and.b32 %r47, %r1, 96;
|
| 254 |
+
setp.eq.b32 %p23, %r47, 0;
|
| 255 |
+
and.pred %p19, %p23, %p20;
|
| 256 |
+
// begin inline asm
|
| 257 |
+
@%p19 st.global.b32 [ %rd73 + 0 ], { %r31 };
|
| 258 |
+
// end inline asm
|
| 259 |
+
.loc 1 43 4 // cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py:43:4
|
| 260 |
+
ret;
|
| 261 |
+
$L__tmp5:
|
| 262 |
+
$L__func_end0:
|
| 263 |
+
// -- End function
|
| 264 |
+
}
|
| 265 |
+
.file 1 "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py"
|
| 266 |
+
.file 2 "/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py"
|
| 267 |
+
.section .debug_abbrev
|
| 268 |
+
{
|
| 269 |
+
.b8 1 // Abbreviation Code
|
| 270 |
+
.b8 17 // DW_TAG_compile_unit
|
| 271 |
+
.b8 1 // DW_CHILDREN_yes
|
| 272 |
+
.b8 37 // DW_AT_producer
|
| 273 |
+
.b8 8 // DW_FORM_string
|
| 274 |
+
.b8 19 // DW_AT_language
|
| 275 |
+
.b8 5 // DW_FORM_data2
|
| 276 |
+
.b8 3 // DW_AT_name
|
| 277 |
+
.b8 8 // DW_FORM_string
|
| 278 |
+
.b8 16 // DW_AT_stmt_list
|
| 279 |
+
.b8 6 // DW_FORM_data4
|
| 280 |
+
.b8 27 // DW_AT_comp_dir
|
| 281 |
+
.b8 8 // DW_FORM_string
|
| 282 |
+
.b8 0 // EOM(1)
|
| 283 |
+
.b8 0 // EOM(2)
|
| 284 |
+
.b8 2 // Abbreviation Code
|
| 285 |
+
.b8 46 // DW_TAG_subprogram
|
| 286 |
+
.b8 0 // DW_CHILDREN_no
|
| 287 |
+
.b8 3 // DW_AT_name
|
| 288 |
+
.b8 8 // DW_FORM_string
|
| 289 |
+
.b8 32 // DW_AT_inline
|
| 290 |
+
.b8 11 // DW_FORM_data1
|
| 291 |
+
.b8 0 // EOM(1)
|
| 292 |
+
.b8 0 // EOM(2)
|
| 293 |
+
.b8 3 // Abbreviation Code
|
| 294 |
+
.b8 46 // DW_TAG_subprogram
|
| 295 |
+
.b8 1 // DW_CHILDREN_yes
|
| 296 |
+
.b8 17 // DW_AT_low_pc
|
| 297 |
+
.b8 1 // DW_FORM_addr
|
| 298 |
+
.b8 18 // DW_AT_high_pc
|
| 299 |
+
.b8 1 // DW_FORM_addr
|
| 300 |
+
.b8 49 // DW_AT_abstract_origin
|
| 301 |
+
.b8 19 // DW_FORM_ref4
|
| 302 |
+
.b8 0 // EOM(1)
|
| 303 |
+
.b8 0 // EOM(2)
|
| 304 |
+
.b8 4 // Abbreviation Code
|
| 305 |
+
.b8 29 // DW_TAG_inlined_subroutine
|
| 306 |
+
.b8 0 // DW_CHILDREN_no
|
| 307 |
+
.b8 49 // DW_AT_abstract_origin
|
| 308 |
+
.b8 19 // DW_FORM_ref4
|
| 309 |
+
.b8 17 // DW_AT_low_pc
|
| 310 |
+
.b8 1 // DW_FORM_addr
|
| 311 |
+
.b8 18 // DW_AT_high_pc
|
| 312 |
+
.b8 1 // DW_FORM_addr
|
| 313 |
+
.b8 88 // DW_AT_call_file
|
| 314 |
+
.b8 11 // DW_FORM_data1
|
| 315 |
+
.b8 89 // DW_AT_call_line
|
| 316 |
+
.b8 11 // DW_FORM_data1
|
| 317 |
+
.b8 87 // DW_AT_call_column
|
| 318 |
+
.b8 11 // DW_FORM_data1
|
| 319 |
+
.b8 0 // EOM(1)
|
| 320 |
+
.b8 0 // EOM(2)
|
| 321 |
+
.b8 0 // EOM(3)
|
| 322 |
+
}
|
| 323 |
+
.section .debug_info
|
| 324 |
+
{
|
| 325 |
+
.b32 238 // Length of Unit
|
| 326 |
+
.b8 2 // DWARF version number
|
| 327 |
+
.b8 0
|
| 328 |
+
.b32 .debug_abbrev // Offset Into Abbrev. Section
|
| 329 |
+
.b8 8 // Address Size (in bytes)
|
| 330 |
+
.b8 1 // Abbrev [1] 0xb:0xe7 DW_TAG_compile_unit
|
| 331 |
+
.b8 116 // DW_AT_producer
|
| 332 |
+
.b8 114
|
| 333 |
+
.b8 105
|
| 334 |
+
.b8 116
|
| 335 |
+
.b8 111
|
| 336 |
+
.b8 110
|
| 337 |
+
.b8 0
|
| 338 |
+
.b8 2 // DW_AT_language
|
| 339 |
+
.b8 0
|
| 340 |
+
.b8 99 // DW_AT_name
|
| 341 |
+
.b8 119
|
| 342 |
+
.b8 99
|
| 343 |
+
.b8 50
|
| 344 |
+
.b8 106
|
| 345 |
+
.b8 99
|
| 346 |
+
.b8 116
|
| 347 |
+
.b8 122
|
| 348 |
+
.b8 54
|
| 349 |
+
.b8 55
|
| 350 |
+
.b8 115
|
| 351 |
+
.b8 55
|
| 352 |
+
.b8 110
|
| 353 |
+
.b8 54
|
| 354 |
+
.b8 102
|
| 355 |
+
.b8 111
|
| 356 |
+
.b8 116
|
| 357 |
+
.b8 108
|
| 358 |
+
.b8 50
|
| 359 |
+
.b8 50
|
| 360 |
+
.b8 103
|
| 361 |
+
.b8 114
|
| 362 |
+
.b8 52
|
| 363 |
+
.b8 52
|
| 364 |
+
.b8 103
|
| 365 |
+
.b8 108
|
| 366 |
+
.b8 105
|
| 367 |
+
.b8 106
|
| 368 |
+
.b8 114
|
| 369 |
+
.b8 99
|
| 370 |
+
.b8 102
|
| 371 |
+
.b8 120
|
| 372 |
+
.b8 112
|
| 373 |
+
.b8 101
|
| 374 |
+
.b8 98
|
| 375 |
+
.b8 108
|
| 376 |
+
.b8 118
|
| 377 |
+
.b8 107
|
| 378 |
+
.b8 51
|
| 379 |
+
.b8 50
|
| 380 |
+
.b8 104
|
| 381 |
+
.b8 111
|
| 382 |
+
.b8 114
|
| 383 |
+
.b8 110
|
| 384 |
+
.b8 108
|
| 385 |
+
.b8 108
|
| 386 |
+
.b8 108
|
| 387 |
+
.b8 122
|
| 388 |
+
.b8 110
|
| 389 |
+
.b8 120
|
| 390 |
+
.b8 50
|
| 391 |
+
.b8 109
|
| 392 |
+
.b8 46
|
| 393 |
+
.b8 112
|
| 394 |
+
.b8 121
|
| 395 |
+
.b8 0
|
| 396 |
+
.b32 .debug_line // DW_AT_stmt_list
|
| 397 |
+
.b8 47 // DW_AT_comp_dir
|
| 398 |
+
.b8 119
|
| 399 |
+
.b8 111
|
| 400 |
+
.b8 114
|
| 401 |
+
.b8 107
|
| 402 |
+
.b8 115
|
| 403 |
+
.b8 112
|
| 404 |
+
.b8 97
|
| 405 |
+
.b8 99
|
| 406 |
+
.b8 101
|
| 407 |
+
.b8 47
|
| 408 |
+
.b8 104
|
| 409 |
+
.b8 97
|
| 410 |
+
.b8 110
|
| 411 |
+
.b8 114
|
| 412 |
+
.b8 117
|
| 413 |
+
.b8 105
|
| 414 |
+
.b8 47
|
| 415 |
+
.b8 83
|
| 416 |
+
.b8 112
|
| 417 |
+
.b8 101
|
| 418 |
+
.b8 99
|
| 419 |
+
.b8 70
|
| 420 |
+
.b8 111
|
| 421 |
+
.b8 114
|
| 422 |
+
.b8 103
|
| 423 |
+
.b8 101
|
| 424 |
+
.b8 45
|
| 425 |
+
.b8 101
|
| 426 |
+
.b8 120
|
| 427 |
+
.b8 116
|
| 428 |
+
.b8 47
|
| 429 |
+
.b8 99
|
| 430 |
+
.b8 97
|
| 431 |
+
.b8 99
|
| 432 |
+
.b8 104
|
| 433 |
+
.b8 101
|
| 434 |
+
.b8 47
|
| 435 |
+
.b8 99
|
| 436 |
+
.b8 111
|
| 437 |
+
.b8 109
|
| 438 |
+
.b8 112
|
| 439 |
+
.b8 105
|
| 440 |
+
.b8 108
|
| 441 |
+
.b8 101
|
| 442 |
+
.b8 100
|
| 443 |
+
.b8 95
|
| 444 |
+
.b8 107
|
| 445 |
+
.b8 101
|
| 446 |
+
.b8 114
|
| 447 |
+
.b8 110
|
| 448 |
+
.b8 101
|
| 449 |
+
.b8 108
|
| 450 |
+
.b8 115
|
| 451 |
+
.b8 47
|
| 452 |
+
.b8 119
|
| 453 |
+
.b8 99
|
| 454 |
+
.b8 0
|
| 455 |
+
.b8 2 // Abbrev [2] 0x8b:0x38 DW_TAG_subprogram
|
| 456 |
+
.b8 116 // DW_AT_name
|
| 457 |
+
.b8 114
|
| 458 |
+
.b8 105
|
| 459 |
+
.b8 116
|
| 460 |
+
.b8 111
|
| 461 |
+
.b8 110
|
| 462 |
+
.b8 95
|
| 463 |
+
.b8 114
|
| 464 |
+
.b8 101
|
| 465 |
+
.b8 100
|
| 466 |
+
.b8 95
|
| 467 |
+
.b8 102
|
| 468 |
+
.b8 117
|
| 469 |
+
.b8 115
|
| 470 |
+
.b8 101
|
| 471 |
+
.b8 100
|
| 472 |
+
.b8 95
|
| 473 |
+
.b8 95
|
| 474 |
+
.b8 116
|
| 475 |
+
.b8 111
|
| 476 |
+
.b8 95
|
| 477 |
+
.b8 99
|
| 478 |
+
.b8 111
|
| 479 |
+
.b8 112
|
| 480 |
+
.b8 121
|
| 481 |
+
.b8 95
|
| 482 |
+
.b8 99
|
| 483 |
+
.b8 108
|
| 484 |
+
.b8 111
|
| 485 |
+
.b8 110
|
| 486 |
+
.b8 101
|
| 487 |
+
.b8 95
|
| 488 |
+
.b8 115
|
| 489 |
+
.b8 108
|
| 490 |
+
.b8 105
|
| 491 |
+
.b8 99
|
| 492 |
+
.b8 101
|
| 493 |
+
.b8 95
|
| 494 |
+
.b8 115
|
| 495 |
+
.b8 117
|
| 496 |
+
.b8 109
|
| 497 |
+
.b8 95
|
| 498 |
+
.b8 116
|
| 499 |
+
.b8 114
|
| 500 |
+
.b8 97
|
| 501 |
+
.b8 110
|
| 502 |
+
.b8 115
|
| 503 |
+
.b8 112
|
| 504 |
+
.b8 111
|
| 505 |
+
.b8 115
|
| 506 |
+
.b8 101
|
| 507 |
+
.b8 95
|
| 508 |
+
.b8 53
|
| 509 |
+
.b8 0
|
| 510 |
+
.b8 1 // DW_AT_inline
|
| 511 |
+
.b8 3 // Abbrev [3] 0xc3:0x2e DW_TAG_subprogram
|
| 512 |
+
.b64 $L__func_begin0 // DW_AT_low_pc
|
| 513 |
+
.b64 $L__func_end0 // DW_AT_high_pc
|
| 514 |
+
.b32 139 // DW_AT_abstract_origin
|
| 515 |
+
.b8 4 // Abbrev [4] 0xd8:0x18 DW_TAG_inlined_subroutine
|
| 516 |
+
.b32 139 // DW_AT_abstract_origin
|
| 517 |
+
.b64 $L__tmp1 // DW_AT_low_pc
|
| 518 |
+
.b64 $L__tmp4 // DW_AT_high_pc
|
| 519 |
+
.b8 1 // DW_AT_call_file
|
| 520 |
+
.b8 41 // DW_AT_call_line
|
| 521 |
+
.b8 25 // DW_AT_call_column
|
| 522 |
+
.b8 0 // End Of Children Mark
|
| 523 |
+
.b8 0 // End Of Children Mark
|
| 524 |
+
}
|
| 525 |
+
.section .debug_macinfo { }
|
SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.source
ADDED
|
@@ -0,0 +1,193 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":18:0)
|
| 2 |
+
#loc41 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":285:0)
|
| 3 |
+
#loc43 = loc(unknown)
|
| 4 |
+
#loc46 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":260:0)
|
| 5 |
+
#loc50 = loc("in_ptr0"(#loc))
|
| 6 |
+
#loc51 = loc("out_ptr1"(#loc))
|
| 7 |
+
#loc52 = loc("ks0"(#loc))
|
| 8 |
+
#loc53 = loc("ks1"(#loc))
|
| 9 |
+
#loc54 = loc("xnumel"(#loc))
|
| 10 |
+
#loc55 = loc("r0_numel"(#loc))
|
| 11 |
+
#loc85 = loc("input"(#loc41))
|
| 12 |
+
#loc86 = loc("a"(#loc46))
|
| 13 |
+
#loc87 = loc("b"(#loc46))
|
| 14 |
+
module {
|
| 15 |
+
tt.func public @triton_red_fused__to_copy_clone_slice_sum_transpose_5(%in_ptr0: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %out_ptr1: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("out_ptr1"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %xnumel: i32 loc("xnumel"(#loc)), %r0_numel: i32 loc("r0_numel"(#loc))) attributes {noinline = false} {
|
| 16 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc56)
|
| 17 |
+
%xoffset_0 = arith.constant 32 : i32 loc(#loc57)
|
| 18 |
+
%xoffset_1 = arith.constant 32 : i32 loc(#loc57)
|
| 19 |
+
%xoffset_2 = arith.muli %xoffset, %xoffset_1 : i32 loc(#loc57)
|
| 20 |
+
%xindex = tt.make_range {end = 32 : i32, start = 0 : i32} : tensor<32xi32> loc(#loc58)
|
| 21 |
+
%xindex_3 = tt.expand_dims %xindex {axis = 1 : i32} : tensor<32xi32> -> tensor<32x1xi32> loc(#loc59)
|
| 22 |
+
%xindex_4 = tt.splat %xoffset_2 : i32 -> tensor<32x1xi32> loc(#loc60)
|
| 23 |
+
%xindex_5 = arith.addi %xindex_4, %xindex_3 : tensor<32x1xi32> loc(#loc60)
|
| 24 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<32x1xi32> loc(#loc61)
|
| 25 |
+
%xmask_6 = arith.cmpi slt, %xindex_5, %xmask : tensor<32x1xi32> loc(#loc61)
|
| 26 |
+
%r0_base = tt.make_range {end = 16 : i32, start = 0 : i32} : tensor<16xi32> loc(#loc62)
|
| 27 |
+
%r0_base_7 = tt.expand_dims %r0_base {axis = 0 : i32} : tensor<16xi32> -> tensor<1x16xi32> loc(#loc63)
|
| 28 |
+
%x0 = arith.extsi %xindex_5 : tensor<32x1xi32> to tensor<32x1xi64> loc(#loc64)
|
| 29 |
+
%x0_8 = tt.splat %ks0 : i64 -> tensor<32x1xi64> loc(#loc64)
|
| 30 |
+
%x0_9 = arith.remsi %x0, %x0_8 : tensor<32x1xi64> loc(#loc64)
|
| 31 |
+
%x1 = arith.extsi %xindex_5 : tensor<32x1xi32> to tensor<32x1xi64> loc(#loc65)
|
| 32 |
+
%x1_10 = tt.splat %ks0 : i64 -> tensor<32x1xi64> loc(#loc65)
|
| 33 |
+
%x1_11 = arith.divsi %x1, %x1_10 : tensor<32x1xi64> loc(#loc65)
|
| 34 |
+
%_tmp3 = arith.constant 0 : i64 loc(#loc66)
|
| 35 |
+
%_tmp3_12 = arith.constant dense<0> : tensor<32x16xi64> loc(#loc66)
|
| 36 |
+
%c0_i32 = arith.constant 0 : i32 loc(#loc12)
|
| 37 |
+
%c16_i32 = arith.constant 16 : i32 loc(#loc12)
|
| 38 |
+
%0 = arith.bitcast %c0_i32 : i32 to i32 loc(#loc12)
|
| 39 |
+
%1 = arith.bitcast %r0_numel : i32 to i32 loc(#loc12)
|
| 40 |
+
%2 = arith.bitcast %c16_i32 : i32 to i32 loc(#loc12)
|
| 41 |
+
%3 = ub.poison : i32 loc(#loc12)
|
| 42 |
+
%_tmp3_13 = scf.for %r0_offset = %0 to %1 step %2 iter_args(%_tmp3_18 = %_tmp3_12) -> (tensor<32x16xi64>) : i32 {
|
| 43 |
+
%r0_index = tt.splat %r0_offset : i32 -> tensor<1x16xi32> loc(#loc68)
|
| 44 |
+
%r0_index_19 = arith.addi %r0_index, %r0_base_7 : tensor<1x16xi32> loc(#loc68)
|
| 45 |
+
%r0_mask = tt.splat %r0_numel : i32 -> tensor<1x16xi32> loc(#loc69)
|
| 46 |
+
%r0_mask_20 = arith.cmpi slt, %r0_index_19, %r0_mask : tensor<1x16xi32> loc(#loc69)
|
| 47 |
+
%tmp0 = arith.extsi %r0_index_19 : tensor<1x16xi32> to tensor<1x16xi64> loc(#loc70)
|
| 48 |
+
%tmp0_21 = tt.splat %ks0 : i64 -> tensor<1x16xi64> loc(#loc70)
|
| 49 |
+
%tmp0_22 = arith.muli %tmp0_21, %tmp0 : tensor<1x16xi64> loc(#loc70)
|
| 50 |
+
%tmp0_23 = tt.broadcast %x0_9 : tensor<32x1xi64> -> tensor<32x16xi64> loc(#loc71)
|
| 51 |
+
%tmp0_24 = tt.broadcast %tmp0_22 : tensor<1x16xi64> -> tensor<32x16xi64> loc(#loc71)
|
| 52 |
+
%tmp0_25 = arith.addi %tmp0_23, %tmp0_24 : tensor<32x16xi64> loc(#loc71)
|
| 53 |
+
%tmp0_26 = arith.muli %ks0, %ks1 : i64 loc(#loc72)
|
| 54 |
+
%tmp0_27 = tt.splat %tmp0_26 : i64 -> tensor<32x1xi64> loc(#loc73)
|
| 55 |
+
%tmp0_28 = arith.muli %tmp0_27, %x1_11 : tensor<32x1xi64> loc(#loc73)
|
| 56 |
+
%tmp0_29 = tt.broadcast %tmp0_28 : tensor<32x1xi64> -> tensor<32x16xi64> loc(#loc74)
|
| 57 |
+
%tmp0_30 = arith.addi %tmp0_25, %tmp0_29 : tensor<32x16xi64> loc(#loc74)
|
| 58 |
+
%tmp0_31 = tt.splat %in_ptr0 : !tt.ptr<i32> -> tensor<32x16x!tt.ptr<i32>> loc(#loc75)
|
| 59 |
+
%tmp0_32 = tt.addptr %tmp0_31, %tmp0_30 : tensor<32x16x!tt.ptr<i32>>, tensor<32x16xi64> loc(#loc75)
|
| 60 |
+
%tmp0_33 = tt.broadcast %r0_mask_20 : tensor<1x16xi1> -> tensor<32x16xi1> loc(#loc76)
|
| 61 |
+
%tmp0_34 = tt.broadcast %xmask_6 : tensor<32x1xi1> -> tensor<32x16xi1> loc(#loc76)
|
| 62 |
+
%tmp0_35 = arith.andi %tmp0_33, %tmp0_34 : tensor<32x16xi1> loc(#loc76)
|
| 63 |
+
%tmp0_36 = arith.constant 0.000000e+00 : f32 loc(#loc77)
|
| 64 |
+
%tmp0_37 = arith.constant dense<0.000000e+00> : tensor<32x16xf32> loc(#loc77)
|
| 65 |
+
%tmp0_38 = arith.fptosi %tmp0_37 : tensor<32x16xf32> to tensor<32x16xi32> loc(#loc77)
|
| 66 |
+
%tmp0_39 = tt.load %tmp0_32, %tmp0_35, %tmp0_38 evictionPolicy = evict_last : tensor<32x16x!tt.ptr<i32>> loc(#loc77)
|
| 67 |
+
%tmp1 = arith.extsi %tmp0_39 : tensor<32x16xi32> to tensor<32x16xi64> loc(#loc78)
|
| 68 |
+
%tmp4 = arith.addi %_tmp3_18, %tmp1 : tensor<32x16xi64> loc(#loc79)
|
| 69 |
+
%_tmp3_40 = tt.broadcast %r0_mask_20 : tensor<1x16xi1> -> tensor<32x16xi1> loc(#loc80)
|
| 70 |
+
%_tmp3_41 = tt.broadcast %xmask_6 : tensor<32x1xi1> -> tensor<32x16xi1> loc(#loc80)
|
| 71 |
+
%_tmp3_42 = arith.andi %_tmp3_40, %_tmp3_41 : tensor<32x16xi1> loc(#loc80)
|
| 72 |
+
%_tmp3_43 = arith.select %_tmp3_42, %tmp4, %_tmp3_18 : tensor<32x16xi1>, tensor<32x16xi64> loc(#loc81)
|
| 73 |
+
scf.yield %_tmp3_43 : tensor<32x16xi64> loc(#loc27)
|
| 74 |
+
} loc(#loc67)
|
| 75 |
+
%tmp3 = tt.call @"triton.language.standard.sum__i64S32_16S__(1,)cconstexpr_1__(2,)cconstexpr_False__(3,)cNone"(%_tmp3_13) : (tensor<32x16xi64>) -> tensor<32xi64> loc(#loc82)
|
| 76 |
+
%tmp3_14 = tt.expand_dims %tmp3 {axis = 1 : i32} : tensor<32xi64> -> tensor<32x1xi64> loc(#loc83)
|
| 77 |
+
%tmp5 = arith.trunci %tmp3_14 : tensor<32x1xi64> to tensor<32x1xi32> loc(#loc84)
|
| 78 |
+
%c1_i32 = arith.constant 1 : i32 loc(#loc31)
|
| 79 |
+
%4 = arith.extsi %c1_i32 : i32 to i64 loc(#loc31)
|
| 80 |
+
%5 = arith.cmpi sge, %4, %ks0 : i64 loc(#loc31)
|
| 81 |
+
%c1_i32_15 = arith.constant 1 : i32 loc(#loc32)
|
| 82 |
+
%c1_i32_16 = arith.constant 1 : i32 loc(#loc32)
|
| 83 |
+
%6 = arith.extui %5 : i1 to i32 loc(#loc32)
|
| 84 |
+
%7 = arith.muli %c1_i32_16, %6 : i32 loc(#loc32)
|
| 85 |
+
%c1_i32_17 = arith.constant 1 : i32 loc(#loc33)
|
| 86 |
+
%8 = arith.extsi %c1_i32_17 : i32 to i64 loc(#loc33)
|
| 87 |
+
%9 = arith.cmpi sgt, %ks0, %8 : i64 loc(#loc33)
|
| 88 |
+
%10 = arith.extui %9 : i1 to i64 loc(#loc34)
|
| 89 |
+
%11 = arith.muli %ks0, %10 : i64 loc(#loc34)
|
| 90 |
+
%12 = arith.extsi %7 : i32 to i64 loc(#loc35)
|
| 91 |
+
%13 = arith.addi %12, %11 : i64 loc(#loc35)
|
| 92 |
+
%14 = tt.splat %13 : i64 -> tensor<32x1xi64> loc(#loc36)
|
| 93 |
+
%15 = arith.muli %x1_11, %14 : tensor<32x1xi64> loc(#loc36)
|
| 94 |
+
%16 = arith.addi %x0_9, %15 : tensor<32x1xi64> loc(#loc37)
|
| 95 |
+
%17 = tt.splat %out_ptr1 : !tt.ptr<i32> -> tensor<32x1x!tt.ptr<i32>> loc(#loc38)
|
| 96 |
+
%18 = tt.addptr %17, %16 : tensor<32x1x!tt.ptr<i32>>, tensor<32x1xi64> loc(#loc38)
|
| 97 |
+
tt.store %18, %tmp5, %xmask_6 : tensor<32x1x!tt.ptr<i32>> loc(#loc39)
|
| 98 |
+
tt.return loc(#loc40)
|
| 99 |
+
} loc(#loc)
|
| 100 |
+
tt.func private @"triton.language.standard.sum__i64S32_16S__(1,)cconstexpr_1__(2,)cconstexpr_False__(3,)cNone"(%input: tensor<32x16xi64> loc("input"(#loc41))) -> tensor<32xi64> attributes {noinline = false} {
|
| 101 |
+
%0 = "tt.reduce"(%input) <{axis = 1 : i32}> ({
|
| 102 |
+
^bb0(%arg1: i64 loc(unknown), %arg2: i64 loc(unknown)):
|
| 103 |
+
%2 = tt.call @triton.language.standard._sum_combine__i64_i64__(%arg1, %arg2) : (i64, i64) -> i64 loc(#loc42)
|
| 104 |
+
tt.reduce.return %2 : i64 loc(#loc42)
|
| 105 |
+
}) : (tensor<32x16xi64>) -> tensor<32xi64> loc(#loc42)
|
| 106 |
+
tt.return %0 : tensor<32xi64> loc(#loc44)
|
| 107 |
+
^bb1: // no predecessors
|
| 108 |
+
%1 = ub.poison : tensor<32xi64> loc(#loc45)
|
| 109 |
+
tt.return %1 : tensor<32xi64> loc(#loc45)
|
| 110 |
+
} loc(#loc41)
|
| 111 |
+
tt.func private @triton.language.standard._sum_combine__i64_i64__(%a: i64 loc("a"(#loc46)), %b: i64 loc("b"(#loc46))) -> i64 attributes {noinline = false} {
|
| 112 |
+
%0 = arith.addi %a, %b : i64 loc(#loc47)
|
| 113 |
+
tt.return %0 : i64 loc(#loc48)
|
| 114 |
+
^bb1: // no predecessors
|
| 115 |
+
%1 = ub.poison : i64 loc(#loc49)
|
| 116 |
+
tt.return %1 : i64 loc(#loc49)
|
| 117 |
+
} loc(#loc46)
|
| 118 |
+
} loc(#loc)
|
| 119 |
+
#loc1 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":21:28)
|
| 120 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":21:33)
|
| 121 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":22:36)
|
| 122 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":22:44)
|
| 123 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":22:23)
|
| 124 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":23:21)
|
| 125 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":24:27)
|
| 126 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":24:37)
|
| 127 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":26:19)
|
| 128 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":27:19)
|
| 129 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":28:43)
|
| 130 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":30:40)
|
| 131 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":31:31)
|
| 132 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":32:29)
|
| 133 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:43)
|
| 134 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:39)
|
| 135 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:54)
|
| 136 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:58)
|
| 137 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:50)
|
| 138 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:34)
|
| 139 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:73)
|
| 140 |
+
#loc22 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:63)
|
| 141 |
+
#loc23 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":37:23)
|
| 142 |
+
#loc24 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":39:23)
|
| 143 |
+
#loc25 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":40:35)
|
| 144 |
+
#loc26 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":40:48)
|
| 145 |
+
#loc27 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":40:8)
|
| 146 |
+
#loc28 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":41:25)
|
| 147 |
+
#loc29 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":41:28)
|
| 148 |
+
#loc30 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":42:19)
|
| 149 |
+
#loc31 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:49)
|
| 150 |
+
#loc32 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:41)
|
| 151 |
+
#loc33 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:75)
|
| 152 |
+
#loc34 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:66)
|
| 153 |
+
#loc35 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:57)
|
| 154 |
+
#loc36 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:34)
|
| 155 |
+
#loc37 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:30)
|
| 156 |
+
#loc38 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:25)
|
| 157 |
+
#loc39 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:88)
|
| 158 |
+
#loc40 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:4)
|
| 159 |
+
#loc42 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:36)
|
| 160 |
+
#loc44 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:11)
|
| 161 |
+
#loc45 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:4)
|
| 162 |
+
#loc47 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:15)
|
| 163 |
+
#loc48 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:11)
|
| 164 |
+
#loc49 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:4)
|
| 165 |
+
#loc56 = loc("xoffset"(#loc1))
|
| 166 |
+
#loc57 = loc("xoffset"(#loc2))
|
| 167 |
+
#loc58 = loc("xindex"(#loc3))
|
| 168 |
+
#loc59 = loc("xindex"(#loc4))
|
| 169 |
+
#loc60 = loc("xindex"(#loc5))
|
| 170 |
+
#loc61 = loc("xmask"(#loc6))
|
| 171 |
+
#loc62 = loc("r0_base"(#loc7))
|
| 172 |
+
#loc63 = loc("r0_base"(#loc8))
|
| 173 |
+
#loc64 = loc("x0"(#loc9))
|
| 174 |
+
#loc65 = loc("x1"(#loc10))
|
| 175 |
+
#loc66 = loc("_tmp3"(#loc11))
|
| 176 |
+
#loc67 = loc("_tmp3"(#loc12))
|
| 177 |
+
#loc68 = loc("r0_index"(#loc13))
|
| 178 |
+
#loc69 = loc("r0_mask"(#loc14))
|
| 179 |
+
#loc70 = loc("tmp0"(#loc15))
|
| 180 |
+
#loc71 = loc("tmp0"(#loc16))
|
| 181 |
+
#loc72 = loc("tmp0"(#loc17))
|
| 182 |
+
#loc73 = loc("tmp0"(#loc18))
|
| 183 |
+
#loc74 = loc("tmp0"(#loc19))
|
| 184 |
+
#loc75 = loc("tmp0"(#loc20))
|
| 185 |
+
#loc76 = loc("tmp0"(#loc21))
|
| 186 |
+
#loc77 = loc("tmp0"(#loc22))
|
| 187 |
+
#loc78 = loc("tmp1"(#loc23))
|
| 188 |
+
#loc79 = loc("tmp4"(#loc24))
|
| 189 |
+
#loc80 = loc("_tmp3"(#loc25))
|
| 190 |
+
#loc81 = loc("_tmp3"(#loc26))
|
| 191 |
+
#loc82 = loc("tmp3"(#loc28))
|
| 192 |
+
#loc83 = loc("tmp3"(#loc29))
|
| 193 |
+
#loc84 = loc("tmp5"(#loc30))
|
SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttgir
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#blocked = #ttg.blocked<{sizePerThread = [1, 1], threadsPerWarp = [32, 1], warpsPerCTA = [1, 4], order = [0, 1]}>
|
| 2 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":18:0)
|
| 3 |
+
#loc1 = loc(unknown)
|
| 4 |
+
#loc26 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":41:25)
|
| 5 |
+
#loc40 = loc("in_ptr0"(#loc))
|
| 6 |
+
#loc41 = loc("out_ptr1"(#loc))
|
| 7 |
+
#loc42 = loc("ks0"(#loc))
|
| 8 |
+
#loc43 = loc("ks1"(#loc))
|
| 9 |
+
#loc44 = loc("xnumel"(#loc))
|
| 10 |
+
#loc45 = loc("r0_numel"(#loc))
|
| 11 |
+
#loc68 = loc("tmp3"(#loc26))
|
| 12 |
+
#loc73 = loc(callsite(#loc1 at #loc68))
|
| 13 |
+
module attributes {"ttg.num-ctas" = 1 : i32, "ttg.num-warps" = 4 : i32, ttg.target = "cuda:90", "ttg.threads-per-warp" = 32 : i32} {
|
| 14 |
+
tt.func public @triton_red_fused__to_copy_clone_slice_sum_transpose_5(%in_ptr0: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %out_ptr1: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("out_ptr1"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %xnumel: i32 loc("xnumel"(#loc)), %r0_numel: i32 loc("r0_numel"(#loc))) attributes {noinline = false} {
|
| 15 |
+
%cst = arith.constant dense<0> : tensor<32x16xi64, #blocked> loc(#loc1)
|
| 16 |
+
%c0_i32 = arith.constant 0 : i32 loc(#loc1)
|
| 17 |
+
%c16_i32 = arith.constant 16 : i32 loc(#loc1)
|
| 18 |
+
%c1_i64 = arith.constant 1 : i64 loc(#loc1)
|
| 19 |
+
%c32_i32 = arith.constant 32 : i32 loc(#loc1)
|
| 20 |
+
%cst_0 = arith.constant dense<0> : tensor<32x16xi32, #blocked> loc(#loc1)
|
| 21 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc46)
|
| 22 |
+
%xoffset_1 = arith.muli %xoffset, %c32_i32 : i32 loc(#loc47)
|
| 23 |
+
%xindex = tt.make_range {end = 32 : i32, start = 0 : i32} : tensor<32xi32, #ttg.slice<{dim = 1, parent = #blocked}>> loc(#loc48)
|
| 24 |
+
%xindex_2 = tt.expand_dims %xindex {axis = 1 : i32} : tensor<32xi32, #ttg.slice<{dim = 1, parent = #blocked}>> -> tensor<32x1xi32, #blocked> loc(#loc48)
|
| 25 |
+
%xindex_3 = tt.splat %xoffset_1 : i32 -> tensor<32x1xi32, #blocked> loc(#loc49)
|
| 26 |
+
%xindex_4 = arith.addi %xindex_3, %xindex_2 : tensor<32x1xi32, #blocked> loc(#loc49)
|
| 27 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<32x1xi32, #blocked> loc(#loc50)
|
| 28 |
+
%xmask_5 = arith.cmpi slt, %xindex_4, %xmask : tensor<32x1xi32, #blocked> loc(#loc50)
|
| 29 |
+
%r0_base = tt.make_range {end = 16 : i32, start = 0 : i32} : tensor<16xi32, #ttg.slice<{dim = 0, parent = #blocked}>> loc(#loc51)
|
| 30 |
+
%r0_base_6 = tt.expand_dims %r0_base {axis = 0 : i32} : tensor<16xi32, #ttg.slice<{dim = 0, parent = #blocked}>> -> tensor<1x16xi32, #blocked> loc(#loc51)
|
| 31 |
+
%x0 = arith.extsi %xindex_4 : tensor<32x1xi32, #blocked> to tensor<32x1xi64, #blocked> loc(#loc52)
|
| 32 |
+
%x0_7 = tt.splat %ks0 : i64 -> tensor<32x1xi64, #blocked> loc(#loc52)
|
| 33 |
+
%x0_8 = arith.remsi %x0, %x0_7 : tensor<32x1xi64, #blocked> loc(#loc52)
|
| 34 |
+
%x1 = arith.divsi %x0, %x0_7 : tensor<32x1xi64, #blocked> loc(#loc53)
|
| 35 |
+
%r0_mask = tt.splat %r0_numel : i32 -> tensor<1x16xi32, #blocked> loc(#loc54)
|
| 36 |
+
%tmp0 = tt.splat %ks0 : i64 -> tensor<1x16xi64, #blocked> loc(#loc55)
|
| 37 |
+
%tmp0_9 = tt.broadcast %x0_8 : tensor<32x1xi64, #blocked> -> tensor<32x16xi64, #blocked> loc(#loc56)
|
| 38 |
+
%tmp0_10 = arith.muli %ks0, %ks1 : i64 loc(#loc57)
|
| 39 |
+
%tmp0_11 = tt.splat %tmp0_10 : i64 -> tensor<32x1xi64, #blocked> loc(#loc58)
|
| 40 |
+
%tmp0_12 = arith.muli %tmp0_11, %x1 : tensor<32x1xi64, #blocked> loc(#loc58)
|
| 41 |
+
%tmp0_13 = tt.broadcast %tmp0_12 : tensor<32x1xi64, #blocked> -> tensor<32x16xi64, #blocked> loc(#loc59)
|
| 42 |
+
%tmp0_14 = tt.splat %in_ptr0 : !tt.ptr<i32> -> tensor<32x16x!tt.ptr<i32>, #blocked> loc(#loc60)
|
| 43 |
+
%tmp0_15 = tt.broadcast %xmask_5 : tensor<32x1xi1, #blocked> -> tensor<32x16xi1, #blocked> loc(#loc61)
|
| 44 |
+
%_tmp3 = scf.for %_tmp3_17 = %c0_i32 to %r0_numel step %c16_i32 iter_args(%_tmp3_18 = %cst) -> (tensor<32x16xi64, #blocked>) : i32 {
|
| 45 |
+
%r0_index = tt.splat %_tmp3_17 : i32 -> tensor<1x16xi32, #blocked> loc(#loc63)
|
| 46 |
+
%r0_index_19 = arith.addi %r0_index, %r0_base_6 : tensor<1x16xi32, #blocked> loc(#loc63)
|
| 47 |
+
%r0_mask_20 = arith.cmpi slt, %r0_index_19, %r0_mask : tensor<1x16xi32, #blocked> loc(#loc54)
|
| 48 |
+
%tmp0_21 = arith.extsi %r0_index_19 : tensor<1x16xi32, #blocked> to tensor<1x16xi64, #blocked> loc(#loc55)
|
| 49 |
+
%tmp0_22 = arith.muli %tmp0, %tmp0_21 : tensor<1x16xi64, #blocked> loc(#loc55)
|
| 50 |
+
%tmp0_23 = tt.broadcast %tmp0_22 : tensor<1x16xi64, #blocked> -> tensor<32x16xi64, #blocked> loc(#loc56)
|
| 51 |
+
%tmp0_24 = arith.addi %tmp0_9, %tmp0_23 : tensor<32x16xi64, #blocked> loc(#loc56)
|
| 52 |
+
%tmp0_25 = arith.addi %tmp0_24, %tmp0_13 : tensor<32x16xi64, #blocked> loc(#loc59)
|
| 53 |
+
%tmp0_26 = tt.addptr %tmp0_14, %tmp0_25 : tensor<32x16x!tt.ptr<i32>, #blocked>, tensor<32x16xi64, #blocked> loc(#loc60)
|
| 54 |
+
%tmp0_27 = tt.broadcast %r0_mask_20 : tensor<1x16xi1, #blocked> -> tensor<32x16xi1, #blocked> loc(#loc61)
|
| 55 |
+
%tmp0_28 = arith.andi %tmp0_27, %tmp0_15 : tensor<32x16xi1, #blocked> loc(#loc61)
|
| 56 |
+
%tmp0_29 = tt.load %tmp0_26, %tmp0_28, %cst_0 evictionPolicy = evict_last : tensor<32x16x!tt.ptr<i32>, #blocked> loc(#loc64)
|
| 57 |
+
%tmp1 = arith.extsi %tmp0_29 : tensor<32x16xi32, #blocked> to tensor<32x16xi64, #blocked> loc(#loc65)
|
| 58 |
+
%tmp4 = arith.addi %_tmp3_18, %tmp1 : tensor<32x16xi64, #blocked> loc(#loc66)
|
| 59 |
+
%_tmp3_30 = arith.select %tmp0_28, %tmp4, %_tmp3_18 : tensor<32x16xi1, #blocked>, tensor<32x16xi64, #blocked> loc(#loc67)
|
| 60 |
+
scf.yield %_tmp3_30 : tensor<32x16xi64, #blocked> loc(#loc24)
|
| 61 |
+
} loc(#loc62)
|
| 62 |
+
%tmp3 = "tt.reduce"(%_tmp3) <{axis = 1 : i32}> ({
|
| 63 |
+
^bb0(%tmp3_17: i64 loc(callsite(#loc1 at #loc68)), %tmp3_18: i64 loc(callsite(#loc1 at #loc68))):
|
| 64 |
+
%tmp3_19 = arith.addi %tmp3_17, %tmp3_18 : i64 loc(#loc74)
|
| 65 |
+
tt.reduce.return %tmp3_19 : i64 loc(#loc72)
|
| 66 |
+
}) : (tensor<32x16xi64, #blocked>) -> tensor<32xi64, #ttg.slice<{dim = 1, parent = #blocked}>> loc(#loc72)
|
| 67 |
+
%tmp3_16 = tt.expand_dims %tmp3 {axis = 1 : i32} : tensor<32xi64, #ttg.slice<{dim = 1, parent = #blocked}>> -> tensor<32x1xi64, #blocked> loc(#loc69)
|
| 68 |
+
%tmp5 = arith.trunci %tmp3_16 : tensor<32x1xi64, #blocked> to tensor<32x1xi32, #blocked> loc(#loc70)
|
| 69 |
+
%0 = arith.cmpi sle, %ks0, %c1_i64 : i64 loc(#loc30)
|
| 70 |
+
%1 = arith.cmpi sgt, %ks0, %c1_i64 : i64 loc(#loc31)
|
| 71 |
+
%2 = arith.extui %1 : i1 to i64 loc(#loc32)
|
| 72 |
+
%3 = arith.muli %ks0, %2 : i64 loc(#loc32)
|
| 73 |
+
%4 = arith.extui %0 : i1 to i64 loc(#loc71)
|
| 74 |
+
%5 = arith.addi %4, %3 : i64 loc(#loc33)
|
| 75 |
+
%6 = tt.splat %5 : i64 -> tensor<32x1xi64, #blocked> loc(#loc35)
|
| 76 |
+
%7 = arith.muli %x1, %6 : tensor<32x1xi64, #blocked> loc(#loc35)
|
| 77 |
+
%8 = arith.addi %x0_8, %7 : tensor<32x1xi64, #blocked> loc(#loc36)
|
| 78 |
+
%9 = tt.splat %out_ptr1 : !tt.ptr<i32> -> tensor<32x1x!tt.ptr<i32>, #blocked> loc(#loc37)
|
| 79 |
+
%10 = tt.addptr %9, %8 : tensor<32x1x!tt.ptr<i32>, #blocked>, tensor<32x1xi64, #blocked> loc(#loc37)
|
| 80 |
+
tt.store %10, %tmp5, %xmask_5 : tensor<32x1x!tt.ptr<i32>, #blocked> loc(#loc38)
|
| 81 |
+
tt.return loc(#loc39)
|
| 82 |
+
} loc(#loc)
|
| 83 |
+
} loc(#loc)
|
| 84 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":21:28)
|
| 85 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":21:33)
|
| 86 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":22:44)
|
| 87 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":22:23)
|
| 88 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":23:21)
|
| 89 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":24:37)
|
| 90 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":26:19)
|
| 91 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":27:19)
|
| 92 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":32:29)
|
| 93 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:43)
|
| 94 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:39)
|
| 95 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:54)
|
| 96 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:58)
|
| 97 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:50)
|
| 98 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:34)
|
| 99 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:73)
|
| 100 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":30:40)
|
| 101 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":31:31)
|
| 102 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:63)
|
| 103 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":37:23)
|
| 104 |
+
#loc22 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":39:23)
|
| 105 |
+
#loc23 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":40:48)
|
| 106 |
+
#loc24 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":40:8)
|
| 107 |
+
#loc25 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:36)
|
| 108 |
+
#loc27 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:15)
|
| 109 |
+
#loc28 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":41:28)
|
| 110 |
+
#loc29 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":42:19)
|
| 111 |
+
#loc30 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:49)
|
| 112 |
+
#loc31 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:75)
|
| 113 |
+
#loc32 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:66)
|
| 114 |
+
#loc33 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:57)
|
| 115 |
+
#loc34 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:41)
|
| 116 |
+
#loc35 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:34)
|
| 117 |
+
#loc36 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:30)
|
| 118 |
+
#loc37 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:25)
|
| 119 |
+
#loc38 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:88)
|
| 120 |
+
#loc39 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:4)
|
| 121 |
+
#loc46 = loc("xoffset"(#loc2))
|
| 122 |
+
#loc47 = loc("xoffset"(#loc3))
|
| 123 |
+
#loc48 = loc("xindex"(#loc4))
|
| 124 |
+
#loc49 = loc("xindex"(#loc5))
|
| 125 |
+
#loc50 = loc("xmask"(#loc6))
|
| 126 |
+
#loc51 = loc("r0_base"(#loc7))
|
| 127 |
+
#loc52 = loc("x0"(#loc8))
|
| 128 |
+
#loc53 = loc("x1"(#loc9))
|
| 129 |
+
#loc54 = loc("r0_mask"(#loc10))
|
| 130 |
+
#loc55 = loc("tmp0"(#loc11))
|
| 131 |
+
#loc56 = loc("tmp0"(#loc12))
|
| 132 |
+
#loc57 = loc("tmp0"(#loc13))
|
| 133 |
+
#loc58 = loc("tmp0"(#loc14))
|
| 134 |
+
#loc59 = loc("tmp0"(#loc15))
|
| 135 |
+
#loc60 = loc("tmp0"(#loc16))
|
| 136 |
+
#loc61 = loc("tmp0"(#loc17))
|
| 137 |
+
#loc62 = loc("_tmp3"(#loc18))
|
| 138 |
+
#loc63 = loc("r0_index"(#loc19))
|
| 139 |
+
#loc64 = loc("tmp0"(#loc20))
|
| 140 |
+
#loc65 = loc("tmp1"(#loc21))
|
| 141 |
+
#loc66 = loc("tmp4"(#loc22))
|
| 142 |
+
#loc67 = loc("_tmp3"(#loc23))
|
| 143 |
+
#loc69 = loc("tmp3"(#loc28))
|
| 144 |
+
#loc70 = loc("tmp5"(#loc29))
|
| 145 |
+
#loc71 = loc(fused[#loc33, #loc34])
|
| 146 |
+
#loc72 = loc(callsite(#loc25 at #loc68))
|
| 147 |
+
#loc74 = loc(callsite(#loc27 at #loc72))
|
SpecForge-ext/cache/compiled_kernels/triton/0/2HBOMUT44J5WFCUWYGRFAAS3HGVNDHLHT7HCSXUCAOIKU6XGJNTA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttir
ADDED
|
@@ -0,0 +1,152 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":18:0)
|
| 2 |
+
#loc1 = loc(unknown)
|
| 3 |
+
#loc29 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":41:25)
|
| 4 |
+
#loc43 = loc("in_ptr0"(#loc))
|
| 5 |
+
#loc44 = loc("out_ptr1"(#loc))
|
| 6 |
+
#loc45 = loc("ks0"(#loc))
|
| 7 |
+
#loc46 = loc("ks1"(#loc))
|
| 8 |
+
#loc47 = loc("xnumel"(#loc))
|
| 9 |
+
#loc48 = loc("r0_numel"(#loc))
|
| 10 |
+
#loc74 = loc("tmp3"(#loc29))
|
| 11 |
+
#loc79 = loc(callsite(#loc1 at #loc74))
|
| 12 |
+
module {
|
| 13 |
+
tt.func public @triton_red_fused__to_copy_clone_slice_sum_transpose_5(%in_ptr0: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %out_ptr1: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("out_ptr1"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %xnumel: i32 loc("xnumel"(#loc)), %r0_numel: i32 loc("r0_numel"(#loc))) attributes {noinline = false} {
|
| 14 |
+
%c1_i64 = arith.constant 1 : i64 loc(#loc1)
|
| 15 |
+
%cst = arith.constant dense<0> : tensor<32x16xi32> loc(#loc1)
|
| 16 |
+
%c16_i32 = arith.constant 16 : i32 loc(#loc2)
|
| 17 |
+
%c0_i32 = arith.constant 0 : i32 loc(#loc2)
|
| 18 |
+
%_tmp3 = arith.constant dense<0> : tensor<32x16xi64> loc(#loc49)
|
| 19 |
+
%c32_i32 = arith.constant 32 : i32 loc(#loc1)
|
| 20 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc50)
|
| 21 |
+
%xoffset_0 = arith.muli %xoffset, %c32_i32 : i32 loc(#loc51)
|
| 22 |
+
%xindex = tt.make_range {end = 32 : i32, start = 0 : i32} : tensor<32xi32> loc(#loc52)
|
| 23 |
+
%xindex_1 = tt.expand_dims %xindex {axis = 1 : i32} : tensor<32xi32> -> tensor<32x1xi32> loc(#loc53)
|
| 24 |
+
%xindex_2 = tt.splat %xoffset_0 : i32 -> tensor<32x1xi32> loc(#loc54)
|
| 25 |
+
%xindex_3 = arith.addi %xindex_2, %xindex_1 : tensor<32x1xi32> loc(#loc54)
|
| 26 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<32x1xi32> loc(#loc55)
|
| 27 |
+
%xmask_4 = arith.cmpi slt, %xindex_3, %xmask : tensor<32x1xi32> loc(#loc55)
|
| 28 |
+
%r0_base = tt.make_range {end = 16 : i32, start = 0 : i32} : tensor<16xi32> loc(#loc56)
|
| 29 |
+
%r0_base_5 = tt.expand_dims %r0_base {axis = 0 : i32} : tensor<16xi32> -> tensor<1x16xi32> loc(#loc57)
|
| 30 |
+
%x0 = arith.extsi %xindex_3 : tensor<32x1xi32> to tensor<32x1xi64> loc(#loc58)
|
| 31 |
+
%x0_6 = tt.splat %ks0 : i64 -> tensor<32x1xi64> loc(#loc58)
|
| 32 |
+
%x0_7 = arith.remsi %x0, %x0_6 : tensor<32x1xi64> loc(#loc58)
|
| 33 |
+
%x1 = arith.divsi %x0, %x0_6 : tensor<32x1xi64> loc(#loc59)
|
| 34 |
+
%_tmp3_8 = scf.for %r0_offset = %c0_i32 to %r0_numel step %c16_i32 iter_args(%_tmp3_10 = %_tmp3) -> (tensor<32x16xi64>) : i32 {
|
| 35 |
+
%r0_index = tt.splat %r0_offset : i32 -> tensor<1x16xi32> loc(#loc61)
|
| 36 |
+
%r0_index_11 = arith.addi %r0_index, %r0_base_5 : tensor<1x16xi32> loc(#loc61)
|
| 37 |
+
%r0_mask = tt.splat %r0_numel : i32 -> tensor<1x16xi32> loc(#loc62)
|
| 38 |
+
%r0_mask_12 = arith.cmpi slt, %r0_index_11, %r0_mask : tensor<1x16xi32> loc(#loc62)
|
| 39 |
+
%tmp0 = arith.extsi %r0_index_11 : tensor<1x16xi32> to tensor<1x16xi64> loc(#loc63)
|
| 40 |
+
%tmp0_13 = tt.splat %ks0 : i64 -> tensor<1x16xi64> loc(#loc63)
|
| 41 |
+
%tmp0_14 = arith.muli %tmp0_13, %tmp0 : tensor<1x16xi64> loc(#loc63)
|
| 42 |
+
%tmp0_15 = tt.broadcast %x0_7 : tensor<32x1xi64> -> tensor<32x16xi64> loc(#loc64)
|
| 43 |
+
%tmp0_16 = tt.broadcast %tmp0_14 : tensor<1x16xi64> -> tensor<32x16xi64> loc(#loc64)
|
| 44 |
+
%tmp0_17 = arith.addi %tmp0_15, %tmp0_16 : tensor<32x16xi64> loc(#loc64)
|
| 45 |
+
%tmp0_18 = arith.muli %ks0, %ks1 : i64 loc(#loc65)
|
| 46 |
+
%tmp0_19 = tt.splat %tmp0_18 : i64 -> tensor<32x1xi64> loc(#loc66)
|
| 47 |
+
%tmp0_20 = arith.muli %tmp0_19, %x1 : tensor<32x1xi64> loc(#loc66)
|
| 48 |
+
%tmp0_21 = tt.broadcast %tmp0_20 : tensor<32x1xi64> -> tensor<32x16xi64> loc(#loc67)
|
| 49 |
+
%tmp0_22 = arith.addi %tmp0_17, %tmp0_21 : tensor<32x16xi64> loc(#loc67)
|
| 50 |
+
%tmp0_23 = tt.splat %in_ptr0 : !tt.ptr<i32> -> tensor<32x16x!tt.ptr<i32>> loc(#loc68)
|
| 51 |
+
%tmp0_24 = tt.addptr %tmp0_23, %tmp0_22 : tensor<32x16x!tt.ptr<i32>>, tensor<32x16xi64> loc(#loc68)
|
| 52 |
+
%tmp0_25 = tt.broadcast %r0_mask_12 : tensor<1x16xi1> -> tensor<32x16xi1> loc(#loc69)
|
| 53 |
+
%tmp0_26 = tt.broadcast %xmask_4 : tensor<32x1xi1> -> tensor<32x16xi1> loc(#loc69)
|
| 54 |
+
%tmp0_27 = arith.andi %tmp0_25, %tmp0_26 : tensor<32x16xi1> loc(#loc69)
|
| 55 |
+
%tmp0_28 = tt.load %tmp0_24, %tmp0_27, %cst evictionPolicy = evict_last : tensor<32x16x!tt.ptr<i32>> loc(#loc70)
|
| 56 |
+
%tmp1 = arith.extsi %tmp0_28 : tensor<32x16xi32> to tensor<32x16xi64> loc(#loc71)
|
| 57 |
+
%tmp4 = arith.addi %_tmp3_10, %tmp1 : tensor<32x16xi64> loc(#loc72)
|
| 58 |
+
%_tmp3_29 = arith.select %tmp0_27, %tmp4, %_tmp3_10 : tensor<32x16xi1>, tensor<32x16xi64> loc(#loc73)
|
| 59 |
+
scf.yield %_tmp3_29 : tensor<32x16xi64> loc(#loc27)
|
| 60 |
+
} loc(#loc60)
|
| 61 |
+
%tmp3 = "tt.reduce"(%_tmp3_8) <{axis = 1 : i32}> ({
|
| 62 |
+
^bb0(%tmp3_10: i64 loc(callsite(#loc1 at #loc74)), %tmp3_11: i64 loc(callsite(#loc1 at #loc74))):
|
| 63 |
+
%tmp3_12 = arith.addi %tmp3_10, %tmp3_11 : i64 loc(#loc80)
|
| 64 |
+
tt.reduce.return %tmp3_12 : i64 loc(#loc78)
|
| 65 |
+
}) : (tensor<32x16xi64>) -> tensor<32xi64> loc(#loc78)
|
| 66 |
+
%tmp3_9 = tt.expand_dims %tmp3 {axis = 1 : i32} : tensor<32xi64> -> tensor<32x1xi64> loc(#loc75)
|
| 67 |
+
%tmp5 = arith.trunci %tmp3_9 : tensor<32x1xi64> to tensor<32x1xi32> loc(#loc76)
|
| 68 |
+
%0 = arith.cmpi sle, %ks0, %c1_i64 : i64 loc(#loc33)
|
| 69 |
+
%1 = arith.cmpi sgt, %ks0, %c1_i64 : i64 loc(#loc34)
|
| 70 |
+
%2 = arith.extui %1 : i1 to i64 loc(#loc35)
|
| 71 |
+
%3 = arith.muli %ks0, %2 : i64 loc(#loc35)
|
| 72 |
+
%4 = arith.extui %0 : i1 to i64 loc(#loc77)
|
| 73 |
+
%5 = arith.addi %4, %3 : i64 loc(#loc36)
|
| 74 |
+
%6 = tt.splat %5 : i64 -> tensor<32x1xi64> loc(#loc38)
|
| 75 |
+
%7 = arith.muli %x1, %6 : tensor<32x1xi64> loc(#loc38)
|
| 76 |
+
%8 = arith.addi %x0_7, %7 : tensor<32x1xi64> loc(#loc39)
|
| 77 |
+
%9 = tt.splat %out_ptr1 : !tt.ptr<i32> -> tensor<32x1x!tt.ptr<i32>> loc(#loc40)
|
| 78 |
+
%10 = tt.addptr %9, %8 : tensor<32x1x!tt.ptr<i32>>, tensor<32x1xi64> loc(#loc40)
|
| 79 |
+
tt.store %10, %tmp5, %xmask_4 : tensor<32x1x!tt.ptr<i32>> loc(#loc41)
|
| 80 |
+
tt.return loc(#loc42)
|
| 81 |
+
} loc(#loc)
|
| 82 |
+
} loc(#loc)
|
| 83 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":30:40)
|
| 84 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":28:43)
|
| 85 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":21:28)
|
| 86 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":21:33)
|
| 87 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":22:36)
|
| 88 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":22:44)
|
| 89 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":22:23)
|
| 90 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":23:21)
|
| 91 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":24:27)
|
| 92 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":24:37)
|
| 93 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":26:19)
|
| 94 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":27:19)
|
| 95 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":31:31)
|
| 96 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":32:29)
|
| 97 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:43)
|
| 98 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:39)
|
| 99 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:54)
|
| 100 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:58)
|
| 101 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:50)
|
| 102 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:34)
|
| 103 |
+
#loc22 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:73)
|
| 104 |
+
#loc23 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":36:63)
|
| 105 |
+
#loc24 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":37:23)
|
| 106 |
+
#loc25 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":39:23)
|
| 107 |
+
#loc26 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":40:48)
|
| 108 |
+
#loc27 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":40:8)
|
| 109 |
+
#loc28 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:36)
|
| 110 |
+
#loc30 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:15)
|
| 111 |
+
#loc31 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":41:28)
|
| 112 |
+
#loc32 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":42:19)
|
| 113 |
+
#loc33 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:49)
|
| 114 |
+
#loc34 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:75)
|
| 115 |
+
#loc35 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:66)
|
| 116 |
+
#loc36 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:57)
|
| 117 |
+
#loc37 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:41)
|
| 118 |
+
#loc38 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:34)
|
| 119 |
+
#loc39 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:30)
|
| 120 |
+
#loc40 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:25)
|
| 121 |
+
#loc41 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:88)
|
| 122 |
+
#loc42 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wc/cwc2jctz67s7n6fotl22gr44glijrcfxpeblvk32hornlllznx2m.py":43:4)
|
| 123 |
+
#loc49 = loc("_tmp3"(#loc3))
|
| 124 |
+
#loc50 = loc("xoffset"(#loc4))
|
| 125 |
+
#loc51 = loc("xoffset"(#loc5))
|
| 126 |
+
#loc52 = loc("xindex"(#loc6))
|
| 127 |
+
#loc53 = loc("xindex"(#loc7))
|
| 128 |
+
#loc54 = loc("xindex"(#loc8))
|
| 129 |
+
#loc55 = loc("xmask"(#loc9))
|
| 130 |
+
#loc56 = loc("r0_base"(#loc10))
|
| 131 |
+
#loc57 = loc("r0_base"(#loc11))
|
| 132 |
+
#loc58 = loc("x0"(#loc12))
|
| 133 |
+
#loc59 = loc("x1"(#loc13))
|
| 134 |
+
#loc60 = loc("_tmp3"(#loc2))
|
| 135 |
+
#loc61 = loc("r0_index"(#loc14))
|
| 136 |
+
#loc62 = loc("r0_mask"(#loc15))
|
| 137 |
+
#loc63 = loc("tmp0"(#loc16))
|
| 138 |
+
#loc64 = loc("tmp0"(#loc17))
|
| 139 |
+
#loc65 = loc("tmp0"(#loc18))
|
| 140 |
+
#loc66 = loc("tmp0"(#loc19))
|
| 141 |
+
#loc67 = loc("tmp0"(#loc20))
|
| 142 |
+
#loc68 = loc("tmp0"(#loc21))
|
| 143 |
+
#loc69 = loc("tmp0"(#loc22))
|
| 144 |
+
#loc70 = loc("tmp0"(#loc23))
|
| 145 |
+
#loc71 = loc("tmp1"(#loc24))
|
| 146 |
+
#loc72 = loc("tmp4"(#loc25))
|
| 147 |
+
#loc73 = loc("_tmp3"(#loc26))
|
| 148 |
+
#loc75 = loc("tmp3"(#loc31))
|
| 149 |
+
#loc76 = loc("tmp5"(#loc32))
|
| 150 |
+
#loc77 = loc(fused[#loc36, #loc37])
|
| 151 |
+
#loc78 = loc(callsite(#loc28 at #loc74))
|
| 152 |
+
#loc80 = loc(callsite(#loc30 at #loc78))
|
SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/__grp__triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"child_paths": {"triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.source": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.source", "triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ttir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ttir", "triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ttgir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ttgir", "triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.llir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.llir", "triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ptx": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ptx", "triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.cubin": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.cubin", "triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.json": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.json"}}
|
SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.cubin
ADDED
|
Binary file (76 kB). View file
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"hash": "d4e9ec88be03aebb42041f53f2cbb9bb30afcd0e87a7a62c3ff7f5fe862eba44", "target": {"backend": "cuda", "arch": 90, "warp_size": 32}, "num_warps": 4, "num_ctas": 1, "num_stages": 1, "warp_size": 32, "maxnreg": null, "cluster_dims": [1, 1, 1], "ptx_version": null, "ptx_options": null, "ir_override": null, "enable_fp_fusion": true, "launch_cooperative_grid": false, "launch_pdl": false, "supported_fp8_dtypes": ["fp8e4b15", "fp8e4nv", "fp8e5"], "deprecated_fp8_dot_operand_dtypes": ["fp8e4b15"], "default_dot_input_precision": "tf32", "allowed_dot_input_precisions": ["tf32", "tf32x3", "ieee"], "max_num_imprecise_acc_default": 1073741824, "extern_libs": [["libdevice", "/workspace/specforge/lib/python3.11/site-packages/triton/backends/nvidia/lib/libdevice.10.bc"]], "debug": true, "backend_name": "cuda", "sanitize_overflow": false, "arch": "sm90", "instrumentation_mode": "", "triton_version": "3.5.1", "tensordesc_meta": [], "shared": 0, "tmem_size": 0, "global_scratch_size": 0, "global_scratch_align": 1, "profile_scratch_size": 0, "profile_scratch_align": 1, "name": "triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0"}
|
SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.llir
ADDED
|
@@ -0,0 +1,667 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
; ModuleID = 'LLVMDialectModule'
|
| 2 |
+
source_filename = "LLVMDialectModule"
|
| 3 |
+
target datalayout = "e-p3:32:32-p4:32:32-p5:32:32-p6:32:32-p7:32:32-i64:64-i128:128-v16:16-v32:32-n16:32:64"
|
| 4 |
+
|
| 5 |
+
@assertFunc_1 = internal constant [8 x i8] c"unknown\00"
|
| 6 |
+
@assertFile_1 = internal constant [114 x i8] c"/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py\00"
|
| 7 |
+
@assertMessage_1 = internal constant [38 x i8] c"index out of bounds: 0 <= tmp25 < ks4\00"
|
| 8 |
+
@assertFunc_0 = internal constant [8 x i8] c"unknown\00"
|
| 9 |
+
@assertFile_0 = internal constant [114 x i8] c"/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py\00"
|
| 10 |
+
@assertMessage_0 = internal constant [37 x i8] c"index out of bounds: 0 <= tmp5 < ks2\00"
|
| 11 |
+
|
| 12 |
+
; Function Attrs: noreturn
|
| 13 |
+
declare !dbg !5 void @__assertfail(ptr, ptr, i32, ptr, i64) local_unnamed_addr #0
|
| 14 |
+
|
| 15 |
+
define ptx_kernel void @triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0(ptr addrspace(1) %0, ptr addrspace(1) %1, ptr addrspace(1) %2, ptr addrspace(1) %3, ptr addrspace(1) %4, i64 %5, i64 %6, i64 %7, i64 %8, i64 %9, i32 %10, ptr addrspace(1) readnone captures(none) %11, ptr addrspace(1) readnone captures(none) %12) local_unnamed_addr #1 !dbg !9 {
|
| 16 |
+
%14 = tail call i32 @llvm.nvvm.read.ptx.sreg.ctaid.x(), !dbg !10
|
| 17 |
+
%15 = shl i32 %14, 10, !dbg !11
|
| 18 |
+
%16 = tail call i32 @llvm.nvvm.read.ptx.sreg.tid.x(), !dbg !12
|
| 19 |
+
%17 = shl nuw nsw i32 %16, 3, !dbg !12
|
| 20 |
+
%18 = and i32 %17, 1016, !dbg !12
|
| 21 |
+
%19 = or disjoint i32 %18, %15, !dbg !13
|
| 22 |
+
%20 = or disjoint i32 %19, 1, !dbg !13
|
| 23 |
+
%21 = or disjoint i32 %19, 2, !dbg !13
|
| 24 |
+
%22 = or disjoint i32 %19, 3, !dbg !13
|
| 25 |
+
%23 = or disjoint i32 %19, 4, !dbg !13
|
| 26 |
+
%24 = or disjoint i32 %19, 5, !dbg !13
|
| 27 |
+
%25 = or disjoint i32 %19, 6, !dbg !13
|
| 28 |
+
%26 = or disjoint i32 %19, 7, !dbg !13
|
| 29 |
+
%27 = insertelement <8 x i32> poison, i32 %26, i64 0, !dbg !14
|
| 30 |
+
%28 = insertelement <8 x i32> %27, i32 %25, i64 1, !dbg !14
|
| 31 |
+
%29 = insertelement <8 x i32> %28, i32 %24, i64 2, !dbg !14
|
| 32 |
+
%30 = insertelement <8 x i32> %29, i32 %23, i64 3, !dbg !14
|
| 33 |
+
%31 = insertelement <8 x i32> %30, i32 %22, i64 4, !dbg !14
|
| 34 |
+
%32 = insertelement <8 x i32> %31, i32 %21, i64 5, !dbg !14
|
| 35 |
+
%33 = insertelement <8 x i32> %32, i32 %20, i64 6, !dbg !14
|
| 36 |
+
%34 = insertelement <8 x i32> %33, i32 %19, i64 7, !dbg !14
|
| 37 |
+
%35 = sext <8 x i32> %34 to <8 x i64>, !dbg !14
|
| 38 |
+
%36 = extractelement <8 x i64> %35, i64 7, !dbg !15
|
| 39 |
+
%37 = sdiv i64 %36, %5, !dbg !14
|
| 40 |
+
%38 = extractelement <8 x i64> %35, i64 6, !dbg !15
|
| 41 |
+
%39 = sdiv i64 %38, %5, !dbg !14
|
| 42 |
+
%40 = extractelement <8 x i64> %35, i64 5, !dbg !15
|
| 43 |
+
%41 = sdiv i64 %40, %5, !dbg !14
|
| 44 |
+
%42 = extractelement <8 x i64> %35, i64 4, !dbg !15
|
| 45 |
+
%43 = sdiv i64 %42, %5, !dbg !14
|
| 46 |
+
%44 = extractelement <8 x i64> %35, i64 3, !dbg !15
|
| 47 |
+
%45 = sdiv i64 %44, %5, !dbg !14
|
| 48 |
+
%46 = extractelement <8 x i64> %35, i64 2, !dbg !15
|
| 49 |
+
%47 = sdiv i64 %46, %5, !dbg !14
|
| 50 |
+
%48 = extractelement <8 x i64> %35, i64 1, !dbg !15
|
| 51 |
+
%49 = sdiv i64 %48, %5, !dbg !14
|
| 52 |
+
%50 = extractelement <8 x i64> %35, i64 0, !dbg !15
|
| 53 |
+
%51 = sdiv i64 %50, %5, !dbg !14
|
| 54 |
+
%52 = srem i64 %37, %6, !dbg !16
|
| 55 |
+
%53 = srem i64 %39, %6, !dbg !16
|
| 56 |
+
%54 = srem i64 %41, %6, !dbg !16
|
| 57 |
+
%55 = srem i64 %43, %6, !dbg !16
|
| 58 |
+
%56 = srem i64 %45, %6, !dbg !16
|
| 59 |
+
%57 = srem i64 %47, %6, !dbg !16
|
| 60 |
+
%58 = srem i64 %49, %6, !dbg !16
|
| 61 |
+
%59 = srem i64 %51, %6, !dbg !16
|
| 62 |
+
%60 = getelementptr bfloat, ptr addrspace(1) %0, i64 %36, !dbg !15
|
| 63 |
+
%61 = getelementptr bfloat, ptr addrspace(1) %0, i64 %38, !dbg !15
|
| 64 |
+
%62 = getelementptr bfloat, ptr addrspace(1) %0, i64 %40, !dbg !15
|
| 65 |
+
%63 = getelementptr bfloat, ptr addrspace(1) %0, i64 %42, !dbg !15
|
| 66 |
+
%64 = getelementptr bfloat, ptr addrspace(1) %0, i64 %44, !dbg !15
|
| 67 |
+
%65 = getelementptr bfloat, ptr addrspace(1) %0, i64 %46, !dbg !15
|
| 68 |
+
%66 = getelementptr bfloat, ptr addrspace(1) %0, i64 %48, !dbg !15
|
| 69 |
+
%67 = getelementptr bfloat, ptr addrspace(1) %0, i64 %50, !dbg !15
|
| 70 |
+
%68 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !17
|
| 71 |
+
%69 = getelementptr i64, ptr addrspace(1) %1, i64 %52, !dbg !18
|
| 72 |
+
%70 = getelementptr i64, ptr addrspace(1) %1, i64 %53, !dbg !18
|
| 73 |
+
%71 = getelementptr i64, ptr addrspace(1) %1, i64 %54, !dbg !18
|
| 74 |
+
%72 = getelementptr i64, ptr addrspace(1) %1, i64 %55, !dbg !18
|
| 75 |
+
%73 = getelementptr i64, ptr addrspace(1) %1, i64 %56, !dbg !18
|
| 76 |
+
%74 = getelementptr i64, ptr addrspace(1) %1, i64 %57, !dbg !18
|
| 77 |
+
%75 = getelementptr i64, ptr addrspace(1) %1, i64 %58, !dbg !18
|
| 78 |
+
%76 = getelementptr i64, ptr addrspace(1) %1, i64 %59, !dbg !18
|
| 79 |
+
%77 = insertelement <8 x i32> poison, i32 %19, i64 0, !dbg !19
|
| 80 |
+
%78 = insertelement <8 x i32> %77, i32 %20, i64 1, !dbg !19
|
| 81 |
+
%79 = insertelement <8 x i32> %78, i32 %21, i64 2, !dbg !19
|
| 82 |
+
%80 = insertelement <8 x i32> %79, i32 %22, i64 3, !dbg !19
|
| 83 |
+
%81 = insertelement <8 x i32> %80, i32 %23, i64 4, !dbg !19
|
| 84 |
+
%82 = insertelement <8 x i32> %81, i32 %24, i64 5, !dbg !19
|
| 85 |
+
%83 = insertelement <8 x i32> %82, i32 %25, i64 6, !dbg !19
|
| 86 |
+
%84 = insertelement <8 x i32> %83, i32 %26, i64 7, !dbg !19
|
| 87 |
+
%85 = insertelement <8 x i32> poison, i32 %10, i64 0, !dbg !19
|
| 88 |
+
%86 = shufflevector <8 x i32> %85, <8 x i32> poison, <8 x i32> zeroinitializer, !dbg !19
|
| 89 |
+
%87 = icmp slt <8 x i32> %84, %86, !dbg !19
|
| 90 |
+
%88 = extractelement <8 x i1> %87, i64 0, !dbg !17
|
| 91 |
+
%89 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %60, i64 %68, i1 %88) #4, !dbg !17
|
| 92 |
+
%90 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !17
|
| 93 |
+
%91 = extractelement <8 x i1> %87, i64 1, !dbg !17
|
| 94 |
+
%92 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %61, i64 %90, i1 %91) #4, !dbg !17
|
| 95 |
+
%93 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !17
|
| 96 |
+
%94 = extractelement <8 x i1> %87, i64 2, !dbg !17
|
| 97 |
+
%95 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %62, i64 %93, i1 %94) #4, !dbg !17
|
| 98 |
+
%96 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !17
|
| 99 |
+
%97 = extractelement <8 x i1> %87, i64 3, !dbg !17
|
| 100 |
+
%98 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %63, i64 %96, i1 %97) #4, !dbg !17
|
| 101 |
+
%99 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !17
|
| 102 |
+
%100 = extractelement <8 x i1> %87, i64 4, !dbg !17
|
| 103 |
+
%101 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %64, i64 %99, i1 %100) #4, !dbg !17
|
| 104 |
+
%102 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !17
|
| 105 |
+
%103 = extractelement <8 x i1> %87, i64 5, !dbg !17
|
| 106 |
+
%104 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %65, i64 %102, i1 %103) #4, !dbg !17
|
| 107 |
+
%105 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !17
|
| 108 |
+
%106 = extractelement <8 x i1> %87, i64 6, !dbg !17
|
| 109 |
+
%107 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %66, i64 %105, i1 %106) #4, !dbg !17
|
| 110 |
+
%108 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !17
|
| 111 |
+
%109 = extractelement <8 x i1> %87, i64 7, !dbg !17
|
| 112 |
+
%110 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %67, i64 %108, i1 %109) #4, !dbg !17
|
| 113 |
+
%111 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !20
|
| 114 |
+
%112 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b64 { $0 }, [ $1 + 0 ], $2;", "=l,l,l,b"(ptr addrspace(1) %69, i64 %111, i1 %88) #4, !dbg !20
|
| 115 |
+
%113 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !20
|
| 116 |
+
%114 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b64 { $0 }, [ $1 + 0 ], $2;", "=l,l,l,b"(ptr addrspace(1) %70, i64 %113, i1 %91) #4, !dbg !20
|
| 117 |
+
%115 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !20
|
| 118 |
+
%116 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b64 { $0 }, [ $1 + 0 ], $2;", "=l,l,l,b"(ptr addrspace(1) %71, i64 %115, i1 %94) #4, !dbg !20
|
| 119 |
+
%117 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !20
|
| 120 |
+
%118 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b64 { $0 }, [ $1 + 0 ], $2;", "=l,l,l,b"(ptr addrspace(1) %72, i64 %117, i1 %97) #4, !dbg !20
|
| 121 |
+
%119 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !20
|
| 122 |
+
%120 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b64 { $0 }, [ $1 + 0 ], $2;", "=l,l,l,b"(ptr addrspace(1) %73, i64 %119, i1 %100) #4, !dbg !20
|
| 123 |
+
%121 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !20
|
| 124 |
+
%122 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b64 { $0 }, [ $1 + 0 ], $2;", "=l,l,l,b"(ptr addrspace(1) %74, i64 %121, i1 %103) #4, !dbg !20
|
| 125 |
+
%123 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !20
|
| 126 |
+
%124 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b64 { $0 }, [ $1 + 0 ], $2;", "=l,l,l,b"(ptr addrspace(1) %75, i64 %123, i1 %106) #4, !dbg !20
|
| 127 |
+
%125 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !20
|
| 128 |
+
%126 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b64 { $0 }, [ $1 + 0 ], $2;", "=l,l,l,b"(ptr addrspace(1) %76, i64 %125, i1 %109) #4, !dbg !20
|
| 129 |
+
%127 = insertelement <8 x i64> poison, i64 %112, i64 0, !dbg !21
|
| 130 |
+
%128 = insertelement <8 x i64> %127, i64 %114, i64 1, !dbg !21
|
| 131 |
+
%129 = insertelement <8 x i64> %128, i64 %116, i64 2, !dbg !21
|
| 132 |
+
%130 = insertelement <8 x i64> %129, i64 %118, i64 3, !dbg !21
|
| 133 |
+
%131 = insertelement <8 x i64> %130, i64 %120, i64 4, !dbg !21
|
| 134 |
+
%132 = insertelement <8 x i64> %131, i64 %122, i64 5, !dbg !21
|
| 135 |
+
%133 = insertelement <8 x i64> %132, i64 %124, i64 6, !dbg !21
|
| 136 |
+
%134 = insertelement <8 x i64> %133, i64 %126, i64 7, !dbg !21
|
| 137 |
+
%135 = icmp slt <8 x i64> %134, zeroinitializer, !dbg !21
|
| 138 |
+
%136 = insertelement <8 x i64> poison, i64 %7, i64 0, !dbg !22
|
| 139 |
+
%137 = shufflevector <8 x i64> %136, <8 x i64> poison, <8 x i32> zeroinitializer, !dbg !22
|
| 140 |
+
%138 = select <8 x i1> %135, <8 x i64> %137, <8 x i64> zeroinitializer, !dbg !22
|
| 141 |
+
%139 = add <8 x i64> %138, %134, !dbg !22
|
| 142 |
+
%140 = icmp slt <8 x i64> %139, zeroinitializer, !dbg !23
|
| 143 |
+
%141 = icmp sge <8 x i64> %139, %137, !dbg !24
|
| 144 |
+
%142 = or <8 x i1> %140, %141, !dbg !25
|
| 145 |
+
%143 = and <8 x i1> %87, %142, !dbg !26
|
| 146 |
+
%144 = bitcast <8 x i1> %143 to i8, !dbg !27
|
| 147 |
+
%.not = icmp eq i8 %144, 0, !dbg !27
|
| 148 |
+
br i1 %.not, label %146, label %145, !dbg !27
|
| 149 |
+
|
| 150 |
+
145: ; preds = %13
|
| 151 |
+
tail call void @__assertfail(ptr nonnull @assertMessage_0, ptr nonnull @assertFile_0, i32 32, ptr nonnull @assertFunc_0, i64 1), !dbg !27
|
| 152 |
+
unreachable, !dbg !27
|
| 153 |
+
|
| 154 |
+
146: ; preds = %13
|
| 155 |
+
%147 = insertelement <8 x i64> poison, i64 %8, i64 0, !dbg !28
|
| 156 |
+
%148 = shufflevector <8 x i64> %147, <8 x i64> poison, <8 x i32> zeroinitializer, !dbg !28
|
| 157 |
+
%149 = srem <8 x i64> %35, %148, !dbg !28
|
| 158 |
+
tail call void @llvm.nvvm.barrier.cta.sync.aligned.all(i32 0), !dbg !27
|
| 159 |
+
%150 = extractelement <8 x i64> %139, i64 0, !dbg !29
|
| 160 |
+
%151 = mul i64 %150, %8, !dbg !29
|
| 161 |
+
%152 = extractelement <8 x i64> %139, i64 1, !dbg !29
|
| 162 |
+
%153 = mul i64 %152, %8, !dbg !29
|
| 163 |
+
%154 = extractelement <8 x i64> %139, i64 2, !dbg !29
|
| 164 |
+
%155 = mul i64 %154, %8, !dbg !29
|
| 165 |
+
%156 = extractelement <8 x i64> %139, i64 3, !dbg !29
|
| 166 |
+
%157 = mul i64 %156, %8, !dbg !29
|
| 167 |
+
%158 = extractelement <8 x i64> %139, i64 4, !dbg !29
|
| 168 |
+
%159 = mul i64 %158, %8, !dbg !29
|
| 169 |
+
%160 = extractelement <8 x i64> %139, i64 5, !dbg !29
|
| 170 |
+
%161 = mul i64 %160, %8, !dbg !29
|
| 171 |
+
%162 = extractelement <8 x i64> %139, i64 6, !dbg !29
|
| 172 |
+
%163 = mul i64 %162, %8, !dbg !29
|
| 173 |
+
%164 = extractelement <8 x i64> %139, i64 7, !dbg !29
|
| 174 |
+
%165 = mul i64 %164, %8, !dbg !29
|
| 175 |
+
%166 = extractelement <8 x i64> %149, i64 7, !dbg !30
|
| 176 |
+
%167 = getelementptr bfloat, ptr addrspace(1) %2, i64 %166, !dbg !31
|
| 177 |
+
%168 = getelementptr bfloat, ptr addrspace(1) %167, i64 %151, !dbg !31
|
| 178 |
+
%169 = extractelement <8 x i64> %149, i64 6, !dbg !30
|
| 179 |
+
%170 = getelementptr bfloat, ptr addrspace(1) %2, i64 %169, !dbg !31
|
| 180 |
+
%171 = getelementptr bfloat, ptr addrspace(1) %170, i64 %153, !dbg !31
|
| 181 |
+
%172 = extractelement <8 x i64> %149, i64 5, !dbg !30
|
| 182 |
+
%173 = getelementptr bfloat, ptr addrspace(1) %2, i64 %172, !dbg !31
|
| 183 |
+
%174 = getelementptr bfloat, ptr addrspace(1) %173, i64 %155, !dbg !31
|
| 184 |
+
%175 = extractelement <8 x i64> %149, i64 4, !dbg !30
|
| 185 |
+
%176 = getelementptr bfloat, ptr addrspace(1) %2, i64 %175, !dbg !31
|
| 186 |
+
%177 = getelementptr bfloat, ptr addrspace(1) %176, i64 %157, !dbg !31
|
| 187 |
+
%178 = extractelement <8 x i64> %149, i64 3, !dbg !30
|
| 188 |
+
%179 = getelementptr bfloat, ptr addrspace(1) %2, i64 %178, !dbg !31
|
| 189 |
+
%180 = getelementptr bfloat, ptr addrspace(1) %179, i64 %159, !dbg !31
|
| 190 |
+
%181 = extractelement <8 x i64> %149, i64 2, !dbg !30
|
| 191 |
+
%182 = getelementptr bfloat, ptr addrspace(1) %2, i64 %181, !dbg !31
|
| 192 |
+
%183 = getelementptr bfloat, ptr addrspace(1) %182, i64 %161, !dbg !31
|
| 193 |
+
%184 = extractelement <8 x i64> %149, i64 1, !dbg !30
|
| 194 |
+
%185 = getelementptr bfloat, ptr addrspace(1) %2, i64 %184, !dbg !31
|
| 195 |
+
%186 = getelementptr bfloat, ptr addrspace(1) %185, i64 %163, !dbg !31
|
| 196 |
+
%187 = extractelement <8 x i64> %149, i64 0, !dbg !30
|
| 197 |
+
%188 = getelementptr bfloat, ptr addrspace(1) %2, i64 %187, !dbg !31
|
| 198 |
+
%189 = getelementptr bfloat, ptr addrspace(1) %188, i64 %165, !dbg !31
|
| 199 |
+
%190 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !32
|
| 200 |
+
%191 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %168, i64 %190, i1 %88) #4, !dbg !32
|
| 201 |
+
%192 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !32
|
| 202 |
+
%193 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %171, i64 %192, i1 %91) #4, !dbg !32
|
| 203 |
+
%194 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !32
|
| 204 |
+
%195 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %174, i64 %194, i1 %94) #4, !dbg !32
|
| 205 |
+
%196 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !32
|
| 206 |
+
%197 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %177, i64 %196, i1 %97) #4, !dbg !32
|
| 207 |
+
%198 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !32
|
| 208 |
+
%199 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %180, i64 %198, i1 %100) #4, !dbg !32
|
| 209 |
+
%200 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !32
|
| 210 |
+
%201 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %183, i64 %200, i1 %103) #4, !dbg !32
|
| 211 |
+
%202 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !32
|
| 212 |
+
%203 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %186, i64 %202, i1 %106) #4, !dbg !32
|
| 213 |
+
%204 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !32
|
| 214 |
+
%205 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %189, i64 %204, i1 %109) #4, !dbg !32
|
| 215 |
+
%206 = sdiv i64 %8, 2, !dbg !33
|
| 216 |
+
%207 = sub i64 %8, %206, !dbg !34
|
| 217 |
+
%208 = icmp slt i64 %166, %207, !dbg !35
|
| 218 |
+
%209 = icmp slt i64 %169, %207, !dbg !35
|
| 219 |
+
%210 = icmp slt i64 %172, %207, !dbg !35
|
| 220 |
+
%211 = icmp slt i64 %175, %207, !dbg !35
|
| 221 |
+
%212 = icmp slt i64 %178, %207, !dbg !35
|
| 222 |
+
%213 = icmp slt i64 %181, %207, !dbg !35
|
| 223 |
+
%214 = icmp slt i64 %184, %207, !dbg !35
|
| 224 |
+
%215 = icmp slt i64 %187, %207, !dbg !35
|
| 225 |
+
%216 = sub nsw i64 %36, %166, !dbg !30
|
| 226 |
+
%217 = sub nsw i64 %38, %169, !dbg !30
|
| 227 |
+
%218 = sub nsw i64 %40, %172, !dbg !30
|
| 228 |
+
%219 = sub nsw i64 %42, %175, !dbg !30
|
| 229 |
+
%220 = sub nsw i64 %44, %178, !dbg !30
|
| 230 |
+
%221 = sub nsw i64 %46, %181, !dbg !30
|
| 231 |
+
%222 = sub nsw i64 %48, %184, !dbg !30
|
| 232 |
+
%223 = sub nsw i64 %50, %187, !dbg !30
|
| 233 |
+
%224 = getelementptr bfloat, ptr addrspace(1) %0, i64 %216, !dbg !36
|
| 234 |
+
%225 = getelementptr bfloat, ptr addrspace(1) %224, i64 %206, !dbg !36
|
| 235 |
+
%226 = getelementptr bfloat, ptr addrspace(1) %225, i64 %166, !dbg !36
|
| 236 |
+
%227 = getelementptr bfloat, ptr addrspace(1) %0, i64 %217, !dbg !36
|
| 237 |
+
%228 = getelementptr bfloat, ptr addrspace(1) %227, i64 %206, !dbg !36
|
| 238 |
+
%229 = getelementptr bfloat, ptr addrspace(1) %228, i64 %169, !dbg !36
|
| 239 |
+
%230 = getelementptr bfloat, ptr addrspace(1) %0, i64 %218, !dbg !36
|
| 240 |
+
%231 = getelementptr bfloat, ptr addrspace(1) %230, i64 %206, !dbg !36
|
| 241 |
+
%232 = getelementptr bfloat, ptr addrspace(1) %231, i64 %172, !dbg !36
|
| 242 |
+
%233 = getelementptr bfloat, ptr addrspace(1) %0, i64 %219, !dbg !36
|
| 243 |
+
%234 = getelementptr bfloat, ptr addrspace(1) %233, i64 %206, !dbg !36
|
| 244 |
+
%235 = getelementptr bfloat, ptr addrspace(1) %234, i64 %175, !dbg !36
|
| 245 |
+
%236 = getelementptr bfloat, ptr addrspace(1) %0, i64 %220, !dbg !36
|
| 246 |
+
%237 = getelementptr bfloat, ptr addrspace(1) %236, i64 %206, !dbg !36
|
| 247 |
+
%238 = getelementptr bfloat, ptr addrspace(1) %237, i64 %178, !dbg !36
|
| 248 |
+
%239 = getelementptr bfloat, ptr addrspace(1) %0, i64 %221, !dbg !36
|
| 249 |
+
%240 = getelementptr bfloat, ptr addrspace(1) %239, i64 %206, !dbg !36
|
| 250 |
+
%241 = getelementptr bfloat, ptr addrspace(1) %240, i64 %181, !dbg !36
|
| 251 |
+
%242 = getelementptr bfloat, ptr addrspace(1) %0, i64 %222, !dbg !36
|
| 252 |
+
%243 = getelementptr bfloat, ptr addrspace(1) %242, i64 %206, !dbg !36
|
| 253 |
+
%244 = getelementptr bfloat, ptr addrspace(1) %243, i64 %184, !dbg !36
|
| 254 |
+
%245 = getelementptr bfloat, ptr addrspace(1) %0, i64 %223, !dbg !36
|
| 255 |
+
%246 = getelementptr bfloat, ptr addrspace(1) %245, i64 %206, !dbg !36
|
| 256 |
+
%247 = getelementptr bfloat, ptr addrspace(1) %246, i64 %187, !dbg !36
|
| 257 |
+
%248 = and i1 %88, %208, !dbg !37
|
| 258 |
+
%249 = and i1 %91, %209, !dbg !37
|
| 259 |
+
%250 = and i1 %94, %210, !dbg !37
|
| 260 |
+
%251 = and i1 %97, %211, !dbg !37
|
| 261 |
+
%252 = and i1 %100, %212, !dbg !37
|
| 262 |
+
%253 = and i1 %103, %213, !dbg !37
|
| 263 |
+
%254 = and i1 %106, %214, !dbg !37
|
| 264 |
+
%255 = and i1 %109, %215, !dbg !37
|
| 265 |
+
%256 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !38
|
| 266 |
+
%257 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %226, i64 %256, i1 %248) #4, !dbg !38
|
| 267 |
+
%258 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !38
|
| 268 |
+
%259 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %229, i64 %258, i1 %249) #4, !dbg !38
|
| 269 |
+
%260 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !38
|
| 270 |
+
%261 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %232, i64 %260, i1 %250) #4, !dbg !38
|
| 271 |
+
%262 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !38
|
| 272 |
+
%263 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %235, i64 %262, i1 %251) #4, !dbg !38
|
| 273 |
+
%264 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !38
|
| 274 |
+
%265 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %238, i64 %264, i1 %252) #4, !dbg !38
|
| 275 |
+
%266 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !38
|
| 276 |
+
%267 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %241, i64 %266, i1 %253) #4, !dbg !38
|
| 277 |
+
%268 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !38
|
| 278 |
+
%269 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %244, i64 %268, i1 %254) #4, !dbg !38
|
| 279 |
+
%270 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !38
|
| 280 |
+
%271 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %247, i64 %270, i1 %255) #4, !dbg !38
|
| 281 |
+
%272 = insertelement <8 x i64> poison, i64 %207, i64 0, !dbg !39
|
| 282 |
+
%273 = shufflevector <8 x i64> %272, <8 x i64> poison, <8 x i32> zeroinitializer, !dbg !39
|
| 283 |
+
%274 = icmp sge <8 x i64> %149, %273, !dbg !39
|
| 284 |
+
%275 = sub i64 %166, %8, !dbg !40
|
| 285 |
+
%276 = sub i64 %169, %8, !dbg !40
|
| 286 |
+
%277 = sub i64 %172, %8, !dbg !40
|
| 287 |
+
%278 = sub i64 %175, %8, !dbg !40
|
| 288 |
+
%279 = sub i64 %178, %8, !dbg !40
|
| 289 |
+
%280 = sub i64 %181, %8, !dbg !40
|
| 290 |
+
%281 = sub i64 %184, %8, !dbg !40
|
| 291 |
+
%282 = sub i64 %187, %8, !dbg !40
|
| 292 |
+
%283 = getelementptr bfloat, ptr addrspace(1) %224, i64 %275, !dbg !41
|
| 293 |
+
%284 = getelementptr bfloat, ptr addrspace(1) %283, i64 %206, !dbg !41
|
| 294 |
+
%285 = getelementptr bfloat, ptr addrspace(1) %227, i64 %276, !dbg !41
|
| 295 |
+
%286 = getelementptr bfloat, ptr addrspace(1) %285, i64 %206, !dbg !41
|
| 296 |
+
%287 = getelementptr bfloat, ptr addrspace(1) %230, i64 %277, !dbg !41
|
| 297 |
+
%288 = getelementptr bfloat, ptr addrspace(1) %287, i64 %206, !dbg !41
|
| 298 |
+
%289 = getelementptr bfloat, ptr addrspace(1) %233, i64 %278, !dbg !41
|
| 299 |
+
%290 = getelementptr bfloat, ptr addrspace(1) %289, i64 %206, !dbg !41
|
| 300 |
+
%291 = getelementptr bfloat, ptr addrspace(1) %236, i64 %279, !dbg !41
|
| 301 |
+
%292 = getelementptr bfloat, ptr addrspace(1) %291, i64 %206, !dbg !41
|
| 302 |
+
%293 = getelementptr bfloat, ptr addrspace(1) %239, i64 %280, !dbg !41
|
| 303 |
+
%294 = getelementptr bfloat, ptr addrspace(1) %293, i64 %206, !dbg !41
|
| 304 |
+
%295 = getelementptr bfloat, ptr addrspace(1) %242, i64 %281, !dbg !41
|
| 305 |
+
%296 = getelementptr bfloat, ptr addrspace(1) %295, i64 %206, !dbg !41
|
| 306 |
+
%297 = getelementptr bfloat, ptr addrspace(1) %245, i64 %282, !dbg !41
|
| 307 |
+
%298 = getelementptr bfloat, ptr addrspace(1) %297, i64 %206, !dbg !41
|
| 308 |
+
%299 = extractelement <8 x i1> %274, i64 7, !dbg !42
|
| 309 |
+
%300 = and i1 %88, %299, !dbg !42
|
| 310 |
+
%301 = extractelement <8 x i1> %274, i64 6, !dbg !42
|
| 311 |
+
%302 = and i1 %91, %301, !dbg !42
|
| 312 |
+
%303 = extractelement <8 x i1> %274, i64 5, !dbg !42
|
| 313 |
+
%304 = and i1 %94, %303, !dbg !42
|
| 314 |
+
%305 = extractelement <8 x i1> %274, i64 4, !dbg !42
|
| 315 |
+
%306 = and i1 %97, %305, !dbg !42
|
| 316 |
+
%307 = extractelement <8 x i1> %274, i64 3, !dbg !42
|
| 317 |
+
%308 = and i1 %100, %307, !dbg !42
|
| 318 |
+
%309 = extractelement <8 x i1> %274, i64 2, !dbg !42
|
| 319 |
+
%310 = and i1 %103, %309, !dbg !42
|
| 320 |
+
%311 = extractelement <8 x i1> %274, i64 1, !dbg !42
|
| 321 |
+
%312 = and i1 %106, %311, !dbg !42
|
| 322 |
+
%313 = extractelement <8 x i1> %274, i64 0, !dbg !42
|
| 323 |
+
%314 = and i1 %109, %313, !dbg !42
|
| 324 |
+
%315 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !43
|
| 325 |
+
%316 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %284, i64 %315, i1 %300) #4, !dbg !43
|
| 326 |
+
%317 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !43
|
| 327 |
+
%318 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %286, i64 %317, i1 %302) #4, !dbg !43
|
| 328 |
+
%319 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !43
|
| 329 |
+
%320 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %288, i64 %319, i1 %304) #4, !dbg !43
|
| 330 |
+
%321 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !43
|
| 331 |
+
%322 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %290, i64 %321, i1 %306) #4, !dbg !43
|
| 332 |
+
%323 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !43
|
| 333 |
+
%324 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %292, i64 %323, i1 %308) #4, !dbg !43
|
| 334 |
+
%325 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !43
|
| 335 |
+
%326 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %294, i64 %325, i1 %310) #4, !dbg !43
|
| 336 |
+
%327 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !43
|
| 337 |
+
%328 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %296, i64 %327, i1 %312) #4, !dbg !43
|
| 338 |
+
%329 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !43
|
| 339 |
+
%330 = tail call i16 asm sideeffect "mov.u16 $0, $1;\0A\09@$4 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $2 + 0 ], $3;", "=c,c,l,l,b"(i16 0, ptr addrspace(1) %298, i64 %329, i1 %314) #4, !dbg !43
|
| 340 |
+
%331 = insertelement <8 x i64> poison, i64 %9, i64 0, !dbg !44
|
| 341 |
+
%332 = shufflevector <8 x i64> %331, <8 x i64> poison, <8 x i32> zeroinitializer, !dbg !44
|
| 342 |
+
%333 = select <8 x i1> %135, <8 x i64> %332, <8 x i64> zeroinitializer, !dbg !44
|
| 343 |
+
%334 = add <8 x i64> %333, %134, !dbg !44
|
| 344 |
+
%335 = icmp slt <8 x i64> %334, zeroinitializer, !dbg !45
|
| 345 |
+
%336 = icmp sge <8 x i64> %334, %332, !dbg !46
|
| 346 |
+
%337 = or <8 x i1> %335, %336, !dbg !47
|
| 347 |
+
%338 = and <8 x i1> %87, %337, !dbg !48
|
| 348 |
+
%339 = bitcast <8 x i1> %338 to i8, !dbg !49
|
| 349 |
+
%.not87 = icmp eq i8 %339, 0, !dbg !49
|
| 350 |
+
br i1 %.not87, label %341, label %340, !dbg !49
|
| 351 |
+
|
| 352 |
+
340: ; preds = %146
|
| 353 |
+
tail call void @__assertfail(ptr nonnull @assertMessage_1, ptr nonnull @assertFile_1, i32 52, ptr nonnull @assertFunc_1, i64 1), !dbg !49
|
| 354 |
+
unreachable, !dbg !49
|
| 355 |
+
|
| 356 |
+
341: ; preds = %146
|
| 357 |
+
%342 = bitcast i16 %271 to bfloat, !dbg !38
|
| 358 |
+
%343 = fpext bfloat %342 to float, !dbg !50
|
| 359 |
+
%344 = fsub float 0.000000e+00, %343, !dbg !51
|
| 360 |
+
%345 = bitcast i16 %330 to bfloat, !dbg !43
|
| 361 |
+
%346 = fpext bfloat %345 to float, !dbg !52
|
| 362 |
+
%347 = select i1 %215, float %344, float %346, !dbg !53
|
| 363 |
+
%348 = bitcast i16 %269 to bfloat, !dbg !38
|
| 364 |
+
%349 = fpext bfloat %348 to float, !dbg !50
|
| 365 |
+
%350 = fsub float 0.000000e+00, %349, !dbg !51
|
| 366 |
+
%351 = bitcast i16 %328 to bfloat, !dbg !43
|
| 367 |
+
%352 = fpext bfloat %351 to float, !dbg !52
|
| 368 |
+
%353 = select i1 %214, float %350, float %352, !dbg !53
|
| 369 |
+
%354 = bitcast i16 %267 to bfloat, !dbg !38
|
| 370 |
+
%355 = fpext bfloat %354 to float, !dbg !50
|
| 371 |
+
%356 = fsub float 0.000000e+00, %355, !dbg !51
|
| 372 |
+
%357 = bitcast i16 %326 to bfloat, !dbg !43
|
| 373 |
+
%358 = fpext bfloat %357 to float, !dbg !52
|
| 374 |
+
%359 = select i1 %213, float %356, float %358, !dbg !53
|
| 375 |
+
%360 = bitcast i16 %265 to bfloat, !dbg !38
|
| 376 |
+
%361 = fpext bfloat %360 to float, !dbg !50
|
| 377 |
+
%362 = fsub float 0.000000e+00, %361, !dbg !51
|
| 378 |
+
%363 = bitcast i16 %324 to bfloat, !dbg !43
|
| 379 |
+
%364 = fpext bfloat %363 to float, !dbg !52
|
| 380 |
+
%365 = select i1 %212, float %362, float %364, !dbg !53
|
| 381 |
+
%366 = bitcast i16 %263 to bfloat, !dbg !38
|
| 382 |
+
%367 = fpext bfloat %366 to float, !dbg !50
|
| 383 |
+
%368 = fsub float 0.000000e+00, %367, !dbg !51
|
| 384 |
+
%369 = bitcast i16 %322 to bfloat, !dbg !43
|
| 385 |
+
%370 = fpext bfloat %369 to float, !dbg !52
|
| 386 |
+
%371 = select i1 %211, float %368, float %370, !dbg !53
|
| 387 |
+
%372 = bitcast i16 %261 to bfloat, !dbg !38
|
| 388 |
+
%373 = fpext bfloat %372 to float, !dbg !50
|
| 389 |
+
%374 = fsub float 0.000000e+00, %373, !dbg !51
|
| 390 |
+
%375 = bitcast i16 %320 to bfloat, !dbg !43
|
| 391 |
+
%376 = fpext bfloat %375 to float, !dbg !52
|
| 392 |
+
%377 = select i1 %210, float %374, float %376, !dbg !53
|
| 393 |
+
%378 = bitcast i16 %259 to bfloat, !dbg !38
|
| 394 |
+
%379 = fpext bfloat %378 to float, !dbg !50
|
| 395 |
+
%380 = fsub float 0.000000e+00, %379, !dbg !51
|
| 396 |
+
%381 = bitcast i16 %318 to bfloat, !dbg !43
|
| 397 |
+
%382 = fpext bfloat %381 to float, !dbg !52
|
| 398 |
+
%383 = select i1 %209, float %380, float %382, !dbg !53
|
| 399 |
+
%384 = bitcast i16 %257 to bfloat, !dbg !38
|
| 400 |
+
%385 = fpext bfloat %384 to float, !dbg !50
|
| 401 |
+
%386 = fsub float 0.000000e+00, %385, !dbg !51
|
| 402 |
+
%387 = bitcast i16 %316 to bfloat, !dbg !43
|
| 403 |
+
%388 = fpext bfloat %387 to float, !dbg !52
|
| 404 |
+
%389 = select i1 %208, float %386, float %388, !dbg !53
|
| 405 |
+
%390 = bitcast i16 %110 to bfloat, !dbg !17
|
| 406 |
+
%391 = fpext bfloat %390 to float, !dbg !54
|
| 407 |
+
%392 = bitcast i16 %107 to bfloat, !dbg !17
|
| 408 |
+
%393 = fpext bfloat %392 to float, !dbg !54
|
| 409 |
+
%394 = bitcast i16 %104 to bfloat, !dbg !17
|
| 410 |
+
%395 = fpext bfloat %394 to float, !dbg !54
|
| 411 |
+
%396 = bitcast i16 %101 to bfloat, !dbg !17
|
| 412 |
+
%397 = fpext bfloat %396 to float, !dbg !54
|
| 413 |
+
%398 = bitcast i16 %98 to bfloat, !dbg !17
|
| 414 |
+
%399 = fpext bfloat %398 to float, !dbg !54
|
| 415 |
+
%400 = bitcast i16 %95 to bfloat, !dbg !17
|
| 416 |
+
%401 = fpext bfloat %400 to float, !dbg !54
|
| 417 |
+
%402 = bitcast i16 %92 to bfloat, !dbg !17
|
| 418 |
+
%403 = fpext bfloat %402 to float, !dbg !54
|
| 419 |
+
%404 = bitcast i16 %89 to bfloat, !dbg !17
|
| 420 |
+
%405 = fpext bfloat %404 to float, !dbg !54
|
| 421 |
+
tail call void @llvm.nvvm.barrier.cta.sync.aligned.all(i32 0), !dbg !49
|
| 422 |
+
%406 = extractelement <8 x i64> %334, i64 0, !dbg !55
|
| 423 |
+
%407 = mul i64 %406, %8, !dbg !55
|
| 424 |
+
%408 = extractelement <8 x i64> %334, i64 1, !dbg !55
|
| 425 |
+
%409 = mul i64 %408, %8, !dbg !55
|
| 426 |
+
%410 = extractelement <8 x i64> %334, i64 2, !dbg !55
|
| 427 |
+
%411 = mul i64 %410, %8, !dbg !55
|
| 428 |
+
%412 = extractelement <8 x i64> %334, i64 3, !dbg !55
|
| 429 |
+
%413 = mul i64 %412, %8, !dbg !55
|
| 430 |
+
%414 = extractelement <8 x i64> %334, i64 4, !dbg !55
|
| 431 |
+
%415 = mul i64 %414, %8, !dbg !55
|
| 432 |
+
%416 = extractelement <8 x i64> %334, i64 5, !dbg !55
|
| 433 |
+
%417 = mul i64 %416, %8, !dbg !55
|
| 434 |
+
%418 = extractelement <8 x i64> %334, i64 6, !dbg !55
|
| 435 |
+
%419 = mul i64 %418, %8, !dbg !55
|
| 436 |
+
%420 = extractelement <8 x i64> %334, i64 7, !dbg !55
|
| 437 |
+
%421 = mul i64 %420, %8, !dbg !55
|
| 438 |
+
%422 = getelementptr bfloat, ptr addrspace(1) %3, i64 %166, !dbg !56
|
| 439 |
+
%423 = getelementptr bfloat, ptr addrspace(1) %422, i64 %407, !dbg !56
|
| 440 |
+
%424 = getelementptr bfloat, ptr addrspace(1) %3, i64 %169, !dbg !56
|
| 441 |
+
%425 = getelementptr bfloat, ptr addrspace(1) %424, i64 %409, !dbg !56
|
| 442 |
+
%426 = getelementptr bfloat, ptr addrspace(1) %3, i64 %172, !dbg !56
|
| 443 |
+
%427 = getelementptr bfloat, ptr addrspace(1) %426, i64 %411, !dbg !56
|
| 444 |
+
%428 = getelementptr bfloat, ptr addrspace(1) %3, i64 %175, !dbg !56
|
| 445 |
+
%429 = getelementptr bfloat, ptr addrspace(1) %428, i64 %413, !dbg !56
|
| 446 |
+
%430 = getelementptr bfloat, ptr addrspace(1) %3, i64 %178, !dbg !56
|
| 447 |
+
%431 = getelementptr bfloat, ptr addrspace(1) %430, i64 %415, !dbg !56
|
| 448 |
+
%432 = getelementptr bfloat, ptr addrspace(1) %3, i64 %181, !dbg !56
|
| 449 |
+
%433 = getelementptr bfloat, ptr addrspace(1) %432, i64 %417, !dbg !56
|
| 450 |
+
%434 = getelementptr bfloat, ptr addrspace(1) %3, i64 %184, !dbg !56
|
| 451 |
+
%435 = getelementptr bfloat, ptr addrspace(1) %434, i64 %419, !dbg !56
|
| 452 |
+
%436 = getelementptr bfloat, ptr addrspace(1) %3, i64 %187, !dbg !56
|
| 453 |
+
%437 = getelementptr bfloat, ptr addrspace(1) %436, i64 %421, !dbg !56
|
| 454 |
+
%438 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !57
|
| 455 |
+
%439 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %423, i64 %438, i1 %88) #4, !dbg !57
|
| 456 |
+
%440 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !57
|
| 457 |
+
%441 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %425, i64 %440, i1 %91) #4, !dbg !57
|
| 458 |
+
%442 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !57
|
| 459 |
+
%443 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %427, i64 %442, i1 %94) #4, !dbg !57
|
| 460 |
+
%444 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !57
|
| 461 |
+
%445 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %429, i64 %444, i1 %97) #4, !dbg !57
|
| 462 |
+
%446 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !57
|
| 463 |
+
%447 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %431, i64 %446, i1 %100) #4, !dbg !57
|
| 464 |
+
%448 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !57
|
| 465 |
+
%449 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %433, i64 %448, i1 %103) #4, !dbg !57
|
| 466 |
+
%450 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !57
|
| 467 |
+
%451 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %435, i64 %450, i1 %106) #4, !dbg !57
|
| 468 |
+
%452 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #4, !dbg !57
|
| 469 |
+
%453 = tail call i16 asm sideeffect "mov.u16 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b16 { $0 }, [ $1 + 0 ], $2;", "=c,l,l,b"(ptr addrspace(1) %437, i64 %452, i1 %109) #4, !dbg !57
|
| 470 |
+
%454 = insertelement <2 x i16> poison, i16 %191, i64 0, !dbg !32
|
| 471 |
+
%455 = insertelement <2 x i16> %454, i16 %439, i64 1, !dbg !32
|
| 472 |
+
%456 = bitcast <2 x i16> %455 to <2 x bfloat>, !dbg !32
|
| 473 |
+
%457 = fpext <2 x bfloat> %456 to <2 x float>, !dbg !58
|
| 474 |
+
%458 = insertelement <2 x float> poison, float %405, i64 0, !dbg !59
|
| 475 |
+
%459 = insertelement <2 x float> %458, float %389, i64 1, !dbg !59
|
| 476 |
+
%460 = fmul <2 x float> %459, %457, !dbg !59
|
| 477 |
+
%461 = insertelement <2 x i16> poison, i16 %193, i64 0, !dbg !32
|
| 478 |
+
%462 = insertelement <2 x i16> %461, i16 %441, i64 1, !dbg !32
|
| 479 |
+
%463 = bitcast <2 x i16> %462 to <2 x bfloat>, !dbg !32
|
| 480 |
+
%464 = fpext <2 x bfloat> %463 to <2 x float>, !dbg !58
|
| 481 |
+
%465 = insertelement <2 x float> poison, float %403, i64 0, !dbg !59
|
| 482 |
+
%466 = insertelement <2 x float> %465, float %383, i64 1, !dbg !59
|
| 483 |
+
%467 = fmul <2 x float> %466, %464, !dbg !59
|
| 484 |
+
%468 = insertelement <2 x i16> poison, i16 %195, i64 0, !dbg !32
|
| 485 |
+
%469 = insertelement <2 x i16> %468, i16 %443, i64 1, !dbg !32
|
| 486 |
+
%470 = bitcast <2 x i16> %469 to <2 x bfloat>, !dbg !32
|
| 487 |
+
%471 = fpext <2 x bfloat> %470 to <2 x float>, !dbg !58
|
| 488 |
+
%472 = insertelement <2 x float> poison, float %401, i64 0, !dbg !59
|
| 489 |
+
%473 = insertelement <2 x float> %472, float %377, i64 1, !dbg !59
|
| 490 |
+
%474 = fmul <2 x float> %473, %471, !dbg !59
|
| 491 |
+
%475 = insertelement <2 x i16> poison, i16 %197, i64 0, !dbg !32
|
| 492 |
+
%476 = insertelement <2 x i16> %475, i16 %445, i64 1, !dbg !32
|
| 493 |
+
%477 = bitcast <2 x i16> %476 to <2 x bfloat>, !dbg !32
|
| 494 |
+
%478 = fpext <2 x bfloat> %477 to <2 x float>, !dbg !58
|
| 495 |
+
%479 = insertelement <2 x float> poison, float %399, i64 0, !dbg !59
|
| 496 |
+
%480 = insertelement <2 x float> %479, float %371, i64 1, !dbg !59
|
| 497 |
+
%481 = fmul <2 x float> %480, %478, !dbg !59
|
| 498 |
+
%482 = insertelement <2 x i16> poison, i16 %199, i64 0, !dbg !32
|
| 499 |
+
%483 = insertelement <2 x i16> %482, i16 %447, i64 1, !dbg !32
|
| 500 |
+
%484 = bitcast <2 x i16> %483 to <2 x bfloat>, !dbg !32
|
| 501 |
+
%485 = fpext <2 x bfloat> %484 to <2 x float>, !dbg !58
|
| 502 |
+
%486 = insertelement <2 x float> poison, float %397, i64 0, !dbg !59
|
| 503 |
+
%487 = insertelement <2 x float> %486, float %365, i64 1, !dbg !59
|
| 504 |
+
%488 = fmul <2 x float> %487, %485, !dbg !59
|
| 505 |
+
%489 = insertelement <2 x i16> poison, i16 %201, i64 0, !dbg !32
|
| 506 |
+
%490 = insertelement <2 x i16> %489, i16 %449, i64 1, !dbg !32
|
| 507 |
+
%491 = bitcast <2 x i16> %490 to <2 x bfloat>, !dbg !32
|
| 508 |
+
%492 = fpext <2 x bfloat> %491 to <2 x float>, !dbg !58
|
| 509 |
+
%493 = insertelement <2 x float> poison, float %395, i64 0, !dbg !59
|
| 510 |
+
%494 = insertelement <2 x float> %493, float %359, i64 1, !dbg !59
|
| 511 |
+
%495 = fmul <2 x float> %494, %492, !dbg !59
|
| 512 |
+
%496 = insertelement <2 x i16> poison, i16 %203, i64 0, !dbg !32
|
| 513 |
+
%497 = insertelement <2 x i16> %496, i16 %451, i64 1, !dbg !32
|
| 514 |
+
%498 = bitcast <2 x i16> %497 to <2 x bfloat>, !dbg !32
|
| 515 |
+
%499 = fpext <2 x bfloat> %498 to <2 x float>, !dbg !58
|
| 516 |
+
%500 = insertelement <2 x float> poison, float %393, i64 0, !dbg !59
|
| 517 |
+
%501 = insertelement <2 x float> %500, float %353, i64 1, !dbg !59
|
| 518 |
+
%502 = fmul <2 x float> %501, %499, !dbg !59
|
| 519 |
+
%503 = insertelement <2 x i16> poison, i16 %205, i64 0, !dbg !32
|
| 520 |
+
%504 = insertelement <2 x i16> %503, i16 %453, i64 1, !dbg !32
|
| 521 |
+
%505 = bitcast <2 x i16> %504 to <2 x bfloat>, !dbg !32
|
| 522 |
+
%506 = fpext <2 x bfloat> %505 to <2 x float>, !dbg !58
|
| 523 |
+
%507 = insertelement <2 x float> poison, float %391, i64 0, !dbg !59
|
| 524 |
+
%508 = insertelement <2 x float> %507, float %347, i64 1, !dbg !59
|
| 525 |
+
%509 = fmul <2 x float> %508, %506, !dbg !59
|
| 526 |
+
%shift = shufflevector <2 x float> %460, <2 x float> poison, <2 x i32> <i32 1, i32 poison>, !dbg !60
|
| 527 |
+
%foldExtExtBinop = fadd <2 x float> %460, %shift, !dbg !60
|
| 528 |
+
%510 = extractelement <2 x float> %foldExtExtBinop, i64 0, !dbg !60
|
| 529 |
+
%shift66 = shufflevector <2 x float> %467, <2 x float> poison, <2 x i32> <i32 1, i32 poison>, !dbg !60
|
| 530 |
+
%foldExtExtBinop67 = fadd <2 x float> %467, %shift66, !dbg !60
|
| 531 |
+
%511 = extractelement <2 x float> %foldExtExtBinop67, i64 0, !dbg !60
|
| 532 |
+
%shift69 = shufflevector <2 x float> %474, <2 x float> poison, <2 x i32> <i32 1, i32 poison>, !dbg !60
|
| 533 |
+
%foldExtExtBinop70 = fadd <2 x float> %474, %shift69, !dbg !60
|
| 534 |
+
%512 = extractelement <2 x float> %foldExtExtBinop70, i64 0, !dbg !60
|
| 535 |
+
%shift72 = shufflevector <2 x float> %481, <2 x float> poison, <2 x i32> <i32 1, i32 poison>, !dbg !60
|
| 536 |
+
%foldExtExtBinop73 = fadd <2 x float> %481, %shift72, !dbg !60
|
| 537 |
+
%513 = extractelement <2 x float> %foldExtExtBinop73, i64 0, !dbg !60
|
| 538 |
+
%shift75 = shufflevector <2 x float> %488, <2 x float> poison, <2 x i32> <i32 1, i32 poison>, !dbg !60
|
| 539 |
+
%foldExtExtBinop76 = fadd <2 x float> %488, %shift75, !dbg !60
|
| 540 |
+
%514 = extractelement <2 x float> %foldExtExtBinop76, i64 0, !dbg !60
|
| 541 |
+
%shift78 = shufflevector <2 x float> %495, <2 x float> poison, <2 x i32> <i32 1, i32 poison>, !dbg !60
|
| 542 |
+
%foldExtExtBinop79 = fadd <2 x float> %495, %shift78, !dbg !60
|
| 543 |
+
%515 = extractelement <2 x float> %foldExtExtBinop79, i64 0, !dbg !60
|
| 544 |
+
%shift81 = shufflevector <2 x float> %502, <2 x float> poison, <2 x i32> <i32 1, i32 poison>, !dbg !60
|
| 545 |
+
%foldExtExtBinop82 = fadd <2 x float> %502, %shift81, !dbg !60
|
| 546 |
+
%516 = extractelement <2 x float> %foldExtExtBinop82, i64 0, !dbg !60
|
| 547 |
+
%shift84 = shufflevector <2 x float> %509, <2 x float> poison, <2 x i32> <i32 1, i32 poison>, !dbg !60
|
| 548 |
+
%foldExtExtBinop85 = fadd <2 x float> %509, %shift84, !dbg !60
|
| 549 |
+
%517 = extractelement <2 x float> %foldExtExtBinop85, i64 0, !dbg !60
|
| 550 |
+
%518 = getelementptr bfloat, ptr addrspace(1) %4, i64 %36, !dbg !61
|
| 551 |
+
%519 = getelementptr bfloat, ptr addrspace(1) %4, i64 %38, !dbg !61
|
| 552 |
+
%520 = getelementptr bfloat, ptr addrspace(1) %4, i64 %40, !dbg !61
|
| 553 |
+
%521 = getelementptr bfloat, ptr addrspace(1) %4, i64 %42, !dbg !61
|
| 554 |
+
%522 = getelementptr bfloat, ptr addrspace(1) %4, i64 %44, !dbg !61
|
| 555 |
+
%523 = getelementptr bfloat, ptr addrspace(1) %4, i64 %46, !dbg !61
|
| 556 |
+
%524 = getelementptr bfloat, ptr addrspace(1) %4, i64 %48, !dbg !61
|
| 557 |
+
%525 = getelementptr bfloat, ptr addrspace(1) %4, i64 %50, !dbg !61
|
| 558 |
+
%526 = fptrunc float %510 to bfloat, !dbg !62
|
| 559 |
+
%527 = fptrunc float %511 to bfloat, !dbg !62
|
| 560 |
+
%528 = fptrunc float %512 to bfloat, !dbg !62
|
| 561 |
+
%529 = fptrunc float %513 to bfloat, !dbg !62
|
| 562 |
+
%530 = fptrunc float %514 to bfloat, !dbg !62
|
| 563 |
+
%531 = fptrunc float %515 to bfloat, !dbg !62
|
| 564 |
+
%532 = fptrunc float %516 to bfloat, !dbg !62
|
| 565 |
+
%533 = fptrunc float %517 to bfloat, !dbg !62
|
| 566 |
+
%534 = bitcast bfloat %526 to i16, !dbg !62
|
| 567 |
+
tail call void asm sideeffect "@$2 st.global.b16 [ $1 + 0 ], { $0 };", "c,l,b"(i16 %534, ptr addrspace(1) %518, i1 %88) #4, !dbg !62
|
| 568 |
+
%535 = bitcast bfloat %527 to i16, !dbg !62
|
| 569 |
+
tail call void asm sideeffect "@$2 st.global.b16 [ $1 + 0 ], { $0 };", "c,l,b"(i16 %535, ptr addrspace(1) %519, i1 %91) #4, !dbg !62
|
| 570 |
+
%536 = bitcast bfloat %528 to i16, !dbg !62
|
| 571 |
+
tail call void asm sideeffect "@$2 st.global.b16 [ $1 + 0 ], { $0 };", "c,l,b"(i16 %536, ptr addrspace(1) %520, i1 %94) #4, !dbg !62
|
| 572 |
+
%537 = bitcast bfloat %529 to i16, !dbg !62
|
| 573 |
+
tail call void asm sideeffect "@$2 st.global.b16 [ $1 + 0 ], { $0 };", "c,l,b"(i16 %537, ptr addrspace(1) %521, i1 %97) #4, !dbg !62
|
| 574 |
+
%538 = bitcast bfloat %530 to i16, !dbg !62
|
| 575 |
+
tail call void asm sideeffect "@$2 st.global.b16 [ $1 + 0 ], { $0 };", "c,l,b"(i16 %538, ptr addrspace(1) %522, i1 %100) #4, !dbg !62
|
| 576 |
+
%539 = bitcast bfloat %531 to i16, !dbg !62
|
| 577 |
+
tail call void asm sideeffect "@$2 st.global.b16 [ $1 + 0 ], { $0 };", "c,l,b"(i16 %539, ptr addrspace(1) %523, i1 %103) #4, !dbg !62
|
| 578 |
+
%540 = bitcast bfloat %532 to i16, !dbg !62
|
| 579 |
+
tail call void asm sideeffect "@$2 st.global.b16 [ $1 + 0 ], { $0 };", "c,l,b"(i16 %540, ptr addrspace(1) %524, i1 %106) #4, !dbg !62
|
| 580 |
+
%541 = bitcast bfloat %533 to i16, !dbg !62
|
| 581 |
+
tail call void asm sideeffect "@$2 st.global.b16 [ $1 + 0 ], { $0 };", "c,l,b"(i16 %541, ptr addrspace(1) %525, i1 %109) #4, !dbg !62
|
| 582 |
+
ret void, !dbg !63
|
| 583 |
+
}
|
| 584 |
+
|
| 585 |
+
; Function Attrs: mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none)
|
| 586 |
+
declare noundef range(i32 0, 2147483647) i32 @llvm.nvvm.read.ptx.sreg.ctaid.x() #2
|
| 587 |
+
|
| 588 |
+
; Function Attrs: mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none)
|
| 589 |
+
declare noundef range(i32 0, 1024) i32 @llvm.nvvm.read.ptx.sreg.tid.x() #2
|
| 590 |
+
|
| 591 |
+
; Function Attrs: convergent nocallback nounwind
|
| 592 |
+
declare void @llvm.nvvm.barrier.cta.sync.aligned.all(i32) #3
|
| 593 |
+
|
| 594 |
+
attributes #0 = { noreturn }
|
| 595 |
+
attributes #1 = { "nvvm.reqntid"="128" }
|
| 596 |
+
attributes #2 = { mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none) }
|
| 597 |
+
attributes #3 = { convergent nocallback nounwind }
|
| 598 |
+
attributes #4 = { nounwind }
|
| 599 |
+
|
| 600 |
+
!llvm.dbg.cu = !{!0}
|
| 601 |
+
!llvm.module.flags = !{!2, !3}
|
| 602 |
+
!llvm.ident = !{!4}
|
| 603 |
+
|
| 604 |
+
!0 = distinct !DICompileUnit(language: DW_LANG_C, file: !1, producer: "triton", isOptimized: true, runtimeVersion: 0, emissionKind: LineTablesOnly)
|
| 605 |
+
!1 = !DIFile(filename: "cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py", directory: "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al")
|
| 606 |
+
!2 = !{i32 2, !"Debug Info Version", i32 3}
|
| 607 |
+
!3 = !{i32 4, !"nvvm-reflect-ftz", i32 1}
|
| 608 |
+
!4 = !{!"clang version 3.8.0 (tags/RELEASE_380/final)"}
|
| 609 |
+
!5 = !DISubprogram(name: "__assertfail", linkageName: "__assertfail", scope: !6, file: !6, type: !7, spFlags: DISPFlagOptimized)
|
| 610 |
+
!6 = !DIFile(filename: "<unknown>", directory: "")
|
| 611 |
+
!7 = !DISubroutineType(cc: DW_CC_normal, types: !8)
|
| 612 |
+
!8 = !{}
|
| 613 |
+
!9 = distinct !DISubprogram(name: "triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0", linkageName: "triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0", scope: !1, file: !1, line: 18, type: !7, scopeLine: 18, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
|
| 614 |
+
!10 = !DILocation(line: 19, column: 28, scope: !9)
|
| 615 |
+
!11 = !DILocation(line: 19, column: 33, scope: !9)
|
| 616 |
+
!12 = !DILocation(line: 20, column: 36, scope: !9)
|
| 617 |
+
!13 = !DILocation(line: 20, column: 23, scope: !9)
|
| 618 |
+
!14 = !DILocation(line: 23, column: 21, scope: !9)
|
| 619 |
+
!15 = !DILocation(line: 26, column: 30, scope: !9)
|
| 620 |
+
!16 = !DILocation(line: 23, column: 28, scope: !9)
|
| 621 |
+
!17 = !DILocation(line: 26, column: 35, scope: !9)
|
| 622 |
+
!18 = !DILocation(line: 27, column: 30, scope: !9)
|
| 623 |
+
!19 = !DILocation(line: 21, column: 21, scope: !9)
|
| 624 |
+
!20 = !DILocation(line: 27, column: 35, scope: !9)
|
| 625 |
+
!21 = !DILocation(line: 30, column: 18, scope: !9)
|
| 626 |
+
!22 = !DILocation(line: 31, column: 32, scope: !9)
|
| 627 |
+
!23 = !DILocation(line: 32, column: 28, scope: !9)
|
| 628 |
+
!24 = !DILocation(line: 32, column: 44, scope: !9)
|
| 629 |
+
!25 = !DILocation(line: 32, column: 37, scope: !9)
|
| 630 |
+
!26 = !DILocation(line: 32, column: 52, scope: !9)
|
| 631 |
+
!27 = !DILocation(line: 32, column: 62, scope: !9)
|
| 632 |
+
!28 = !DILocation(line: 24, column: 19, scope: !9)
|
| 633 |
+
!29 = !DILocation(line: 33, column: 39, scope: !9)
|
| 634 |
+
!30 = !DILocation(line: 40, column: 35, scope: !9)
|
| 635 |
+
!31 = !DILocation(line: 33, column: 30, scope: !9)
|
| 636 |
+
!32 = !DILocation(line: 33, column: 46, scope: !9)
|
| 637 |
+
!33 = !DILocation(line: 38, column: 31, scope: !9)
|
| 638 |
+
!34 = !DILocation(line: 38, column: 18, scope: !9)
|
| 639 |
+
!35 = !DILocation(line: 39, column: 19, scope: !9)
|
| 640 |
+
!36 = !DILocation(line: 40, column: 31, scope: !9)
|
| 641 |
+
!37 = !DILocation(line: 40, column: 68, scope: !9)
|
| 642 |
+
!38 = !DILocation(line: 40, column: 60, scope: !9)
|
| 643 |
+
!39 = !DILocation(line: 44, column: 20, scope: !9)
|
| 644 |
+
!40 = !DILocation(line: 47, column: 47, scope: !9)
|
| 645 |
+
!41 = !DILocation(line: 47, column: 31, scope: !9)
|
| 646 |
+
!42 = !DILocation(line: 47, column: 81, scope: !9)
|
| 647 |
+
!43 = !DILocation(line: 47, column: 73, scope: !9)
|
| 648 |
+
!44 = !DILocation(line: 51, column: 34, scope: !9)
|
| 649 |
+
!45 = !DILocation(line: 52, column: 28, scope: !9)
|
| 650 |
+
!46 = !DILocation(line: 52, column: 46, scope: !9)
|
| 651 |
+
!47 = !DILocation(line: 52, column: 38, scope: !9)
|
| 652 |
+
!48 = !DILocation(line: 52, column: 54, scope: !9)
|
| 653 |
+
!49 = !DILocation(line: 52, column: 64, scope: !9)
|
| 654 |
+
!50 = !DILocation(line: 40, column: 119, scope: !9)
|
| 655 |
+
!51 = !DILocation(line: 41, column: 13, scope: !9)
|
| 656 |
+
!52 = !DILocation(line: 47, column: 132, scope: !9)
|
| 657 |
+
!53 = !DILocation(line: 0, scope: !9)
|
| 658 |
+
!54 = !DILocation(line: 26, column: 75, scope: !9)
|
| 659 |
+
!55 = !DILocation(line: 53, column: 40, scope: !9)
|
| 660 |
+
!56 = !DILocation(line: 53, column: 31, scope: !9)
|
| 661 |
+
!57 = !DILocation(line: 53, column: 48, scope: !9)
|
| 662 |
+
!58 = !DILocation(line: 33, column: 86, scope: !9)
|
| 663 |
+
!59 = !DILocation(line: 34, column: 18, scope: !9)
|
| 664 |
+
!60 = !DILocation(line: 55, column: 19, scope: !9)
|
| 665 |
+
!61 = !DILocation(line: 56, column: 25, scope: !9)
|
| 666 |
+
!62 = !DILocation(line: 56, column: 37, scope: !9)
|
| 667 |
+
!63 = !DILocation(line: 56, column: 4, scope: !9)
|
SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ptx
ADDED
|
@@ -0,0 +1,1534 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
//
|
| 2 |
+
// Generated by LLVM NVPTX Back-End
|
| 3 |
+
//
|
| 4 |
+
|
| 5 |
+
.version 8.7
|
| 6 |
+
.target sm_90a
|
| 7 |
+
.address_size 64
|
| 8 |
+
|
| 9 |
+
// .globl triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0 // -- Begin function triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0
|
| 10 |
+
.extern .func __assertfail
|
| 11 |
+
(
|
| 12 |
+
.param .b64 __assertfail_param_0,
|
| 13 |
+
.param .b64 __assertfail_param_1,
|
| 14 |
+
.param .b32 __assertfail_param_2,
|
| 15 |
+
.param .b64 __assertfail_param_3,
|
| 16 |
+
.param .b64 __assertfail_param_4
|
| 17 |
+
)
|
| 18 |
+
.noreturn;
|
| 19 |
+
.global .align 1 .b8 assertFunc_1[8] = {117, 110, 107, 110, 111, 119, 110};
|
| 20 |
+
.global .align 1 .b8 assertFile_1[114] = {47, 119, 111, 114, 107, 115, 112, 97, 99, 101, 47, 104, 97, 110, 114, 117, 105, 47, 83, 112, 101, 99, 70, 111, 114, 103, 101, 45, 101, 120, 116, 47, 99, 97, 99, 104, 101, 47, 99, 111, 109, 112, 105, 108, 101, 100, 95, 107, 101, 114, 110, 101, 108, 115, 47, 97, 108, 47, 99, 97, 108, 50, 114, 52, 116, 102, 121, 119, 54, 103, 105, 99, 51, 103, 103, 113, 121, 117, 100, 51, 110, 117, 102, 110, 97, 106, 120, 54, 120, 97, 117, 50, 107, 111, 105, 101, 111, 105, 116, 120, 54, 122, 103, 52, 119, 115, 105, 111, 122, 109, 46, 112, 121};
|
| 21 |
+
.global .align 1 .b8 assertMessage_1[38] = {105, 110, 100, 101, 120, 32, 111, 117, 116, 32, 111, 102, 32, 98, 111, 117, 110, 100, 115, 58, 32, 48, 32, 60, 61, 32, 116, 109, 112, 50, 53, 32, 60, 32, 107, 115, 52};
|
| 22 |
+
.global .align 1 .b8 assertFunc_0[8] = {117, 110, 107, 110, 111, 119, 110};
|
| 23 |
+
.global .align 1 .b8 assertFile_0[114] = {47, 119, 111, 114, 107, 115, 112, 97, 99, 101, 47, 104, 97, 110, 114, 117, 105, 47, 83, 112, 101, 99, 70, 111, 114, 103, 101, 45, 101, 120, 116, 47, 99, 97, 99, 104, 101, 47, 99, 111, 109, 112, 105, 108, 101, 100, 95, 107, 101, 114, 110, 101, 108, 115, 47, 97, 108, 47, 99, 97, 108, 50, 114, 52, 116, 102, 121, 119, 54, 103, 105, 99, 51, 103, 103, 113, 121, 117, 100, 51, 110, 117, 102, 110, 97, 106, 120, 54, 120, 97, 117, 50, 107, 111, 105, 101, 111, 105, 116, 120, 54, 122, 103, 52, 119, 115, 105, 111, 122, 109, 46, 112, 121};
|
| 24 |
+
.global .align 1 .b8 assertMessage_0[37] = {105, 110, 100, 101, 120, 32, 111, 117, 116, 32, 111, 102, 32, 98, 111, 117, 110, 100, 115, 58, 32, 48, 32, 60, 61, 32, 116, 109, 112, 53, 32, 60, 32, 107, 115, 50};
|
| 25 |
+
// @triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0
|
| 26 |
+
.visible .entry triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0(
|
| 27 |
+
.param .u64 .ptr .global .align 1 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_0,
|
| 28 |
+
.param .u64 .ptr .global .align 1 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_1,
|
| 29 |
+
.param .u64 .ptr .global .align 1 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_2,
|
| 30 |
+
.param .u64 .ptr .global .align 1 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_3,
|
| 31 |
+
.param .u64 .ptr .global .align 1 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_4,
|
| 32 |
+
.param .u64 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_5,
|
| 33 |
+
.param .u64 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_6,
|
| 34 |
+
.param .u64 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_7,
|
| 35 |
+
.param .u64 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_8,
|
| 36 |
+
.param .u64 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_9,
|
| 37 |
+
.param .u32 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_10,
|
| 38 |
+
.param .u64 .ptr .global .align 1 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_11,
|
| 39 |
+
.param .u64 .ptr .global .align 1 triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_12
|
| 40 |
+
)
|
| 41 |
+
.reqntid 128
|
| 42 |
+
{
|
| 43 |
+
.reg .pred %p<179>;
|
| 44 |
+
.reg .b16 %rs<165>;
|
| 45 |
+
.reg .b32 %r<160>;
|
| 46 |
+
.reg .b64 %rd<500>;
|
| 47 |
+
.loc 1 18 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:18:0
|
| 48 |
+
$L__func_begin0:
|
| 49 |
+
.loc 1 18 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:18:0
|
| 50 |
+
|
| 51 |
+
// %bb.0:
|
| 52 |
+
ld.param.b64 %rd103, [triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_5];
|
| 53 |
+
$L__tmp0:
|
| 54 |
+
.loc 1 19 28 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:19:28
|
| 55 |
+
mov.u32 %r26, %ctaid.x;
|
| 56 |
+
.loc 1 19 33 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:19:33
|
| 57 |
+
shl.b32 %r27, %r26, 10;
|
| 58 |
+
.loc 1 20 36 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:20:36
|
| 59 |
+
mov.u32 %r28, %tid.x;
|
| 60 |
+
shl.b32 %r29, %r28, 3;
|
| 61 |
+
and.b32 %r30, %r29, 1016;
|
| 62 |
+
.loc 1 20 23 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:20:23
|
| 63 |
+
or.b32 %r1, %r30, %r27;
|
| 64 |
+
or.b32 %r2, %r1, 1;
|
| 65 |
+
or.b32 %r3, %r1, 2;
|
| 66 |
+
.loc 1 23 21 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:23:21
|
| 67 |
+
cvt.s64.s32 %rd7, %r2;
|
| 68 |
+
cvt.s64.s32 %rd8, %r1;
|
| 69 |
+
or.b64 %rd108, %rd8, %rd103;
|
| 70 |
+
and.b64 %rd109, %rd108, -4294967296;
|
| 71 |
+
setp.ne.b64 %p9, %rd109, 0;
|
| 72 |
+
@%p9 bra $L__BB0_2;
|
| 73 |
+
bra.uni $L__BB0_1;
|
| 74 |
+
$L__BB0_2:
|
| 75 |
+
div.s64 %rd484, %rd8, %rd103;
|
| 76 |
+
bra.uni $L__BB0_3;
|
| 77 |
+
$L__BB0_1:
|
| 78 |
+
cvt.u32.u64 %r31, %rd103;
|
| 79 |
+
cvt.u32.u64 %r32, %rd8;
|
| 80 |
+
div.u32 %r33, %r32, %r31;
|
| 81 |
+
cvt.u64.u32 %rd484, %r33;
|
| 82 |
+
$L__BB0_3:
|
| 83 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 84 |
+
or.b32 %r4, %r1, 3;
|
| 85 |
+
.loc 1 23 21 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:23:21
|
| 86 |
+
cvt.s64.s32 %rd6, %r3;
|
| 87 |
+
or.b64 %rd110, %rd7, %rd103;
|
| 88 |
+
and.b64 %rd111, %rd110, -4294967296;
|
| 89 |
+
setp.ne.b64 %p10, %rd111, 0;
|
| 90 |
+
@%p10 bra $L__BB0_5;
|
| 91 |
+
bra.uni $L__BB0_4;
|
| 92 |
+
$L__BB0_5:
|
| 93 |
+
div.s64 %rd485, %rd7, %rd103;
|
| 94 |
+
bra.uni $L__BB0_6;
|
| 95 |
+
$L__BB0_4:
|
| 96 |
+
cvt.u32.u64 %r34, %rd103;
|
| 97 |
+
cvt.u32.u64 %r35, %rd7;
|
| 98 |
+
div.u32 %r36, %r35, %r34;
|
| 99 |
+
cvt.u64.u32 %rd485, %r36;
|
| 100 |
+
$L__BB0_6:
|
| 101 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 102 |
+
or.b32 %r5, %r1, 4;
|
| 103 |
+
.loc 1 23 21 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:23:21
|
| 104 |
+
cvt.s64.s32 %rd5, %r4;
|
| 105 |
+
or.b64 %rd112, %rd6, %rd103;
|
| 106 |
+
and.b64 %rd113, %rd112, -4294967296;
|
| 107 |
+
setp.ne.b64 %p11, %rd113, 0;
|
| 108 |
+
@%p11 bra $L__BB0_8;
|
| 109 |
+
bra.uni $L__BB0_7;
|
| 110 |
+
$L__BB0_8:
|
| 111 |
+
div.s64 %rd486, %rd6, %rd103;
|
| 112 |
+
bra.uni $L__BB0_9;
|
| 113 |
+
$L__BB0_7:
|
| 114 |
+
cvt.u32.u64 %r37, %rd103;
|
| 115 |
+
cvt.u32.u64 %r38, %rd6;
|
| 116 |
+
div.u32 %r39, %r38, %r37;
|
| 117 |
+
cvt.u64.u32 %rd486, %r39;
|
| 118 |
+
$L__BB0_9:
|
| 119 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 120 |
+
or.b32 %r6, %r1, 5;
|
| 121 |
+
.loc 1 23 21 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:23:21
|
| 122 |
+
cvt.s64.s32 %rd4, %r5;
|
| 123 |
+
or.b64 %rd114, %rd5, %rd103;
|
| 124 |
+
and.b64 %rd115, %rd114, -4294967296;
|
| 125 |
+
setp.ne.b64 %p12, %rd115, 0;
|
| 126 |
+
@%p12 bra $L__BB0_11;
|
| 127 |
+
bra.uni $L__BB0_10;
|
| 128 |
+
$L__BB0_11:
|
| 129 |
+
div.s64 %rd487, %rd5, %rd103;
|
| 130 |
+
bra.uni $L__BB0_12;
|
| 131 |
+
$L__BB0_10:
|
| 132 |
+
cvt.u32.u64 %r40, %rd103;
|
| 133 |
+
cvt.u32.u64 %r41, %rd5;
|
| 134 |
+
div.u32 %r42, %r41, %r40;
|
| 135 |
+
cvt.u64.u32 %rd487, %r42;
|
| 136 |
+
$L__BB0_12:
|
| 137 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 138 |
+
or.b32 %r7, %r1, 6;
|
| 139 |
+
.loc 1 23 21 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:23:21
|
| 140 |
+
cvt.s64.s32 %rd3, %r6;
|
| 141 |
+
or.b64 %rd116, %rd4, %rd103;
|
| 142 |
+
and.b64 %rd117, %rd116, -4294967296;
|
| 143 |
+
setp.ne.b64 %p13, %rd117, 0;
|
| 144 |
+
@%p13 bra $L__BB0_14;
|
| 145 |
+
bra.uni $L__BB0_13;
|
| 146 |
+
$L__BB0_14:
|
| 147 |
+
div.s64 %rd488, %rd4, %rd103;
|
| 148 |
+
bra.uni $L__BB0_15;
|
| 149 |
+
$L__BB0_13:
|
| 150 |
+
cvt.u32.u64 %r43, %rd103;
|
| 151 |
+
cvt.u32.u64 %r44, %rd4;
|
| 152 |
+
div.u32 %r45, %r44, %r43;
|
| 153 |
+
cvt.u64.u32 %rd488, %r45;
|
| 154 |
+
$L__BB0_15:
|
| 155 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 156 |
+
or.b32 %r8, %r1, 7;
|
| 157 |
+
.loc 1 23 21 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:23:21
|
| 158 |
+
cvt.s64.s32 %rd2, %r7;
|
| 159 |
+
or.b64 %rd118, %rd3, %rd103;
|
| 160 |
+
and.b64 %rd119, %rd118, -4294967296;
|
| 161 |
+
setp.ne.b64 %p14, %rd119, 0;
|
| 162 |
+
@%p14 bra $L__BB0_17;
|
| 163 |
+
bra.uni $L__BB0_16;
|
| 164 |
+
$L__BB0_17:
|
| 165 |
+
div.s64 %rd489, %rd3, %rd103;
|
| 166 |
+
bra.uni $L__BB0_18;
|
| 167 |
+
$L__BB0_16:
|
| 168 |
+
cvt.u32.u64 %r46, %rd103;
|
| 169 |
+
cvt.u32.u64 %r47, %rd3;
|
| 170 |
+
div.u32 %r48, %r47, %r46;
|
| 171 |
+
cvt.u64.u32 %rd489, %r48;
|
| 172 |
+
$L__BB0_18:
|
| 173 |
+
cvt.s64.s32 %rd1, %r8;
|
| 174 |
+
or.b64 %rd120, %rd2, %rd103;
|
| 175 |
+
and.b64 %rd121, %rd120, -4294967296;
|
| 176 |
+
setp.ne.b64 %p15, %rd121, 0;
|
| 177 |
+
@%p15 bra $L__BB0_20;
|
| 178 |
+
bra.uni $L__BB0_19;
|
| 179 |
+
$L__BB0_20:
|
| 180 |
+
div.s64 %rd490, %rd2, %rd103;
|
| 181 |
+
bra.uni $L__BB0_21;
|
| 182 |
+
$L__BB0_19:
|
| 183 |
+
cvt.u32.u64 %r49, %rd103;
|
| 184 |
+
cvt.u32.u64 %r50, %rd2;
|
| 185 |
+
div.u32 %r51, %r50, %r49;
|
| 186 |
+
cvt.u64.u32 %rd490, %r51;
|
| 187 |
+
$L__BB0_21:
|
| 188 |
+
.loc 1 0 21 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0:21
|
| 189 |
+
ld.param.b64 %rd104, [triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_6];
|
| 190 |
+
.loc 1 23 21 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:23:21
|
| 191 |
+
or.b64 %rd122, %rd1, %rd103;
|
| 192 |
+
and.b64 %rd123, %rd122, -4294967296;
|
| 193 |
+
setp.ne.b64 %p16, %rd123, 0;
|
| 194 |
+
@%p16 bra $L__BB0_23;
|
| 195 |
+
bra.uni $L__BB0_22;
|
| 196 |
+
$L__BB0_23:
|
| 197 |
+
div.s64 %rd491, %rd1, %rd103;
|
| 198 |
+
bra.uni $L__BB0_24;
|
| 199 |
+
$L__BB0_22:
|
| 200 |
+
cvt.u32.u64 %r52, %rd103;
|
| 201 |
+
cvt.u32.u64 %r53, %rd1;
|
| 202 |
+
div.u32 %r54, %r53, %r52;
|
| 203 |
+
cvt.u64.u32 %rd491, %r54;
|
| 204 |
+
$L__BB0_24:
|
| 205 |
+
.loc 1 23 28 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:23:28
|
| 206 |
+
or.b64 %rd124, %rd484, %rd104;
|
| 207 |
+
and.b64 %rd125, %rd124, -4294967296;
|
| 208 |
+
setp.ne.b64 %p17, %rd125, 0;
|
| 209 |
+
@%p17 bra $L__BB0_26;
|
| 210 |
+
bra.uni $L__BB0_25;
|
| 211 |
+
$L__BB0_26:
|
| 212 |
+
rem.s64 %rd492, %rd484, %rd104;
|
| 213 |
+
bra.uni $L__BB0_27;
|
| 214 |
+
$L__BB0_25:
|
| 215 |
+
cvt.u32.u64 %r55, %rd104;
|
| 216 |
+
cvt.u32.u64 %r56, %rd484;
|
| 217 |
+
rem.u32 %r57, %r56, %r55;
|
| 218 |
+
cvt.u64.u32 %rd492, %r57;
|
| 219 |
+
$L__BB0_27:
|
| 220 |
+
or.b64 %rd126, %rd485, %rd104;
|
| 221 |
+
and.b64 %rd127, %rd126, -4294967296;
|
| 222 |
+
setp.ne.b64 %p18, %rd127, 0;
|
| 223 |
+
@%p18 bra $L__BB0_29;
|
| 224 |
+
bra.uni $L__BB0_28;
|
| 225 |
+
$L__BB0_29:
|
| 226 |
+
rem.s64 %rd493, %rd485, %rd104;
|
| 227 |
+
bra.uni $L__BB0_30;
|
| 228 |
+
$L__BB0_28:
|
| 229 |
+
cvt.u32.u64 %r58, %rd104;
|
| 230 |
+
cvt.u32.u64 %r59, %rd485;
|
| 231 |
+
rem.u32 %r60, %r59, %r58;
|
| 232 |
+
cvt.u64.u32 %rd493, %r60;
|
| 233 |
+
$L__BB0_30:
|
| 234 |
+
or.b64 %rd128, %rd486, %rd104;
|
| 235 |
+
and.b64 %rd129, %rd128, -4294967296;
|
| 236 |
+
setp.ne.b64 %p19, %rd129, 0;
|
| 237 |
+
@%p19 bra $L__BB0_32;
|
| 238 |
+
bra.uni $L__BB0_31;
|
| 239 |
+
$L__BB0_32:
|
| 240 |
+
rem.s64 %rd494, %rd486, %rd104;
|
| 241 |
+
bra.uni $L__BB0_33;
|
| 242 |
+
$L__BB0_31:
|
| 243 |
+
cvt.u32.u64 %r61, %rd104;
|
| 244 |
+
cvt.u32.u64 %r62, %rd486;
|
| 245 |
+
rem.u32 %r63, %r62, %r61;
|
| 246 |
+
cvt.u64.u32 %rd494, %r63;
|
| 247 |
+
$L__BB0_33:
|
| 248 |
+
or.b64 %rd130, %rd487, %rd104;
|
| 249 |
+
and.b64 %rd131, %rd130, -4294967296;
|
| 250 |
+
setp.ne.b64 %p20, %rd131, 0;
|
| 251 |
+
@%p20 bra $L__BB0_35;
|
| 252 |
+
bra.uni $L__BB0_34;
|
| 253 |
+
$L__BB0_35:
|
| 254 |
+
rem.s64 %rd495, %rd487, %rd104;
|
| 255 |
+
bra.uni $L__BB0_36;
|
| 256 |
+
$L__BB0_34:
|
| 257 |
+
cvt.u32.u64 %r64, %rd104;
|
| 258 |
+
cvt.u32.u64 %r65, %rd487;
|
| 259 |
+
rem.u32 %r66, %r65, %r64;
|
| 260 |
+
cvt.u64.u32 %rd495, %r66;
|
| 261 |
+
$L__BB0_36:
|
| 262 |
+
or.b64 %rd132, %rd488, %rd104;
|
| 263 |
+
and.b64 %rd133, %rd132, -4294967296;
|
| 264 |
+
setp.ne.b64 %p21, %rd133, 0;
|
| 265 |
+
@%p21 bra $L__BB0_38;
|
| 266 |
+
bra.uni $L__BB0_37;
|
| 267 |
+
$L__BB0_38:
|
| 268 |
+
rem.s64 %rd496, %rd488, %rd104;
|
| 269 |
+
bra.uni $L__BB0_39;
|
| 270 |
+
$L__BB0_37:
|
| 271 |
+
cvt.u32.u64 %r67, %rd104;
|
| 272 |
+
cvt.u32.u64 %r68, %rd488;
|
| 273 |
+
rem.u32 %r69, %r68, %r67;
|
| 274 |
+
cvt.u64.u32 %rd496, %r69;
|
| 275 |
+
$L__BB0_39:
|
| 276 |
+
or.b64 %rd134, %rd489, %rd104;
|
| 277 |
+
and.b64 %rd135, %rd134, -4294967296;
|
| 278 |
+
setp.ne.b64 %p22, %rd135, 0;
|
| 279 |
+
@%p22 bra $L__BB0_41;
|
| 280 |
+
bra.uni $L__BB0_40;
|
| 281 |
+
$L__BB0_41:
|
| 282 |
+
rem.s64 %rd497, %rd489, %rd104;
|
| 283 |
+
bra.uni $L__BB0_42;
|
| 284 |
+
$L__BB0_40:
|
| 285 |
+
cvt.u32.u64 %r70, %rd104;
|
| 286 |
+
cvt.u32.u64 %r71, %rd489;
|
| 287 |
+
rem.u32 %r72, %r71, %r70;
|
| 288 |
+
cvt.u64.u32 %rd497, %r72;
|
| 289 |
+
$L__BB0_42:
|
| 290 |
+
or.b64 %rd136, %rd490, %rd104;
|
| 291 |
+
and.b64 %rd137, %rd136, -4294967296;
|
| 292 |
+
setp.ne.b64 %p23, %rd137, 0;
|
| 293 |
+
@%p23 bra $L__BB0_44;
|
| 294 |
+
bra.uni $L__BB0_43;
|
| 295 |
+
$L__BB0_44:
|
| 296 |
+
rem.s64 %rd498, %rd490, %rd104;
|
| 297 |
+
bra.uni $L__BB0_45;
|
| 298 |
+
$L__BB0_43:
|
| 299 |
+
cvt.u32.u64 %r73, %rd104;
|
| 300 |
+
cvt.u32.u64 %r74, %rd490;
|
| 301 |
+
rem.u32 %r75, %r74, %r73;
|
| 302 |
+
cvt.u64.u32 %rd498, %r75;
|
| 303 |
+
$L__BB0_45:
|
| 304 |
+
.loc 1 0 28 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0:28
|
| 305 |
+
ld.param.b32 %r25, [triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_10];
|
| 306 |
+
ld.param.b64 %rd105, [triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_7];
|
| 307 |
+
ld.param.b64 %rd99, [triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_1];
|
| 308 |
+
ld.param.b64 %rd98, [triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_0];
|
| 309 |
+
.loc 1 23 28 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:23:28
|
| 310 |
+
or.b64 %rd138, %rd491, %rd104;
|
| 311 |
+
and.b64 %rd139, %rd138, -4294967296;
|
| 312 |
+
setp.ne.b64 %p24, %rd139, 0;
|
| 313 |
+
@%p24 bra $L__BB0_47;
|
| 314 |
+
bra.uni $L__BB0_46;
|
| 315 |
+
$L__BB0_47:
|
| 316 |
+
rem.s64 %rd499, %rd491, %rd104;
|
| 317 |
+
bra.uni $L__BB0_48;
|
| 318 |
+
$L__BB0_46:
|
| 319 |
+
cvt.u32.u64 %r76, %rd104;
|
| 320 |
+
cvt.u32.u64 %r77, %rd491;
|
| 321 |
+
rem.u32 %r78, %r77, %r76;
|
| 322 |
+
cvt.u64.u32 %rd499, %r78;
|
| 323 |
+
$L__BB0_48:
|
| 324 |
+
.loc 1 26 30 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:26:30
|
| 325 |
+
shl.b64 %rd196, %rd8, 1;
|
| 326 |
+
add.s64 %rd141, %rd98, %rd196;
|
| 327 |
+
shl.b64 %rd197, %rd7, 1;
|
| 328 |
+
add.s64 %rd144, %rd98, %rd197;
|
| 329 |
+
shl.b64 %rd198, %rd6, 1;
|
| 330 |
+
add.s64 %rd147, %rd98, %rd198;
|
| 331 |
+
shl.b64 %rd199, %rd5, 1;
|
| 332 |
+
add.s64 %rd150, %rd98, %rd199;
|
| 333 |
+
shl.b64 %rd200, %rd4, 1;
|
| 334 |
+
add.s64 %rd153, %rd98, %rd200;
|
| 335 |
+
shl.b64 %rd201, %rd3, 1;
|
| 336 |
+
add.s64 %rd156, %rd98, %rd201;
|
| 337 |
+
shl.b64 %rd202, %rd2, 1;
|
| 338 |
+
add.s64 %rd159, %rd98, %rd202;
|
| 339 |
+
shl.b64 %rd203, %rd1, 1;
|
| 340 |
+
add.s64 %rd162, %rd98, %rd203;
|
| 341 |
+
.loc 1 26 35 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:26:35
|
| 342 |
+
// begin inline asm
|
| 343 |
+
mov.u64 %rd140, 0x0;
|
| 344 |
+
createpolicy.fractional.L2::evict_last.b64 %rd140, 1.0;
|
| 345 |
+
// end inline asm
|
| 346 |
+
.loc 1 27 30 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:27:30
|
| 347 |
+
shl.b64 %rd204, %rd492, 3;
|
| 348 |
+
add.s64 %rd166, %rd99, %rd204;
|
| 349 |
+
shl.b64 %rd205, %rd493, 3;
|
| 350 |
+
add.s64 %rd170, %rd99, %rd205;
|
| 351 |
+
shl.b64 %rd206, %rd494, 3;
|
| 352 |
+
add.s64 %rd174, %rd99, %rd206;
|
| 353 |
+
shl.b64 %rd207, %rd495, 3;
|
| 354 |
+
add.s64 %rd178, %rd99, %rd207;
|
| 355 |
+
shl.b64 %rd208, %rd496, 3;
|
| 356 |
+
add.s64 %rd182, %rd99, %rd208;
|
| 357 |
+
shl.b64 %rd209, %rd497, 3;
|
| 358 |
+
add.s64 %rd186, %rd99, %rd209;
|
| 359 |
+
shl.b64 %rd210, %rd498, 3;
|
| 360 |
+
add.s64 %rd190, %rd99, %rd210;
|
| 361 |
+
shl.b64 %rd211, %rd499, 3;
|
| 362 |
+
add.s64 %rd194, %rd99, %rd211;
|
| 363 |
+
.loc 1 21 21 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:21:21
|
| 364 |
+
setp.lt.s32 %p8, %r8, %r25;
|
| 365 |
+
setp.lt.s32 %p7, %r7, %r25;
|
| 366 |
+
setp.lt.s32 %p6, %r6, %r25;
|
| 367 |
+
setp.lt.s32 %p5, %r5, %r25;
|
| 368 |
+
setp.lt.s32 %p4, %r4, %r25;
|
| 369 |
+
setp.lt.s32 %p3, %r3, %r25;
|
| 370 |
+
setp.lt.s32 %p2, %r2, %r25;
|
| 371 |
+
setp.lt.s32 %p1, %r1, %r25;
|
| 372 |
+
.loc 1 26 35 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:26:35
|
| 373 |
+
// begin inline asm
|
| 374 |
+
mov.u16 %rs33, 0x0;
|
| 375 |
+
@%p1 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs33 }, [ %rd141 + 0 ], %rd140;
|
| 376 |
+
// end inline asm
|
| 377 |
+
// begin inline asm
|
| 378 |
+
mov.u64 %rd143, 0x0;
|
| 379 |
+
createpolicy.fractional.L2::evict_last.b64 %rd143, 1.0;
|
| 380 |
+
// end inline asm
|
| 381 |
+
// begin inline asm
|
| 382 |
+
mov.u16 %rs34, 0x0;
|
| 383 |
+
@%p2 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs34 }, [ %rd144 + 0 ], %rd143;
|
| 384 |
+
// end inline asm
|
| 385 |
+
// begin inline asm
|
| 386 |
+
mov.u64 %rd146, 0x0;
|
| 387 |
+
createpolicy.fractional.L2::evict_last.b64 %rd146, 1.0;
|
| 388 |
+
// end inline asm
|
| 389 |
+
// begin inline asm
|
| 390 |
+
mov.u16 %rs35, 0x0;
|
| 391 |
+
@%p3 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs35 }, [ %rd147 + 0 ], %rd146;
|
| 392 |
+
// end inline asm
|
| 393 |
+
// begin inline asm
|
| 394 |
+
mov.u64 %rd149, 0x0;
|
| 395 |
+
createpolicy.fractional.L2::evict_last.b64 %rd149, 1.0;
|
| 396 |
+
// end inline asm
|
| 397 |
+
// begin inline asm
|
| 398 |
+
mov.u16 %rs36, 0x0;
|
| 399 |
+
@%p4 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs36 }, [ %rd150 + 0 ], %rd149;
|
| 400 |
+
// end inline asm
|
| 401 |
+
// begin inline asm
|
| 402 |
+
mov.u64 %rd152, 0x0;
|
| 403 |
+
createpolicy.fractional.L2::evict_last.b64 %rd152, 1.0;
|
| 404 |
+
// end inline asm
|
| 405 |
+
// begin inline asm
|
| 406 |
+
mov.u16 %rs37, 0x0;
|
| 407 |
+
@%p5 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs37 }, [ %rd153 + 0 ], %rd152;
|
| 408 |
+
// end inline asm
|
| 409 |
+
// begin inline asm
|
| 410 |
+
mov.u64 %rd155, 0x0;
|
| 411 |
+
createpolicy.fractional.L2::evict_last.b64 %rd155, 1.0;
|
| 412 |
+
// end inline asm
|
| 413 |
+
// begin inline asm
|
| 414 |
+
mov.u16 %rs38, 0x0;
|
| 415 |
+
@%p6 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs38 }, [ %rd156 + 0 ], %rd155;
|
| 416 |
+
// end inline asm
|
| 417 |
+
// begin inline asm
|
| 418 |
+
mov.u64 %rd158, 0x0;
|
| 419 |
+
createpolicy.fractional.L2::evict_last.b64 %rd158, 1.0;
|
| 420 |
+
// end inline asm
|
| 421 |
+
// begin inline asm
|
| 422 |
+
mov.u16 %rs39, 0x0;
|
| 423 |
+
@%p7 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs39 }, [ %rd159 + 0 ], %rd158;
|
| 424 |
+
// end inline asm
|
| 425 |
+
// begin inline asm
|
| 426 |
+
mov.u64 %rd161, 0x0;
|
| 427 |
+
createpolicy.fractional.L2::evict_last.b64 %rd161, 1.0;
|
| 428 |
+
// end inline asm
|
| 429 |
+
// begin inline asm
|
| 430 |
+
mov.u16 %rs40, 0x0;
|
| 431 |
+
@%p8 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs40 }, [ %rd162 + 0 ], %rd161;
|
| 432 |
+
// end inline asm
|
| 433 |
+
.loc 1 27 35 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:27:35
|
| 434 |
+
// begin inline asm
|
| 435 |
+
mov.u64 %rd164, 0x0;
|
| 436 |
+
createpolicy.fractional.L2::evict_last.b64 %rd164, 1.0;
|
| 437 |
+
// end inline asm
|
| 438 |
+
// begin inline asm
|
| 439 |
+
mov.u64 %rd165, 0x0;
|
| 440 |
+
@%p1 ld.global.L1::evict_last.L2::cache_hint.b64 { %rd165 }, [ %rd166 + 0 ], %rd164;
|
| 441 |
+
// end inline asm
|
| 442 |
+
// begin inline asm
|
| 443 |
+
mov.u64 %rd168, 0x0;
|
| 444 |
+
createpolicy.fractional.L2::evict_last.b64 %rd168, 1.0;
|
| 445 |
+
// end inline asm
|
| 446 |
+
// begin inline asm
|
| 447 |
+
mov.u64 %rd169, 0x0;
|
| 448 |
+
@%p2 ld.global.L1::evict_last.L2::cache_hint.b64 { %rd169 }, [ %rd170 + 0 ], %rd168;
|
| 449 |
+
// end inline asm
|
| 450 |
+
// begin inline asm
|
| 451 |
+
mov.u64 %rd172, 0x0;
|
| 452 |
+
createpolicy.fractional.L2::evict_last.b64 %rd172, 1.0;
|
| 453 |
+
// end inline asm
|
| 454 |
+
// begin inline asm
|
| 455 |
+
mov.u64 %rd173, 0x0;
|
| 456 |
+
@%p3 ld.global.L1::evict_last.L2::cache_hint.b64 { %rd173 }, [ %rd174 + 0 ], %rd172;
|
| 457 |
+
// end inline asm
|
| 458 |
+
// begin inline asm
|
| 459 |
+
mov.u64 %rd176, 0x0;
|
| 460 |
+
createpolicy.fractional.L2::evict_last.b64 %rd176, 1.0;
|
| 461 |
+
// end inline asm
|
| 462 |
+
// begin inline asm
|
| 463 |
+
mov.u64 %rd177, 0x0;
|
| 464 |
+
@%p4 ld.global.L1::evict_last.L2::cache_hint.b64 { %rd177 }, [ %rd178 + 0 ], %rd176;
|
| 465 |
+
// end inline asm
|
| 466 |
+
// begin inline asm
|
| 467 |
+
mov.u64 %rd180, 0x0;
|
| 468 |
+
createpolicy.fractional.L2::evict_last.b64 %rd180, 1.0;
|
| 469 |
+
// end inline asm
|
| 470 |
+
// begin inline asm
|
| 471 |
+
mov.u64 %rd181, 0x0;
|
| 472 |
+
@%p5 ld.global.L1::evict_last.L2::cache_hint.b64 { %rd181 }, [ %rd182 + 0 ], %rd180;
|
| 473 |
+
// end inline asm
|
| 474 |
+
// begin inline asm
|
| 475 |
+
mov.u64 %rd184, 0x0;
|
| 476 |
+
createpolicy.fractional.L2::evict_last.b64 %rd184, 1.0;
|
| 477 |
+
// end inline asm
|
| 478 |
+
// begin inline asm
|
| 479 |
+
mov.u64 %rd185, 0x0;
|
| 480 |
+
@%p6 ld.global.L1::evict_last.L2::cache_hint.b64 { %rd185 }, [ %rd186 + 0 ], %rd184;
|
| 481 |
+
// end inline asm
|
| 482 |
+
// begin inline asm
|
| 483 |
+
mov.u64 %rd188, 0x0;
|
| 484 |
+
createpolicy.fractional.L2::evict_last.b64 %rd188, 1.0;
|
| 485 |
+
// end inline asm
|
| 486 |
+
// begin inline asm
|
| 487 |
+
mov.u64 %rd189, 0x0;
|
| 488 |
+
@%p7 ld.global.L1::evict_last.L2::cache_hint.b64 { %rd189 }, [ %rd190 + 0 ], %rd188;
|
| 489 |
+
// end inline asm
|
| 490 |
+
// begin inline asm
|
| 491 |
+
mov.u64 %rd192, 0x0;
|
| 492 |
+
createpolicy.fractional.L2::evict_last.b64 %rd192, 1.0;
|
| 493 |
+
// end inline asm
|
| 494 |
+
// begin inline asm
|
| 495 |
+
mov.u64 %rd193, 0x0;
|
| 496 |
+
@%p8 ld.global.L1::evict_last.L2::cache_hint.b64 { %rd193 }, [ %rd194 + 0 ], %rd192;
|
| 497 |
+
// end inline asm
|
| 498 |
+
.loc 1 31 32 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:31:32
|
| 499 |
+
shr.s64 %rd212, %rd173, 63;
|
| 500 |
+
and.b64 %rd213, %rd212, %rd105;
|
| 501 |
+
shr.s64 %rd214, %rd177, 63;
|
| 502 |
+
and.b64 %rd215, %rd214, %rd105;
|
| 503 |
+
shr.s64 %rd216, %rd165, 63;
|
| 504 |
+
and.b64 %rd217, %rd216, %rd105;
|
| 505 |
+
shr.s64 %rd218, %rd169, 63;
|
| 506 |
+
and.b64 %rd219, %rd218, %rd105;
|
| 507 |
+
shr.s64 %rd220, %rd189, 63;
|
| 508 |
+
and.b64 %rd221, %rd220, %rd105;
|
| 509 |
+
shr.s64 %rd222, %rd193, 63;
|
| 510 |
+
and.b64 %rd223, %rd222, %rd105;
|
| 511 |
+
shr.s64 %rd224, %rd181, 63;
|
| 512 |
+
and.b64 %rd225, %rd224, %rd105;
|
| 513 |
+
shr.s64 %rd226, %rd185, 63;
|
| 514 |
+
and.b64 %rd227, %rd226, %rd105;
|
| 515 |
+
add.s64 %rd78, %rd227, %rd185;
|
| 516 |
+
add.s64 %rd77, %rd225, %rd181;
|
| 517 |
+
add.s64 %rd80, %rd223, %rd193;
|
| 518 |
+
add.s64 %rd79, %rd221, %rd189;
|
| 519 |
+
add.s64 %rd74, %rd219, %rd169;
|
| 520 |
+
add.s64 %rd73, %rd217, %rd165;
|
| 521 |
+
add.s64 %rd76, %rd215, %rd177;
|
| 522 |
+
add.s64 %rd75, %rd213, %rd173;
|
| 523 |
+
.loc 1 32 28 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:32:28
|
| 524 |
+
setp.lt.s64 %p41, %rd75, 0;
|
| 525 |
+
setp.lt.s64 %p42, %rd76, 0;
|
| 526 |
+
setp.lt.s64 %p43, %rd73, 0;
|
| 527 |
+
setp.lt.s64 %p44, %rd74, 0;
|
| 528 |
+
setp.lt.s64 %p45, %rd79, 0;
|
| 529 |
+
setp.lt.s64 %p46, %rd80, 0;
|
| 530 |
+
setp.lt.s64 %p47, %rd77, 0;
|
| 531 |
+
setp.lt.s64 %p48, %rd78, 0;
|
| 532 |
+
.loc 1 32 44 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:32:44
|
| 533 |
+
setp.ge.s64 %p49, %rd75, %rd105;
|
| 534 |
+
setp.ge.s64 %p50, %rd76, %rd105;
|
| 535 |
+
setp.ge.s64 %p51, %rd73, %rd105;
|
| 536 |
+
setp.ge.s64 %p52, %rd74, %rd105;
|
| 537 |
+
setp.ge.s64 %p53, %rd79, %rd105;
|
| 538 |
+
setp.ge.s64 %p54, %rd80, %rd105;
|
| 539 |
+
setp.ge.s64 %p55, %rd77, %rd105;
|
| 540 |
+
setp.ge.s64 %p56, %rd78, %rd105;
|
| 541 |
+
.loc 1 32 37 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:32:37
|
| 542 |
+
or.pred %p57, %p48, %p56;
|
| 543 |
+
or.pred %p58, %p47, %p55;
|
| 544 |
+
or.pred %p59, %p46, %p54;
|
| 545 |
+
or.pred %p60, %p45, %p53;
|
| 546 |
+
or.pred %p61, %p44, %p52;
|
| 547 |
+
or.pred %p62, %p43, %p51;
|
| 548 |
+
or.pred %p63, %p42, %p50;
|
| 549 |
+
or.pred %p64, %p41, %p49;
|
| 550 |
+
.loc 1 32 52 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:32:52
|
| 551 |
+
and.pred %p65, %p3, %p64;
|
| 552 |
+
selp.b16 %rs41, 1, 0, %p65;
|
| 553 |
+
shl.b16 %rs42, %rs41, 2;
|
| 554 |
+
and.pred %p66, %p4, %p63;
|
| 555 |
+
selp.b16 %rs43, -1, 0, %p66;
|
| 556 |
+
shl.b16 %rs44, %rs43, 3;
|
| 557 |
+
or.b16 %rs45, %rs44, %rs42;
|
| 558 |
+
and.pred %p67, %p1, %p62;
|
| 559 |
+
selp.b16 %rs46, 1, 0, %p67;
|
| 560 |
+
and.pred %p68, %p2, %p61;
|
| 561 |
+
selp.b16 %rs47, -1, 0, %p68;
|
| 562 |
+
shl.b16 %rs48, %rs47, 1;
|
| 563 |
+
or.b16 %rs49, %rs46, %rs48;
|
| 564 |
+
and.b16 %rs50, %rs49, 3;
|
| 565 |
+
or.b16 %rs51, %rs50, %rs45;
|
| 566 |
+
and.b16 %rs52, %rs51, 15;
|
| 567 |
+
and.pred %p69, %p7, %p60;
|
| 568 |
+
selp.b16 %rs53, 1, 0, %p69;
|
| 569 |
+
shl.b16 %rs54, %rs53, 2;
|
| 570 |
+
and.pred %p70, %p8, %p59;
|
| 571 |
+
selp.b16 %rs55, -1, 0, %p70;
|
| 572 |
+
shl.b16 %rs56, %rs55, 3;
|
| 573 |
+
or.b16 %rs57, %rs56, %rs54;
|
| 574 |
+
and.pred %p71, %p5, %p58;
|
| 575 |
+
selp.b16 %rs58, 1, 0, %p71;
|
| 576 |
+
and.pred %p72, %p6, %p57;
|
| 577 |
+
selp.b16 %rs59, -1, 0, %p72;
|
| 578 |
+
shl.b16 %rs60, %rs59, 1;
|
| 579 |
+
or.b16 %rs61, %rs58, %rs60;
|
| 580 |
+
and.b16 %rs62, %rs61, 3;
|
| 581 |
+
or.b16 %rs63, %rs62, %rs57;
|
| 582 |
+
shl.b16 %rs64, %rs63, 4;
|
| 583 |
+
or.b16 %rs65, %rs52, %rs64;
|
| 584 |
+
.loc 1 32 62 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:32:62
|
| 585 |
+
and.b16 %rs66, %rs65, 255;
|
| 586 |
+
setp.eq.b16 %p73, %rs66, 0;
|
| 587 |
+
@%p73 bra $L__BB0_50;
|
| 588 |
+
bra.uni $L__BB0_49;
|
| 589 |
+
$L__BB0_50:
|
| 590 |
+
.loc 1 0 62 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0:62
|
| 591 |
+
ld.param.b64 %rd107, [triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_9];
|
| 592 |
+
ld.param.b64 %rd106, [triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_8];
|
| 593 |
+
ld.param.b64 %rd100, [triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_2];
|
| 594 |
+
.loc 1 24 19 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:24:19
|
| 595 |
+
rem.s64 %rd88, %rd1, %rd106;
|
| 596 |
+
rem.s64 %rd87, %rd2, %rd106;
|
| 597 |
+
rem.s64 %rd86, %rd3, %rd106;
|
| 598 |
+
rem.s64 %rd85, %rd4, %rd106;
|
| 599 |
+
rem.s64 %rd84, %rd5, %rd106;
|
| 600 |
+
rem.s64 %rd83, %rd6, %rd106;
|
| 601 |
+
rem.s64 %rd82, %rd7, %rd106;
|
| 602 |
+
rem.s64 %rd81, %rd8, %rd106;
|
| 603 |
+
.loc 1 32 62 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:32:62
|
| 604 |
+
bar.sync 0;
|
| 605 |
+
.loc 1 33 39 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:39
|
| 606 |
+
mul.lo.s64 %rd306, %rd73, %rd106;
|
| 607 |
+
mul.lo.s64 %rd307, %rd74, %rd106;
|
| 608 |
+
mul.lo.s64 %rd308, %rd75, %rd106;
|
| 609 |
+
mul.lo.s64 %rd309, %rd76, %rd106;
|
| 610 |
+
mul.lo.s64 %rd310, %rd77, %rd106;
|
| 611 |
+
mul.lo.s64 %rd311, %rd78, %rd106;
|
| 612 |
+
mul.lo.s64 %rd312, %rd79, %rd106;
|
| 613 |
+
mul.lo.s64 %rd313, %rd80, %rd106;
|
| 614 |
+
.loc 1 33 30 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:30
|
| 615 |
+
shl.b64 %rd314, %rd81, 1;
|
| 616 |
+
add.s64 %rd315, %rd100, %rd314;
|
| 617 |
+
shl.b64 %rd316, %rd306, 1;
|
| 618 |
+
add.s64 %rd235, %rd315, %rd316;
|
| 619 |
+
shl.b64 %rd317, %rd82, 1;
|
| 620 |
+
add.s64 %rd318, %rd100, %rd317;
|
| 621 |
+
shl.b64 %rd319, %rd307, 1;
|
| 622 |
+
add.s64 %rd238, %rd318, %rd319;
|
| 623 |
+
shl.b64 %rd320, %rd83, 1;
|
| 624 |
+
add.s64 %rd321, %rd100, %rd320;
|
| 625 |
+
shl.b64 %rd322, %rd308, 1;
|
| 626 |
+
add.s64 %rd241, %rd321, %rd322;
|
| 627 |
+
shl.b64 %rd323, %rd84, 1;
|
| 628 |
+
add.s64 %rd324, %rd100, %rd323;
|
| 629 |
+
shl.b64 %rd325, %rd309, 1;
|
| 630 |
+
add.s64 %rd244, %rd324, %rd325;
|
| 631 |
+
shl.b64 %rd326, %rd85, 1;
|
| 632 |
+
add.s64 %rd327, %rd100, %rd326;
|
| 633 |
+
shl.b64 %rd328, %rd310, 1;
|
| 634 |
+
add.s64 %rd247, %rd327, %rd328;
|
| 635 |
+
shl.b64 %rd329, %rd86, 1;
|
| 636 |
+
add.s64 %rd330, %rd100, %rd329;
|
| 637 |
+
shl.b64 %rd331, %rd311, 1;
|
| 638 |
+
add.s64 %rd250, %rd330, %rd331;
|
| 639 |
+
shl.b64 %rd332, %rd87, 1;
|
| 640 |
+
add.s64 %rd333, %rd100, %rd332;
|
| 641 |
+
shl.b64 %rd334, %rd312, 1;
|
| 642 |
+
add.s64 %rd253, %rd333, %rd334;
|
| 643 |
+
shl.b64 %rd335, %rd88, 1;
|
| 644 |
+
add.s64 %rd336, %rd100, %rd335;
|
| 645 |
+
shl.b64 %rd337, %rd313, 1;
|
| 646 |
+
add.s64 %rd256, %rd336, %rd337;
|
| 647 |
+
.loc 1 33 46 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:46
|
| 648 |
+
// begin inline asm
|
| 649 |
+
mov.u64 %rd234, 0x0;
|
| 650 |
+
createpolicy.fractional.L2::evict_last.b64 %rd234, 1.0;
|
| 651 |
+
// end inline asm
|
| 652 |
+
// begin inline asm
|
| 653 |
+
mov.u16 %rs67, 0x0;
|
| 654 |
+
@%p1 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs67 }, [ %rd235 + 0 ], %rd234;
|
| 655 |
+
// end inline asm
|
| 656 |
+
// begin inline asm
|
| 657 |
+
mov.u64 %rd237, 0x0;
|
| 658 |
+
createpolicy.fractional.L2::evict_last.b64 %rd237, 1.0;
|
| 659 |
+
// end inline asm
|
| 660 |
+
// begin inline asm
|
| 661 |
+
mov.u16 %rs68, 0x0;
|
| 662 |
+
@%p2 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs68 }, [ %rd238 + 0 ], %rd237;
|
| 663 |
+
// end inline asm
|
| 664 |
+
// begin inline asm
|
| 665 |
+
mov.u64 %rd240, 0x0;
|
| 666 |
+
createpolicy.fractional.L2::evict_last.b64 %rd240, 1.0;
|
| 667 |
+
// end inline asm
|
| 668 |
+
// begin inline asm
|
| 669 |
+
mov.u16 %rs69, 0x0;
|
| 670 |
+
@%p3 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs69 }, [ %rd241 + 0 ], %rd240;
|
| 671 |
+
// end inline asm
|
| 672 |
+
// begin inline asm
|
| 673 |
+
mov.u64 %rd243, 0x0;
|
| 674 |
+
createpolicy.fractional.L2::evict_last.b64 %rd243, 1.0;
|
| 675 |
+
// end inline asm
|
| 676 |
+
// begin inline asm
|
| 677 |
+
mov.u16 %rs70, 0x0;
|
| 678 |
+
@%p4 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs70 }, [ %rd244 + 0 ], %rd243;
|
| 679 |
+
// end inline asm
|
| 680 |
+
// begin inline asm
|
| 681 |
+
mov.u64 %rd246, 0x0;
|
| 682 |
+
createpolicy.fractional.L2::evict_last.b64 %rd246, 1.0;
|
| 683 |
+
// end inline asm
|
| 684 |
+
// begin inline asm
|
| 685 |
+
mov.u16 %rs71, 0x0;
|
| 686 |
+
@%p5 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs71 }, [ %rd247 + 0 ], %rd246;
|
| 687 |
+
// end inline asm
|
| 688 |
+
// begin inline asm
|
| 689 |
+
mov.u64 %rd249, 0x0;
|
| 690 |
+
createpolicy.fractional.L2::evict_last.b64 %rd249, 1.0;
|
| 691 |
+
// end inline asm
|
| 692 |
+
// begin inline asm
|
| 693 |
+
mov.u16 %rs72, 0x0;
|
| 694 |
+
@%p6 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs72 }, [ %rd250 + 0 ], %rd249;
|
| 695 |
+
// end inline asm
|
| 696 |
+
// begin inline asm
|
| 697 |
+
mov.u64 %rd252, 0x0;
|
| 698 |
+
createpolicy.fractional.L2::evict_last.b64 %rd252, 1.0;
|
| 699 |
+
// end inline asm
|
| 700 |
+
// begin inline asm
|
| 701 |
+
mov.u16 %rs73, 0x0;
|
| 702 |
+
@%p7 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs73 }, [ %rd253 + 0 ], %rd252;
|
| 703 |
+
// end inline asm
|
| 704 |
+
// begin inline asm
|
| 705 |
+
mov.u64 %rd255, 0x0;
|
| 706 |
+
createpolicy.fractional.L2::evict_last.b64 %rd255, 1.0;
|
| 707 |
+
// end inline asm
|
| 708 |
+
// begin inline asm
|
| 709 |
+
mov.u16 %rs74, 0x0;
|
| 710 |
+
@%p8 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs74 }, [ %rd256 + 0 ], %rd255;
|
| 711 |
+
// end inline asm
|
| 712 |
+
.loc 1 38 31 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:38:31
|
| 713 |
+
shr.u64 %rd338, %rd106, 63;
|
| 714 |
+
add.s64 %rd339, %rd106, %rd338;
|
| 715 |
+
shr.s64 %rd340, %rd339, 1;
|
| 716 |
+
.loc 1 38 18 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:38:18
|
| 717 |
+
sub.s64 %rd89, %rd106, %rd340;
|
| 718 |
+
.loc 1 39 19 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:39:19
|
| 719 |
+
setp.lt.s64 %p106, %rd81, %rd89;
|
| 720 |
+
setp.lt.s64 %p107, %rd82, %rd89;
|
| 721 |
+
setp.lt.s64 %p108, %rd83, %rd89;
|
| 722 |
+
setp.lt.s64 %p109, %rd84, %rd89;
|
| 723 |
+
setp.lt.s64 %p110, %rd85, %rd89;
|
| 724 |
+
setp.lt.s64 %p111, %rd86, %rd89;
|
| 725 |
+
setp.lt.s64 %p112, %rd87, %rd89;
|
| 726 |
+
setp.lt.s64 %p113, %rd88, %rd89;
|
| 727 |
+
.loc 1 40 35 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:35
|
| 728 |
+
sub.s64 %rd341, %rd8, %rd81;
|
| 729 |
+
sub.s64 %rd342, %rd7, %rd82;
|
| 730 |
+
sub.s64 %rd343, %rd6, %rd83;
|
| 731 |
+
sub.s64 %rd344, %rd5, %rd84;
|
| 732 |
+
sub.s64 %rd345, %rd4, %rd85;
|
| 733 |
+
sub.s64 %rd346, %rd3, %rd86;
|
| 734 |
+
sub.s64 %rd347, %rd2, %rd87;
|
| 735 |
+
sub.s64 %rd348, %rd1, %rd88;
|
| 736 |
+
.loc 1 40 31 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:31
|
| 737 |
+
shl.b64 %rd349, %rd341, 1;
|
| 738 |
+
add.s64 %rd350, %rd98, %rd349;
|
| 739 |
+
and.b64 %rd351, %rd339, -2;
|
| 740 |
+
add.s64 %rd352, %rd350, %rd351;
|
| 741 |
+
add.s64 %rd259, %rd352, %rd314;
|
| 742 |
+
shl.b64 %rd353, %rd342, 1;
|
| 743 |
+
add.s64 %rd354, %rd98, %rd353;
|
| 744 |
+
add.s64 %rd355, %rd354, %rd351;
|
| 745 |
+
add.s64 %rd262, %rd355, %rd317;
|
| 746 |
+
shl.b64 %rd356, %rd343, 1;
|
| 747 |
+
add.s64 %rd357, %rd98, %rd356;
|
| 748 |
+
add.s64 %rd358, %rd357, %rd351;
|
| 749 |
+
add.s64 %rd265, %rd358, %rd320;
|
| 750 |
+
shl.b64 %rd359, %rd344, 1;
|
| 751 |
+
add.s64 %rd360, %rd98, %rd359;
|
| 752 |
+
add.s64 %rd361, %rd360, %rd351;
|
| 753 |
+
add.s64 %rd268, %rd361, %rd323;
|
| 754 |
+
shl.b64 %rd362, %rd345, 1;
|
| 755 |
+
add.s64 %rd363, %rd98, %rd362;
|
| 756 |
+
add.s64 %rd364, %rd363, %rd351;
|
| 757 |
+
add.s64 %rd271, %rd364, %rd326;
|
| 758 |
+
shl.b64 %rd365, %rd346, 1;
|
| 759 |
+
add.s64 %rd366, %rd98, %rd365;
|
| 760 |
+
add.s64 %rd367, %rd366, %rd351;
|
| 761 |
+
add.s64 %rd274, %rd367, %rd329;
|
| 762 |
+
shl.b64 %rd368, %rd347, 1;
|
| 763 |
+
add.s64 %rd369, %rd98, %rd368;
|
| 764 |
+
add.s64 %rd370, %rd369, %rd351;
|
| 765 |
+
add.s64 %rd277, %rd370, %rd332;
|
| 766 |
+
shl.b64 %rd371, %rd348, 1;
|
| 767 |
+
add.s64 %rd372, %rd98, %rd371;
|
| 768 |
+
add.s64 %rd373, %rd372, %rd351;
|
| 769 |
+
add.s64 %rd280, %rd373, %rd335;
|
| 770 |
+
.loc 1 40 68 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:68
|
| 771 |
+
and.pred %p82, %p1, %p106;
|
| 772 |
+
and.pred %p83, %p2, %p107;
|
| 773 |
+
and.pred %p84, %p3, %p108;
|
| 774 |
+
and.pred %p85, %p4, %p109;
|
| 775 |
+
and.pred %p86, %p5, %p110;
|
| 776 |
+
and.pred %p87, %p6, %p111;
|
| 777 |
+
and.pred %p88, %p7, %p112;
|
| 778 |
+
and.pred %p89, %p8, %p113;
|
| 779 |
+
.loc 1 40 60 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:60
|
| 780 |
+
// begin inline asm
|
| 781 |
+
mov.u64 %rd258, 0x0;
|
| 782 |
+
createpolicy.fractional.L2::evict_last.b64 %rd258, 1.0;
|
| 783 |
+
// end inline asm
|
| 784 |
+
mov.b16 %rs76, 0;
|
| 785 |
+
// begin inline asm
|
| 786 |
+
mov.u16 %rs75, %rs76;
|
| 787 |
+
@%p82 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs75 }, [ %rd259 + 0 ], %rd258;
|
| 788 |
+
// end inline asm
|
| 789 |
+
// begin inline asm
|
| 790 |
+
mov.u64 %rd261, 0x0;
|
| 791 |
+
createpolicy.fractional.L2::evict_last.b64 %rd261, 1.0;
|
| 792 |
+
// end inline asm
|
| 793 |
+
// begin inline asm
|
| 794 |
+
mov.u16 %rs77, %rs76;
|
| 795 |
+
@%p83 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs77 }, [ %rd262 + 0 ], %rd261;
|
| 796 |
+
// end inline asm
|
| 797 |
+
// begin inline asm
|
| 798 |
+
mov.u64 %rd264, 0x0;
|
| 799 |
+
createpolicy.fractional.L2::evict_last.b64 %rd264, 1.0;
|
| 800 |
+
// end inline asm
|
| 801 |
+
// begin inline asm
|
| 802 |
+
mov.u16 %rs79, %rs76;
|
| 803 |
+
@%p84 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs79 }, [ %rd265 + 0 ], %rd264;
|
| 804 |
+
// end inline asm
|
| 805 |
+
// begin inline asm
|
| 806 |
+
mov.u64 %rd267, 0x0;
|
| 807 |
+
createpolicy.fractional.L2::evict_last.b64 %rd267, 1.0;
|
| 808 |
+
// end inline asm
|
| 809 |
+
// begin inline asm
|
| 810 |
+
mov.u16 %rs81, %rs76;
|
| 811 |
+
@%p85 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs81 }, [ %rd268 + 0 ], %rd267;
|
| 812 |
+
// end inline asm
|
| 813 |
+
// begin inline asm
|
| 814 |
+
mov.u64 %rd270, 0x0;
|
| 815 |
+
createpolicy.fractional.L2::evict_last.b64 %rd270, 1.0;
|
| 816 |
+
// end inline asm
|
| 817 |
+
// begin inline asm
|
| 818 |
+
mov.u16 %rs83, %rs76;
|
| 819 |
+
@%p86 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs83 }, [ %rd271 + 0 ], %rd270;
|
| 820 |
+
// end inline asm
|
| 821 |
+
// begin inline asm
|
| 822 |
+
mov.u64 %rd273, 0x0;
|
| 823 |
+
createpolicy.fractional.L2::evict_last.b64 %rd273, 1.0;
|
| 824 |
+
// end inline asm
|
| 825 |
+
// begin inline asm
|
| 826 |
+
mov.u16 %rs85, %rs76;
|
| 827 |
+
@%p87 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs85 }, [ %rd274 + 0 ], %rd273;
|
| 828 |
+
// end inline asm
|
| 829 |
+
// begin inline asm
|
| 830 |
+
mov.u64 %rd276, 0x0;
|
| 831 |
+
createpolicy.fractional.L2::evict_last.b64 %rd276, 1.0;
|
| 832 |
+
// end inline asm
|
| 833 |
+
// begin inline asm
|
| 834 |
+
mov.u16 %rs87, %rs76;
|
| 835 |
+
@%p88 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs87 }, [ %rd277 + 0 ], %rd276;
|
| 836 |
+
// end inline asm
|
| 837 |
+
// begin inline asm
|
| 838 |
+
mov.u64 %rd279, 0x0;
|
| 839 |
+
createpolicy.fractional.L2::evict_last.b64 %rd279, 1.0;
|
| 840 |
+
// end inline asm
|
| 841 |
+
// begin inline asm
|
| 842 |
+
mov.u16 %rs89, %rs76;
|
| 843 |
+
@%p89 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs89 }, [ %rd280 + 0 ], %rd279;
|
| 844 |
+
// end inline asm
|
| 845 |
+
.loc 1 44 20 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:44:20
|
| 846 |
+
setp.ge.s64 %p114, %rd88, %rd89;
|
| 847 |
+
setp.ge.s64 %p115, %rd87, %rd89;
|
| 848 |
+
setp.ge.s64 %p116, %rd86, %rd89;
|
| 849 |
+
setp.ge.s64 %p117, %rd85, %rd89;
|
| 850 |
+
setp.ge.s64 %p118, %rd84, %rd89;
|
| 851 |
+
setp.ge.s64 %p119, %rd83, %rd89;
|
| 852 |
+
setp.ge.s64 %p120, %rd82, %rd89;
|
| 853 |
+
setp.ge.s64 %p121, %rd81, %rd89;
|
| 854 |
+
.loc 1 47 47 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:47
|
| 855 |
+
sub.s64 %rd374, %rd81, %rd106;
|
| 856 |
+
sub.s64 %rd375, %rd82, %rd106;
|
| 857 |
+
sub.s64 %rd376, %rd83, %rd106;
|
| 858 |
+
sub.s64 %rd377, %rd84, %rd106;
|
| 859 |
+
sub.s64 %rd378, %rd85, %rd106;
|
| 860 |
+
sub.s64 %rd379, %rd86, %rd106;
|
| 861 |
+
sub.s64 %rd380, %rd87, %rd106;
|
| 862 |
+
sub.s64 %rd381, %rd88, %rd106;
|
| 863 |
+
.loc 1 47 31 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:31
|
| 864 |
+
shl.b64 %rd382, %rd374, 1;
|
| 865 |
+
add.s64 %rd283, %rd352, %rd382;
|
| 866 |
+
shl.b64 %rd383, %rd375, 1;
|
| 867 |
+
add.s64 %rd286, %rd355, %rd383;
|
| 868 |
+
shl.b64 %rd384, %rd376, 1;
|
| 869 |
+
add.s64 %rd289, %rd358, %rd384;
|
| 870 |
+
shl.b64 %rd385, %rd377, 1;
|
| 871 |
+
add.s64 %rd292, %rd361, %rd385;
|
| 872 |
+
shl.b64 %rd386, %rd378, 1;
|
| 873 |
+
add.s64 %rd295, %rd364, %rd386;
|
| 874 |
+
shl.b64 %rd387, %rd379, 1;
|
| 875 |
+
add.s64 %rd298, %rd367, %rd387;
|
| 876 |
+
shl.b64 %rd388, %rd380, 1;
|
| 877 |
+
add.s64 %rd301, %rd370, %rd388;
|
| 878 |
+
shl.b64 %rd389, %rd381, 1;
|
| 879 |
+
add.s64 %rd304, %rd373, %rd389;
|
| 880 |
+
.loc 1 47 81 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:81
|
| 881 |
+
and.pred %p90, %p1, %p121;
|
| 882 |
+
and.pred %p91, %p2, %p120;
|
| 883 |
+
and.pred %p92, %p3, %p119;
|
| 884 |
+
and.pred %p93, %p4, %p118;
|
| 885 |
+
and.pred %p94, %p5, %p117;
|
| 886 |
+
and.pred %p95, %p6, %p116;
|
| 887 |
+
and.pred %p96, %p7, %p115;
|
| 888 |
+
and.pred %p97, %p8, %p114;
|
| 889 |
+
.loc 1 47 73 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:73
|
| 890 |
+
// begin inline asm
|
| 891 |
+
mov.u64 %rd282, 0x0;
|
| 892 |
+
createpolicy.fractional.L2::evict_last.b64 %rd282, 1.0;
|
| 893 |
+
// end inline asm
|
| 894 |
+
// begin inline asm
|
| 895 |
+
mov.u16 %rs91, %rs76;
|
| 896 |
+
@%p90 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs91 }, [ %rd283 + 0 ], %rd282;
|
| 897 |
+
// end inline asm
|
| 898 |
+
// begin inline asm
|
| 899 |
+
mov.u64 %rd285, 0x0;
|
| 900 |
+
createpolicy.fractional.L2::evict_last.b64 %rd285, 1.0;
|
| 901 |
+
// end inline asm
|
| 902 |
+
// begin inline asm
|
| 903 |
+
mov.u16 %rs93, %rs76;
|
| 904 |
+
@%p91 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs93 }, [ %rd286 + 0 ], %rd285;
|
| 905 |
+
// end inline asm
|
| 906 |
+
// begin inline asm
|
| 907 |
+
mov.u64 %rd288, 0x0;
|
| 908 |
+
createpolicy.fractional.L2::evict_last.b64 %rd288, 1.0;
|
| 909 |
+
// end inline asm
|
| 910 |
+
// begin inline asm
|
| 911 |
+
mov.u16 %rs95, %rs76;
|
| 912 |
+
@%p92 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs95 }, [ %rd289 + 0 ], %rd288;
|
| 913 |
+
// end inline asm
|
| 914 |
+
// begin inline asm
|
| 915 |
+
mov.u64 %rd291, 0x0;
|
| 916 |
+
createpolicy.fractional.L2::evict_last.b64 %rd291, 1.0;
|
| 917 |
+
// end inline asm
|
| 918 |
+
// begin inline asm
|
| 919 |
+
mov.u16 %rs97, %rs76;
|
| 920 |
+
@%p93 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs97 }, [ %rd292 + 0 ], %rd291;
|
| 921 |
+
// end inline asm
|
| 922 |
+
// begin inline asm
|
| 923 |
+
mov.u64 %rd294, 0x0;
|
| 924 |
+
createpolicy.fractional.L2::evict_last.b64 %rd294, 1.0;
|
| 925 |
+
// end inline asm
|
| 926 |
+
// begin inline asm
|
| 927 |
+
mov.u16 %rs99, %rs76;
|
| 928 |
+
@%p94 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs99 }, [ %rd295 + 0 ], %rd294;
|
| 929 |
+
// end inline asm
|
| 930 |
+
// begin inline asm
|
| 931 |
+
mov.u64 %rd297, 0x0;
|
| 932 |
+
createpolicy.fractional.L2::evict_last.b64 %rd297, 1.0;
|
| 933 |
+
// end inline asm
|
| 934 |
+
// begin inline asm
|
| 935 |
+
mov.u16 %rs101, %rs76;
|
| 936 |
+
@%p95 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs101 }, [ %rd298 + 0 ], %rd297;
|
| 937 |
+
// end inline asm
|
| 938 |
+
// begin inline asm
|
| 939 |
+
mov.u64 %rd300, 0x0;
|
| 940 |
+
createpolicy.fractional.L2::evict_last.b64 %rd300, 1.0;
|
| 941 |
+
// end inline asm
|
| 942 |
+
// begin inline asm
|
| 943 |
+
mov.u16 %rs103, %rs76;
|
| 944 |
+
@%p96 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs103 }, [ %rd301 + 0 ], %rd300;
|
| 945 |
+
// end inline asm
|
| 946 |
+
// begin inline asm
|
| 947 |
+
mov.u64 %rd303, 0x0;
|
| 948 |
+
createpolicy.fractional.L2::evict_last.b64 %rd303, 1.0;
|
| 949 |
+
// end inline asm
|
| 950 |
+
// begin inline asm
|
| 951 |
+
mov.u16 %rs105, %rs76;
|
| 952 |
+
@%p97 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs105 }, [ %rd304 + 0 ], %rd303;
|
| 953 |
+
// end inline asm
|
| 954 |
+
.loc 1 51 34 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:51:34
|
| 955 |
+
and.b64 %rd391, %rd212, %rd107;
|
| 956 |
+
and.b64 %rd393, %rd214, %rd107;
|
| 957 |
+
and.b64 %rd395, %rd216, %rd107;
|
| 958 |
+
and.b64 %rd397, %rd218, %rd107;
|
| 959 |
+
and.b64 %rd399, %rd220, %rd107;
|
| 960 |
+
and.b64 %rd401, %rd222, %rd107;
|
| 961 |
+
and.b64 %rd403, %rd224, %rd107;
|
| 962 |
+
and.b64 %rd405, %rd226, %rd107;
|
| 963 |
+
add.s64 %rd95, %rd405, %rd185;
|
| 964 |
+
add.s64 %rd94, %rd403, %rd181;
|
| 965 |
+
add.s64 %rd97, %rd401, %rd193;
|
| 966 |
+
add.s64 %rd96, %rd399, %rd189;
|
| 967 |
+
add.s64 %rd91, %rd397, %rd169;
|
| 968 |
+
add.s64 %rd90, %rd395, %rd165;
|
| 969 |
+
add.s64 %rd93, %rd393, %rd177;
|
| 970 |
+
add.s64 %rd92, %rd391, %rd173;
|
| 971 |
+
.loc 1 52 28 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:52:28
|
| 972 |
+
setp.lt.s64 %p122, %rd92, 0;
|
| 973 |
+
setp.lt.s64 %p123, %rd93, 0;
|
| 974 |
+
setp.lt.s64 %p124, %rd90, 0;
|
| 975 |
+
setp.lt.s64 %p125, %rd91, 0;
|
| 976 |
+
setp.lt.s64 %p126, %rd96, 0;
|
| 977 |
+
setp.lt.s64 %p127, %rd97, 0;
|
| 978 |
+
setp.lt.s64 %p128, %rd94, 0;
|
| 979 |
+
setp.lt.s64 %p129, %rd95, 0;
|
| 980 |
+
.loc 1 52 46 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:52:46
|
| 981 |
+
setp.ge.s64 %p130, %rd92, %rd107;
|
| 982 |
+
setp.ge.s64 %p131, %rd93, %rd107;
|
| 983 |
+
setp.ge.s64 %p132, %rd90, %rd107;
|
| 984 |
+
setp.ge.s64 %p133, %rd91, %rd107;
|
| 985 |
+
setp.ge.s64 %p134, %rd96, %rd107;
|
| 986 |
+
setp.ge.s64 %p135, %rd97, %rd107;
|
| 987 |
+
setp.ge.s64 %p136, %rd94, %rd107;
|
| 988 |
+
setp.ge.s64 %p137, %rd95, %rd107;
|
| 989 |
+
.loc 1 52 38 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:52:38
|
| 990 |
+
or.pred %p138, %p129, %p137;
|
| 991 |
+
or.pred %p139, %p128, %p136;
|
| 992 |
+
or.pred %p140, %p127, %p135;
|
| 993 |
+
or.pred %p141, %p126, %p134;
|
| 994 |
+
or.pred %p142, %p125, %p133;
|
| 995 |
+
or.pred %p143, %p124, %p132;
|
| 996 |
+
or.pred %p144, %p123, %p131;
|
| 997 |
+
or.pred %p145, %p122, %p130;
|
| 998 |
+
.loc 1 52 54 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:52:54
|
| 999 |
+
and.pred %p146, %p3, %p145;
|
| 1000 |
+
selp.b16 %rs107, 1, 0, %p146;
|
| 1001 |
+
shl.b16 %rs108, %rs107, 2;
|
| 1002 |
+
and.pred %p147, %p4, %p144;
|
| 1003 |
+
selp.b16 %rs109, -1, 0, %p147;
|
| 1004 |
+
shl.b16 %rs110, %rs109, 3;
|
| 1005 |
+
or.b16 %rs111, %rs110, %rs108;
|
| 1006 |
+
and.pred %p148, %p1, %p143;
|
| 1007 |
+
selp.b16 %rs112, 1, 0, %p148;
|
| 1008 |
+
and.pred %p149, %p2, %p142;
|
| 1009 |
+
selp.b16 %rs113, -1, 0, %p149;
|
| 1010 |
+
shl.b16 %rs114, %rs113, 1;
|
| 1011 |
+
or.b16 %rs115, %rs112, %rs114;
|
| 1012 |
+
and.b16 %rs116, %rs115, 3;
|
| 1013 |
+
or.b16 %rs117, %rs116, %rs111;
|
| 1014 |
+
and.b16 %rs118, %rs117, 15;
|
| 1015 |
+
and.pred %p150, %p7, %p141;
|
| 1016 |
+
selp.b16 %rs119, 1, 0, %p150;
|
| 1017 |
+
shl.b16 %rs120, %rs119, 2;
|
| 1018 |
+
and.pred %p151, %p8, %p140;
|
| 1019 |
+
selp.b16 %rs121, -1, 0, %p151;
|
| 1020 |
+
shl.b16 %rs122, %rs121, 3;
|
| 1021 |
+
or.b16 %rs123, %rs122, %rs120;
|
| 1022 |
+
and.pred %p152, %p5, %p139;
|
| 1023 |
+
selp.b16 %rs124, 1, 0, %p152;
|
| 1024 |
+
and.pred %p153, %p6, %p138;
|
| 1025 |
+
selp.b16 %rs125, -1, 0, %p153;
|
| 1026 |
+
shl.b16 %rs126, %rs125, 1;
|
| 1027 |
+
or.b16 %rs127, %rs124, %rs126;
|
| 1028 |
+
and.b16 %rs128, %rs127, 3;
|
| 1029 |
+
or.b16 %rs129, %rs128, %rs123;
|
| 1030 |
+
shl.b16 %rs130, %rs129, 4;
|
| 1031 |
+
or.b16 %rs131, %rs118, %rs130;
|
| 1032 |
+
.loc 1 52 64 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:52:64
|
| 1033 |
+
and.b16 %rs132, %rs131, 255;
|
| 1034 |
+
setp.eq.b16 %p154, %rs132, 0;
|
| 1035 |
+
@%p154 bra $L__BB0_52;
|
| 1036 |
+
bra.uni $L__BB0_51;
|
| 1037 |
+
$L__BB0_52:
|
| 1038 |
+
.loc 1 0 64 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0:64
|
| 1039 |
+
ld.param.b64 %rd102, [triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_4];
|
| 1040 |
+
ld.param.b64 %rd101, [triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0_param_3];
|
| 1041 |
+
.loc 1 40 119 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:119
|
| 1042 |
+
cvt.f32.bf16 %r79, %rs89;
|
| 1043 |
+
mov.b32 %r80, 0f00000000;
|
| 1044 |
+
.loc 1 41 13 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:41:13
|
| 1045 |
+
sub.f32 %r81, %r80, %r79;
|
| 1046 |
+
.loc 1 47 132 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:132
|
| 1047 |
+
cvt.f32.bf16 %r82, %rs105;
|
| 1048 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 1049 |
+
selp.f32 %r83, %r81, %r82, %p113;
|
| 1050 |
+
.loc 1 40 119 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:119
|
| 1051 |
+
cvt.f32.bf16 %r84, %rs87;
|
| 1052 |
+
.loc 1 41 13 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:41:13
|
| 1053 |
+
sub.f32 %r85, %r80, %r84;
|
| 1054 |
+
.loc 1 47 132 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:132
|
| 1055 |
+
cvt.f32.bf16 %r86, %rs103;
|
| 1056 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 1057 |
+
selp.f32 %r87, %r85, %r86, %p112;
|
| 1058 |
+
.loc 1 40 119 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:119
|
| 1059 |
+
cvt.f32.bf16 %r88, %rs85;
|
| 1060 |
+
.loc 1 41 13 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:41:13
|
| 1061 |
+
sub.f32 %r89, %r80, %r88;
|
| 1062 |
+
.loc 1 47 132 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:132
|
| 1063 |
+
cvt.f32.bf16 %r90, %rs101;
|
| 1064 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 1065 |
+
selp.f32 %r91, %r89, %r90, %p111;
|
| 1066 |
+
.loc 1 40 119 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:119
|
| 1067 |
+
cvt.f32.bf16 %r92, %rs83;
|
| 1068 |
+
.loc 1 41 13 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:41:13
|
| 1069 |
+
sub.f32 %r93, %r80, %r92;
|
| 1070 |
+
.loc 1 47 132 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:132
|
| 1071 |
+
cvt.f32.bf16 %r94, %rs99;
|
| 1072 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 1073 |
+
selp.f32 %r95, %r93, %r94, %p110;
|
| 1074 |
+
.loc 1 40 119 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:119
|
| 1075 |
+
cvt.f32.bf16 %r96, %rs81;
|
| 1076 |
+
.loc 1 41 13 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:41:13
|
| 1077 |
+
sub.f32 %r97, %r80, %r96;
|
| 1078 |
+
.loc 1 47 132 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:132
|
| 1079 |
+
cvt.f32.bf16 %r98, %rs97;
|
| 1080 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 1081 |
+
selp.f32 %r99, %r97, %r98, %p109;
|
| 1082 |
+
.loc 1 40 119 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:119
|
| 1083 |
+
cvt.f32.bf16 %r100, %rs79;
|
| 1084 |
+
.loc 1 41 13 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:41:13
|
| 1085 |
+
sub.f32 %r101, %r80, %r100;
|
| 1086 |
+
.loc 1 47 132 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:132
|
| 1087 |
+
cvt.f32.bf16 %r102, %rs95;
|
| 1088 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 1089 |
+
selp.f32 %r103, %r101, %r102, %p108;
|
| 1090 |
+
.loc 1 40 119 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:119
|
| 1091 |
+
cvt.f32.bf16 %r104, %rs77;
|
| 1092 |
+
.loc 1 41 13 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:41:13
|
| 1093 |
+
sub.f32 %r105, %r80, %r104;
|
| 1094 |
+
.loc 1 47 132 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:132
|
| 1095 |
+
cvt.f32.bf16 %r106, %rs93;
|
| 1096 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 1097 |
+
selp.f32 %r107, %r105, %r106, %p107;
|
| 1098 |
+
.loc 1 40 119 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:40:119
|
| 1099 |
+
cvt.f32.bf16 %r108, %rs75;
|
| 1100 |
+
.loc 1 41 13 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:41:13
|
| 1101 |
+
sub.f32 %r109, %r80, %r108;
|
| 1102 |
+
.loc 1 47 132 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:47:132
|
| 1103 |
+
cvt.f32.bf16 %r110, %rs91;
|
| 1104 |
+
.loc 1 0 0 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:0
|
| 1105 |
+
selp.f32 %r111, %r109, %r110, %p106;
|
| 1106 |
+
.loc 1 26 75 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:26:75
|
| 1107 |
+
cvt.f32.bf16 %r112, %rs40;
|
| 1108 |
+
cvt.f32.bf16 %r113, %rs39;
|
| 1109 |
+
cvt.f32.bf16 %r114, %rs38;
|
| 1110 |
+
cvt.f32.bf16 %r115, %rs37;
|
| 1111 |
+
cvt.f32.bf16 %r116, %rs36;
|
| 1112 |
+
cvt.f32.bf16 %r117, %rs35;
|
| 1113 |
+
cvt.f32.bf16 %r118, %rs34;
|
| 1114 |
+
cvt.f32.bf16 %r119, %rs33;
|
| 1115 |
+
.loc 1 52 64 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:52:64
|
| 1116 |
+
bar.sync 0;
|
| 1117 |
+
.loc 1 53 40 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:53:40
|
| 1118 |
+
mul.lo.s64 %rd444, %rd90, %rd106;
|
| 1119 |
+
mul.lo.s64 %rd445, %rd91, %rd106;
|
| 1120 |
+
mul.lo.s64 %rd446, %rd92, %rd106;
|
| 1121 |
+
mul.lo.s64 %rd447, %rd93, %rd106;
|
| 1122 |
+
mul.lo.s64 %rd448, %rd94, %rd106;
|
| 1123 |
+
mul.lo.s64 %rd449, %rd95, %rd106;
|
| 1124 |
+
mul.lo.s64 %rd450, %rd96, %rd106;
|
| 1125 |
+
mul.lo.s64 %rd451, %rd97, %rd106;
|
| 1126 |
+
.loc 1 53 31 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:53:31
|
| 1127 |
+
add.s64 %rd453, %rd101, %rd314;
|
| 1128 |
+
shl.b64 %rd454, %rd444, 1;
|
| 1129 |
+
add.s64 %rd413, %rd453, %rd454;
|
| 1130 |
+
add.s64 %rd456, %rd101, %rd317;
|
| 1131 |
+
shl.b64 %rd457, %rd445, 1;
|
| 1132 |
+
add.s64 %rd416, %rd456, %rd457;
|
| 1133 |
+
add.s64 %rd459, %rd101, %rd320;
|
| 1134 |
+
shl.b64 %rd460, %rd446, 1;
|
| 1135 |
+
add.s64 %rd419, %rd459, %rd460;
|
| 1136 |
+
add.s64 %rd462, %rd101, %rd323;
|
| 1137 |
+
shl.b64 %rd463, %rd447, 1;
|
| 1138 |
+
add.s64 %rd422, %rd462, %rd463;
|
| 1139 |
+
add.s64 %rd465, %rd101, %rd326;
|
| 1140 |
+
shl.b64 %rd466, %rd448, 1;
|
| 1141 |
+
add.s64 %rd425, %rd465, %rd466;
|
| 1142 |
+
add.s64 %rd468, %rd101, %rd329;
|
| 1143 |
+
shl.b64 %rd469, %rd449, 1;
|
| 1144 |
+
add.s64 %rd428, %rd468, %rd469;
|
| 1145 |
+
add.s64 %rd471, %rd101, %rd332;
|
| 1146 |
+
shl.b64 %rd472, %rd450, 1;
|
| 1147 |
+
add.s64 %rd431, %rd471, %rd472;
|
| 1148 |
+
add.s64 %rd474, %rd101, %rd335;
|
| 1149 |
+
shl.b64 %rd475, %rd451, 1;
|
| 1150 |
+
add.s64 %rd434, %rd474, %rd475;
|
| 1151 |
+
.loc 1 53 48 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:53:48
|
| 1152 |
+
// begin inline asm
|
| 1153 |
+
mov.u64 %rd414, 0x0;
|
| 1154 |
+
createpolicy.fractional.L2::evict_last.b64 %rd414, 1.0;
|
| 1155 |
+
// end inline asm
|
| 1156 |
+
// begin inline asm
|
| 1157 |
+
mov.u16 %rs133, 0x0;
|
| 1158 |
+
@%p1 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs133 }, [ %rd413 + 0 ], %rd414;
|
| 1159 |
+
// end inline asm
|
| 1160 |
+
// begin inline asm
|
| 1161 |
+
mov.u64 %rd417, 0x0;
|
| 1162 |
+
createpolicy.fractional.L2::evict_last.b64 %rd417, 1.0;
|
| 1163 |
+
// end inline asm
|
| 1164 |
+
// begin inline asm
|
| 1165 |
+
mov.u16 %rs134, 0x0;
|
| 1166 |
+
@%p2 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs134 }, [ %rd416 + 0 ], %rd417;
|
| 1167 |
+
// end inline asm
|
| 1168 |
+
// begin inline asm
|
| 1169 |
+
mov.u64 %rd420, 0x0;
|
| 1170 |
+
createpolicy.fractional.L2::evict_last.b64 %rd420, 1.0;
|
| 1171 |
+
// end inline asm
|
| 1172 |
+
// begin inline asm
|
| 1173 |
+
mov.u16 %rs135, 0x0;
|
| 1174 |
+
@%p3 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs135 }, [ %rd419 + 0 ], %rd420;
|
| 1175 |
+
// end inline asm
|
| 1176 |
+
// begin inline asm
|
| 1177 |
+
mov.u64 %rd423, 0x0;
|
| 1178 |
+
createpolicy.fractional.L2::evict_last.b64 %rd423, 1.0;
|
| 1179 |
+
// end inline asm
|
| 1180 |
+
// begin inline asm
|
| 1181 |
+
mov.u16 %rs136, 0x0;
|
| 1182 |
+
@%p4 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs136 }, [ %rd422 + 0 ], %rd423;
|
| 1183 |
+
// end inline asm
|
| 1184 |
+
// begin inline asm
|
| 1185 |
+
mov.u64 %rd426, 0x0;
|
| 1186 |
+
createpolicy.fractional.L2::evict_last.b64 %rd426, 1.0;
|
| 1187 |
+
// end inline asm
|
| 1188 |
+
// begin inline asm
|
| 1189 |
+
mov.u16 %rs137, 0x0;
|
| 1190 |
+
@%p5 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs137 }, [ %rd425 + 0 ], %rd426;
|
| 1191 |
+
// end inline asm
|
| 1192 |
+
// begin inline asm
|
| 1193 |
+
mov.u64 %rd429, 0x0;
|
| 1194 |
+
createpolicy.fractional.L2::evict_last.b64 %rd429, 1.0;
|
| 1195 |
+
// end inline asm
|
| 1196 |
+
// begin inline asm
|
| 1197 |
+
mov.u16 %rs138, 0x0;
|
| 1198 |
+
@%p6 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs138 }, [ %rd428 + 0 ], %rd429;
|
| 1199 |
+
// end inline asm
|
| 1200 |
+
// begin inline asm
|
| 1201 |
+
mov.u64 %rd432, 0x0;
|
| 1202 |
+
createpolicy.fractional.L2::evict_last.b64 %rd432, 1.0;
|
| 1203 |
+
// end inline asm
|
| 1204 |
+
// begin inline asm
|
| 1205 |
+
mov.u16 %rs139, 0x0;
|
| 1206 |
+
@%p7 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs139 }, [ %rd431 + 0 ], %rd432;
|
| 1207 |
+
// end inline asm
|
| 1208 |
+
// begin inline asm
|
| 1209 |
+
mov.u64 %rd435, 0x0;
|
| 1210 |
+
createpolicy.fractional.L2::evict_last.b64 %rd435, 1.0;
|
| 1211 |
+
// end inline asm
|
| 1212 |
+
// begin inline asm
|
| 1213 |
+
mov.u16 %rs140, 0x0;
|
| 1214 |
+
@%p8 ld.global.L1::evict_last.L2::cache_hint.b16 { %rs140 }, [ %rd434 + 0 ], %rd435;
|
| 1215 |
+
// end inline asm
|
| 1216 |
+
.loc 1 33 46 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:46
|
| 1217 |
+
mov.b32 %r120, {%rs67, %rs133};
|
| 1218 |
+
.loc 1 33 86 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:86
|
| 1219 |
+
mov.b32 {%rs149, %rs150}, %r120;
|
| 1220 |
+
cvt.f32.bf16 %r121, %rs149;
|
| 1221 |
+
cvt.f32.bf16 %r122, %rs150;
|
| 1222 |
+
.loc 1 34 18 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:34:18
|
| 1223 |
+
mul.f32 %r123, %r111, %r122;
|
| 1224 |
+
.loc 1 33 46 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:46
|
| 1225 |
+
mov.b32 %r124, {%rs68, %rs134};
|
| 1226 |
+
.loc 1 33 86 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:86
|
| 1227 |
+
mov.b32 {%rs151, %rs152}, %r124;
|
| 1228 |
+
cvt.f32.bf16 %r125, %rs151;
|
| 1229 |
+
cvt.f32.bf16 %r126, %rs152;
|
| 1230 |
+
.loc 1 34 18 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:34:18
|
| 1231 |
+
mul.f32 %r127, %r107, %r126;
|
| 1232 |
+
.loc 1 33 46 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:46
|
| 1233 |
+
mov.b32 %r128, {%rs69, %rs135};
|
| 1234 |
+
.loc 1 33 86 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:86
|
| 1235 |
+
mov.b32 {%rs153, %rs154}, %r128;
|
| 1236 |
+
cvt.f32.bf16 %r129, %rs153;
|
| 1237 |
+
cvt.f32.bf16 %r130, %rs154;
|
| 1238 |
+
.loc 1 34 18 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:34:18
|
| 1239 |
+
mul.f32 %r131, %r103, %r130;
|
| 1240 |
+
.loc 1 33 46 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:46
|
| 1241 |
+
mov.b32 %r132, {%rs70, %rs136};
|
| 1242 |
+
.loc 1 33 86 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:86
|
| 1243 |
+
mov.b32 {%rs155, %rs156}, %r132;
|
| 1244 |
+
cvt.f32.bf16 %r133, %rs155;
|
| 1245 |
+
cvt.f32.bf16 %r134, %rs156;
|
| 1246 |
+
.loc 1 34 18 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:34:18
|
| 1247 |
+
mul.f32 %r135, %r99, %r134;
|
| 1248 |
+
.loc 1 33 46 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:46
|
| 1249 |
+
mov.b32 %r136, {%rs71, %rs137};
|
| 1250 |
+
.loc 1 33 86 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:86
|
| 1251 |
+
mov.b32 {%rs157, %rs158}, %r136;
|
| 1252 |
+
cvt.f32.bf16 %r137, %rs157;
|
| 1253 |
+
cvt.f32.bf16 %r138, %rs158;
|
| 1254 |
+
.loc 1 34 18 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:34:18
|
| 1255 |
+
mul.f32 %r139, %r95, %r138;
|
| 1256 |
+
.loc 1 33 46 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:46
|
| 1257 |
+
mov.b32 %r140, {%rs72, %rs138};
|
| 1258 |
+
.loc 1 33 86 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:86
|
| 1259 |
+
mov.b32 {%rs159, %rs160}, %r140;
|
| 1260 |
+
cvt.f32.bf16 %r141, %rs159;
|
| 1261 |
+
cvt.f32.bf16 %r142, %rs160;
|
| 1262 |
+
.loc 1 34 18 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:34:18
|
| 1263 |
+
mul.f32 %r143, %r91, %r142;
|
| 1264 |
+
.loc 1 33 46 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:46
|
| 1265 |
+
mov.b32 %r144, {%rs73, %rs139};
|
| 1266 |
+
.loc 1 33 86 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:86
|
| 1267 |
+
mov.b32 {%rs161, %rs162}, %r144;
|
| 1268 |
+
cvt.f32.bf16 %r145, %rs161;
|
| 1269 |
+
cvt.f32.bf16 %r146, %rs162;
|
| 1270 |
+
.loc 1 34 18 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:34:18
|
| 1271 |
+
mul.f32 %r147, %r87, %r146;
|
| 1272 |
+
.loc 1 33 46 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:46
|
| 1273 |
+
mov.b32 %r148, {%rs74, %rs140};
|
| 1274 |
+
.loc 1 33 86 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:33:86
|
| 1275 |
+
mov.b32 {%rs163, %rs164}, %r148;
|
| 1276 |
+
cvt.f32.bf16 %r149, %rs163;
|
| 1277 |
+
cvt.f32.bf16 %r150, %rs164;
|
| 1278 |
+
.loc 1 34 18 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:34:18
|
| 1279 |
+
mul.f32 %r151, %r83, %r150;
|
| 1280 |
+
.loc 1 55 19 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:55:19
|
| 1281 |
+
fma.rn.f32 %r152, %r119, %r121, %r123;
|
| 1282 |
+
fma.rn.f32 %r153, %r118, %r125, %r127;
|
| 1283 |
+
fma.rn.f32 %r154, %r117, %r129, %r131;
|
| 1284 |
+
fma.rn.f32 %r155, %r116, %r133, %r135;
|
| 1285 |
+
fma.rn.f32 %r156, %r115, %r137, %r139;
|
| 1286 |
+
fma.rn.f32 %r157, %r114, %r141, %r143;
|
| 1287 |
+
fma.rn.f32 %r158, %r113, %r145, %r147;
|
| 1288 |
+
fma.rn.f32 %r159, %r112, %r149, %r151;
|
| 1289 |
+
.loc 1 56 25 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:56:25
|
| 1290 |
+
add.s64 %rd436, %rd102, %rd196;
|
| 1291 |
+
add.s64 %rd437, %rd102, %rd197;
|
| 1292 |
+
add.s64 %rd438, %rd102, %rd198;
|
| 1293 |
+
add.s64 %rd439, %rd102, %rd199;
|
| 1294 |
+
add.s64 %rd440, %rd102, %rd200;
|
| 1295 |
+
add.s64 %rd441, %rd102, %rd201;
|
| 1296 |
+
add.s64 %rd442, %rd102, %rd202;
|
| 1297 |
+
add.s64 %rd443, %rd102, %rd203;
|
| 1298 |
+
.loc 1 56 37 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:56:37
|
| 1299 |
+
cvt.rn.bf16.f32 %rs141, %r152;
|
| 1300 |
+
cvt.rn.bf16.f32 %rs142, %r153;
|
| 1301 |
+
cvt.rn.bf16.f32 %rs143, %r154;
|
| 1302 |
+
cvt.rn.bf16.f32 %rs144, %r155;
|
| 1303 |
+
cvt.rn.bf16.f32 %rs145, %r156;
|
| 1304 |
+
cvt.rn.bf16.f32 %rs146, %r157;
|
| 1305 |
+
cvt.rn.bf16.f32 %rs147, %r158;
|
| 1306 |
+
cvt.rn.bf16.f32 %rs148, %r159;
|
| 1307 |
+
// begin inline asm
|
| 1308 |
+
@%p1 st.global.b16 [ %rd436 + 0 ], { %rs141 };
|
| 1309 |
+
// end inline asm
|
| 1310 |
+
// begin inline asm
|
| 1311 |
+
@%p2 st.global.b16 [ %rd437 + 0 ], { %rs142 };
|
| 1312 |
+
// end inline asm
|
| 1313 |
+
// begin inline asm
|
| 1314 |
+
@%p3 st.global.b16 [ %rd438 + 0 ], { %rs143 };
|
| 1315 |
+
// end inline asm
|
| 1316 |
+
// begin inline asm
|
| 1317 |
+
@%p4 st.global.b16 [ %rd439 + 0 ], { %rs144 };
|
| 1318 |
+
// end inline asm
|
| 1319 |
+
// begin inline asm
|
| 1320 |
+
@%p5 st.global.b16 [ %rd440 + 0 ], { %rs145 };
|
| 1321 |
+
// end inline asm
|
| 1322 |
+
// begin inline asm
|
| 1323 |
+
@%p6 st.global.b16 [ %rd441 + 0 ], { %rs146 };
|
| 1324 |
+
// end inline asm
|
| 1325 |
+
// begin inline asm
|
| 1326 |
+
@%p7 st.global.b16 [ %rd442 + 0 ], { %rs147 };
|
| 1327 |
+
// end inline asm
|
| 1328 |
+
// begin inline asm
|
| 1329 |
+
@%p8 st.global.b16 [ %rd443 + 0 ], { %rs148 };
|
| 1330 |
+
// end inline asm
|
| 1331 |
+
.loc 1 56 4 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:56:4
|
| 1332 |
+
ret;
|
| 1333 |
+
$L__BB0_49:
|
| 1334 |
+
.loc 1 32 62 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:32:62
|
| 1335 |
+
{ // callseq 0, 0
|
| 1336 |
+
.param .b64 param0;
|
| 1337 |
+
.param .b64 param1;
|
| 1338 |
+
.param .b32 param2;
|
| 1339 |
+
.param .b64 param3;
|
| 1340 |
+
.param .b64 param4;
|
| 1341 |
+
mov.b64 %rd228, assertFunc_0;
|
| 1342 |
+
cvta.global.u64 %rd229, %rd228;
|
| 1343 |
+
st.param.b64 [param3], %rd229;
|
| 1344 |
+
mov.b64 %rd230, assertFile_0;
|
| 1345 |
+
cvta.global.u64 %rd231, %rd230;
|
| 1346 |
+
st.param.b64 [param1], %rd231;
|
| 1347 |
+
mov.b64 %rd232, assertMessage_0;
|
| 1348 |
+
cvta.global.u64 %rd233, %rd232;
|
| 1349 |
+
st.param.b64 [param0], %rd233;
|
| 1350 |
+
st.param.b64 [param4], 1;
|
| 1351 |
+
st.param.b32 [param2], 32;
|
| 1352 |
+
call.uni __assertfail, (param0, param1, param2, param3, param4);
|
| 1353 |
+
} // callseq 0
|
| 1354 |
+
trap;
|
| 1355 |
+
$L__BB0_51:
|
| 1356 |
+
.loc 1 52 64 // cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py:52:64
|
| 1357 |
+
{ // callseq 1, 0
|
| 1358 |
+
.param .b64 param0;
|
| 1359 |
+
.param .b64 param1;
|
| 1360 |
+
.param .b32 param2;
|
| 1361 |
+
.param .b64 param3;
|
| 1362 |
+
.param .b64 param4;
|
| 1363 |
+
mov.b64 %rd406, assertFunc_1;
|
| 1364 |
+
cvta.global.u64 %rd407, %rd406;
|
| 1365 |
+
st.param.b64 [param3], %rd407;
|
| 1366 |
+
mov.b64 %rd408, assertFile_1;
|
| 1367 |
+
cvta.global.u64 %rd409, %rd408;
|
| 1368 |
+
st.param.b64 [param1], %rd409;
|
| 1369 |
+
mov.b64 %rd410, assertMessage_1;
|
| 1370 |
+
cvta.global.u64 %rd411, %rd410;
|
| 1371 |
+
st.param.b64 [param0], %rd411;
|
| 1372 |
+
st.param.b64 [param4], 1;
|
| 1373 |
+
st.param.b32 [param2], 52;
|
| 1374 |
+
call.uni __assertfail, (param0, param1, param2, param3, param4);
|
| 1375 |
+
} // callseq 1
|
| 1376 |
+
trap;
|
| 1377 |
+
$L__tmp1:
|
| 1378 |
+
$L__func_end0:
|
| 1379 |
+
// -- End function
|
| 1380 |
+
}
|
| 1381 |
+
.file 1 "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py"
|
| 1382 |
+
.section .debug_abbrev
|
| 1383 |
+
{
|
| 1384 |
+
.b8 1 // Abbreviation Code
|
| 1385 |
+
.b8 17 // DW_TAG_compile_unit
|
| 1386 |
+
.b8 0 // DW_CHILDREN_no
|
| 1387 |
+
.b8 37 // DW_AT_producer
|
| 1388 |
+
.b8 8 // DW_FORM_string
|
| 1389 |
+
.b8 19 // DW_AT_language
|
| 1390 |
+
.b8 5 // DW_FORM_data2
|
| 1391 |
+
.b8 3 // DW_AT_name
|
| 1392 |
+
.b8 8 // DW_FORM_string
|
| 1393 |
+
.b8 16 // DW_AT_stmt_list
|
| 1394 |
+
.b8 6 // DW_FORM_data4
|
| 1395 |
+
.b8 27 // DW_AT_comp_dir
|
| 1396 |
+
.b8 8 // DW_FORM_string
|
| 1397 |
+
.b8 0 // EOM(1)
|
| 1398 |
+
.b8 0 // EOM(2)
|
| 1399 |
+
.b8 0 // EOM(3)
|
| 1400 |
+
}
|
| 1401 |
+
.section .debug_info
|
| 1402 |
+
{
|
| 1403 |
+
.b32 135 // Length of Unit
|
| 1404 |
+
.b8 2 // DWARF version number
|
| 1405 |
+
.b8 0
|
| 1406 |
+
.b32 .debug_abbrev // Offset Into Abbrev. Section
|
| 1407 |
+
.b8 8 // Address Size (in bytes)
|
| 1408 |
+
.b8 1 // Abbrev [1] 0xb:0x80 DW_TAG_compile_unit
|
| 1409 |
+
.b8 116 // DW_AT_producer
|
| 1410 |
+
.b8 114
|
| 1411 |
+
.b8 105
|
| 1412 |
+
.b8 116
|
| 1413 |
+
.b8 111
|
| 1414 |
+
.b8 110
|
| 1415 |
+
.b8 0
|
| 1416 |
+
.b8 2 // DW_AT_language
|
| 1417 |
+
.b8 0
|
| 1418 |
+
.b8 99 // DW_AT_name
|
| 1419 |
+
.b8 97
|
| 1420 |
+
.b8 108
|
| 1421 |
+
.b8 50
|
| 1422 |
+
.b8 114
|
| 1423 |
+
.b8 52
|
| 1424 |
+
.b8 116
|
| 1425 |
+
.b8 102
|
| 1426 |
+
.b8 121
|
| 1427 |
+
.b8 119
|
| 1428 |
+
.b8 54
|
| 1429 |
+
.b8 103
|
| 1430 |
+
.b8 105
|
| 1431 |
+
.b8 99
|
| 1432 |
+
.b8 51
|
| 1433 |
+
.b8 103
|
| 1434 |
+
.b8 103
|
| 1435 |
+
.b8 113
|
| 1436 |
+
.b8 121
|
| 1437 |
+
.b8 117
|
| 1438 |
+
.b8 100
|
| 1439 |
+
.b8 51
|
| 1440 |
+
.b8 110
|
| 1441 |
+
.b8 117
|
| 1442 |
+
.b8 102
|
| 1443 |
+
.b8 110
|
| 1444 |
+
.b8 97
|
| 1445 |
+
.b8 106
|
| 1446 |
+
.b8 120
|
| 1447 |
+
.b8 54
|
| 1448 |
+
.b8 120
|
| 1449 |
+
.b8 97
|
| 1450 |
+
.b8 117
|
| 1451 |
+
.b8 50
|
| 1452 |
+
.b8 107
|
| 1453 |
+
.b8 111
|
| 1454 |
+
.b8 105
|
| 1455 |
+
.b8 101
|
| 1456 |
+
.b8 111
|
| 1457 |
+
.b8 105
|
| 1458 |
+
.b8 116
|
| 1459 |
+
.b8 120
|
| 1460 |
+
.b8 54
|
| 1461 |
+
.b8 122
|
| 1462 |
+
.b8 103
|
| 1463 |
+
.b8 52
|
| 1464 |
+
.b8 119
|
| 1465 |
+
.b8 115
|
| 1466 |
+
.b8 105
|
| 1467 |
+
.b8 111
|
| 1468 |
+
.b8 122
|
| 1469 |
+
.b8 109
|
| 1470 |
+
.b8 46
|
| 1471 |
+
.b8 112
|
| 1472 |
+
.b8 121
|
| 1473 |
+
.b8 0
|
| 1474 |
+
.b32 .debug_line // DW_AT_stmt_list
|
| 1475 |
+
.b8 47 // DW_AT_comp_dir
|
| 1476 |
+
.b8 119
|
| 1477 |
+
.b8 111
|
| 1478 |
+
.b8 114
|
| 1479 |
+
.b8 107
|
| 1480 |
+
.b8 115
|
| 1481 |
+
.b8 112
|
| 1482 |
+
.b8 97
|
| 1483 |
+
.b8 99
|
| 1484 |
+
.b8 101
|
| 1485 |
+
.b8 47
|
| 1486 |
+
.b8 104
|
| 1487 |
+
.b8 97
|
| 1488 |
+
.b8 110
|
| 1489 |
+
.b8 114
|
| 1490 |
+
.b8 117
|
| 1491 |
+
.b8 105
|
| 1492 |
+
.b8 47
|
| 1493 |
+
.b8 83
|
| 1494 |
+
.b8 112
|
| 1495 |
+
.b8 101
|
| 1496 |
+
.b8 99
|
| 1497 |
+
.b8 70
|
| 1498 |
+
.b8 111
|
| 1499 |
+
.b8 114
|
| 1500 |
+
.b8 103
|
| 1501 |
+
.b8 101
|
| 1502 |
+
.b8 45
|
| 1503 |
+
.b8 101
|
| 1504 |
+
.b8 120
|
| 1505 |
+
.b8 116
|
| 1506 |
+
.b8 47
|
| 1507 |
+
.b8 99
|
| 1508 |
+
.b8 97
|
| 1509 |
+
.b8 99
|
| 1510 |
+
.b8 104
|
| 1511 |
+
.b8 101
|
| 1512 |
+
.b8 47
|
| 1513 |
+
.b8 99
|
| 1514 |
+
.b8 111
|
| 1515 |
+
.b8 109
|
| 1516 |
+
.b8 112
|
| 1517 |
+
.b8 105
|
| 1518 |
+
.b8 108
|
| 1519 |
+
.b8 101
|
| 1520 |
+
.b8 100
|
| 1521 |
+
.b8 95
|
| 1522 |
+
.b8 107
|
| 1523 |
+
.b8 101
|
| 1524 |
+
.b8 114
|
| 1525 |
+
.b8 110
|
| 1526 |
+
.b8 101
|
| 1527 |
+
.b8 108
|
| 1528 |
+
.b8 115
|
| 1529 |
+
.b8 47
|
| 1530 |
+
.b8 97
|
| 1531 |
+
.b8 108
|
| 1532 |
+
.b8 0
|
| 1533 |
+
}
|
| 1534 |
+
.section .debug_macinfo { }
|
SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.source
ADDED
|
@@ -0,0 +1,299 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":18:0)
|
| 2 |
+
#loc78 = loc("in_ptr0"(#loc))
|
| 3 |
+
#loc79 = loc("in_ptr1"(#loc))
|
| 4 |
+
#loc80 = loc("in_ptr2"(#loc))
|
| 5 |
+
#loc81 = loc("in_ptr3"(#loc))
|
| 6 |
+
#loc82 = loc("out_ptr0"(#loc))
|
| 7 |
+
#loc83 = loc("ks0"(#loc))
|
| 8 |
+
#loc84 = loc("ks1"(#loc))
|
| 9 |
+
#loc85 = loc("ks2"(#loc))
|
| 10 |
+
#loc86 = loc("ks3"(#loc))
|
| 11 |
+
#loc87 = loc("ks4"(#loc))
|
| 12 |
+
#loc88 = loc("xnumel"(#loc))
|
| 13 |
+
module {
|
| 14 |
+
tt.func public @triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0(%in_ptr0: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %in_ptr1: !tt.ptr<i64> {tt.divisibility = 16 : i32} loc("in_ptr1"(#loc)), %in_ptr2: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("in_ptr2"(#loc)), %in_ptr3: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("in_ptr3"(#loc)), %out_ptr0: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("out_ptr0"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %ks2: i64 loc("ks2"(#loc)), %ks3: i64 loc("ks3"(#loc)), %ks4: i64 loc("ks4"(#loc)), %xnumel: i32 loc("xnumel"(#loc))) attributes {noinline = false} {
|
| 15 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc89)
|
| 16 |
+
%xoffset_0 = arith.constant 1024 : i32 loc(#loc90)
|
| 17 |
+
%xoffset_1 = arith.constant 1024 : i32 loc(#loc90)
|
| 18 |
+
%xoffset_2 = arith.muli %xoffset, %xoffset_1 : i32 loc(#loc90)
|
| 19 |
+
%xindex = tt.make_range {end = 1024 : i32, start = 0 : i32} : tensor<1024xi32> loc(#loc91)
|
| 20 |
+
%xindex_3 = tt.splat %xoffset_2 : i32 -> tensor<1024xi32> loc(#loc92)
|
| 21 |
+
%xindex_4 = arith.addi %xindex_3, %xindex : tensor<1024xi32> loc(#loc92)
|
| 22 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<1024xi32> loc(#loc93)
|
| 23 |
+
%xmask_5 = arith.cmpi slt, %xindex_4, %xmask : tensor<1024xi32> loc(#loc93)
|
| 24 |
+
%x2 = arith.extsi %xindex_4 : tensor<1024xi32> to tensor<1024xi64> loc(#loc94)
|
| 25 |
+
%x2_6 = tt.splat %ks0 : i64 -> tensor<1024xi64> loc(#loc94)
|
| 26 |
+
%x2_7 = arith.divsi %x2, %x2_6 : tensor<1024xi64> loc(#loc94)
|
| 27 |
+
%x2_8 = tt.splat %ks1 : i64 -> tensor<1024xi64> loc(#loc95)
|
| 28 |
+
%x2_9 = arith.remsi %x2_7, %x2_8 : tensor<1024xi64> loc(#loc95)
|
| 29 |
+
%x0 = arith.extsi %xindex_4 : tensor<1024xi32> to tensor<1024xi64> loc(#loc96)
|
| 30 |
+
%x0_10 = tt.splat %ks3 : i64 -> tensor<1024xi64> loc(#loc96)
|
| 31 |
+
%x0_11 = arith.remsi %x0, %x0_10 : tensor<1024xi64> loc(#loc96)
|
| 32 |
+
%x5 = arith.extsi %xindex_4 : tensor<1024xi32> to tensor<1024xi64> loc(#loc97)
|
| 33 |
+
%x5_12 = tt.splat %ks3 : i64 -> tensor<1024xi64> loc(#loc97)
|
| 34 |
+
%x5_13 = arith.divsi %x5, %x5_12 : tensor<1024xi64> loc(#loc97)
|
| 35 |
+
%tmp0 = tt.splat %in_ptr0 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>> loc(#loc98)
|
| 36 |
+
%tmp0_14 = tt.addptr %tmp0, %xindex_4 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi32> loc(#loc98)
|
| 37 |
+
%tmp0_15 = tt.load %tmp0_14, %xmask_5 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>> loc(#loc99)
|
| 38 |
+
%tmp0_16 = arith.extf %tmp0_15 : tensor<1024xbf16> to tensor<1024xf32> loc(#loc100)
|
| 39 |
+
%tmp1 = tt.splat %in_ptr1 : !tt.ptr<i64> -> tensor<1024x!tt.ptr<i64>> loc(#loc101)
|
| 40 |
+
%tmp1_17 = tt.addptr %tmp1, %x2_9 : tensor<1024x!tt.ptr<i64>>, tensor<1024xi64> loc(#loc101)
|
| 41 |
+
%tmp1_18 = tt.load %tmp1_17, %xmask_5 evictionPolicy = evict_last : tensor<1024x!tt.ptr<i64>> loc(#loc102)
|
| 42 |
+
%tmp3 = tt.splat %ks2 : i64 -> tensor<1024xi64> loc(#loc103)
|
| 43 |
+
%tmp3_19 = arith.addi %tmp1_18, %tmp3 : tensor<1024xi64> loc(#loc103)
|
| 44 |
+
%tmp4 = arith.constant 0 : i32 loc(#loc104)
|
| 45 |
+
%tmp4_20 = arith.extsi %tmp4 : i32 to i64 loc(#loc104)
|
| 46 |
+
%tmp4_21 = tt.splat %tmp4_20 : i64 -> tensor<1024xi64> loc(#loc104)
|
| 47 |
+
%tmp4_22 = arith.cmpi slt, %tmp1_18, %tmp4_21 : tensor<1024xi64> loc(#loc104)
|
| 48 |
+
%tmp5 = arith.select %tmp4_22, %tmp3_19, %tmp1_18 : tensor<1024xi1>, tensor<1024xi64> loc(#loc105)
|
| 49 |
+
%c0_i32 = arith.constant 0 : i32 loc(#loc18)
|
| 50 |
+
%0 = arith.extsi %c0_i32 : i32 to i64 loc(#loc18)
|
| 51 |
+
%1 = tt.splat %0 : i64 -> tensor<1024xi64> loc(#loc18)
|
| 52 |
+
%2 = arith.cmpi sle, %1, %tmp5 : tensor<1024xi64> loc(#loc18)
|
| 53 |
+
%3 = tt.splat %ks2 : i64 -> tensor<1024xi64> loc(#loc19)
|
| 54 |
+
%4 = arith.cmpi slt, %tmp5, %3 : tensor<1024xi64> loc(#loc19)
|
| 55 |
+
%5 = arith.andi %2, %4 : tensor<1024xi1> loc(#loc20)
|
| 56 |
+
%true = arith.constant true loc(#loc21)
|
| 57 |
+
%cst = arith.constant dense<true> : tensor<1024xi1> loc(#loc21)
|
| 58 |
+
%6 = arith.xori %xmask_5, %cst : tensor<1024xi1> loc(#loc21)
|
| 59 |
+
%7 = arith.ori %5, %6 : tensor<1024xi1> loc(#loc22)
|
| 60 |
+
tt.assert %7, "index out of bounds: 0 <= tmp5 < ks2" : tensor<1024xi1> loc(#loc23)
|
| 61 |
+
%tmp7 = tt.splat %ks3 : i64 -> tensor<1024xi64> loc(#loc106)
|
| 62 |
+
%tmp7_23 = arith.muli %tmp7, %tmp5 : tensor<1024xi64> loc(#loc106)
|
| 63 |
+
%tmp7_24 = arith.addi %x0_11, %tmp7_23 : tensor<1024xi64> loc(#loc107)
|
| 64 |
+
%tmp7_25 = tt.splat %in_ptr2 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>> loc(#loc108)
|
| 65 |
+
%tmp7_26 = tt.addptr %tmp7_25, %tmp7_24 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi64> loc(#loc108)
|
| 66 |
+
%tmp7_27 = tt.load %tmp7_26, %xmask_5 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>> loc(#loc109)
|
| 67 |
+
%tmp7_28 = arith.extf %tmp7_27 : tensor<1024xbf16> to tensor<1024xf32> loc(#loc110)
|
| 68 |
+
%tmp8 = arith.mulf %tmp0_16, %tmp7_28 : tensor<1024xf32> loc(#loc111)
|
| 69 |
+
%tmp10 = arith.constant 0 : i64 loc(#loc112)
|
| 70 |
+
%tmp10_29 = arith.constant dense<0> : tensor<1xi64> loc(#loc112)
|
| 71 |
+
%tmp11 = arith.constant dense<0> : tensor<1024xi64> loc(#loc113)
|
| 72 |
+
%tmp11_30 = arith.cmpi sge, %x0_11, %tmp11 : tensor<1024xi64> loc(#loc113)
|
| 73 |
+
%tmp12 = arith.constant 2 : i32 loc(#loc114)
|
| 74 |
+
%tmp12_31 = arith.constant 2 : i64 loc(#loc114)
|
| 75 |
+
%tmp12_32 = arith.divsi %ks3, %tmp12_31 : i64 loc(#loc114)
|
| 76 |
+
%tmp12_33 = arith.constant -1 : i32 loc(#loc115)
|
| 77 |
+
%tmp12_34 = arith.constant -1 : i64 loc(#loc115)
|
| 78 |
+
%tmp12_35 = arith.muli %tmp12_34, %tmp12_32 : i64 loc(#loc115)
|
| 79 |
+
%tmp12_36 = arith.addi %ks3, %tmp12_35 : i64 loc(#loc116)
|
| 80 |
+
%tmp13 = tt.splat %tmp12_36 : i64 -> tensor<1024xi64> loc(#loc117)
|
| 81 |
+
%tmp13_37 = arith.cmpi slt, %x0_11, %tmp13 : tensor<1024xi64> loc(#loc117)
|
| 82 |
+
%tmp14 = tt.splat %ks3 : i64 -> tensor<1024xi64> loc(#loc118)
|
| 83 |
+
%tmp14_38 = arith.muli %tmp14, %x5_13 : tensor<1024xi64> loc(#loc118)
|
| 84 |
+
%tmp14_39 = arith.constant 2 : i32 loc(#loc119)
|
| 85 |
+
%tmp14_40 = arith.constant 2 : i64 loc(#loc119)
|
| 86 |
+
%tmp14_41 = arith.divsi %ks3, %tmp14_40 : i64 loc(#loc119)
|
| 87 |
+
%tmp14_42 = tt.splat %tmp14_41 : i64 -> tensor<1024xi64> loc(#loc120)
|
| 88 |
+
%tmp14_43 = arith.addi %tmp14_38, %tmp14_42 : tensor<1024xi64> loc(#loc120)
|
| 89 |
+
%tmp14_44 = arith.addi %tmp14_43, %x0_11 : tensor<1024xi64> loc(#loc121)
|
| 90 |
+
%tmp14_45 = tt.splat %in_ptr0 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>> loc(#loc122)
|
| 91 |
+
%tmp14_46 = tt.addptr %tmp14_45, %tmp14_44 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi64> loc(#loc122)
|
| 92 |
+
%tmp14_47 = arith.andi %tmp13_37, %xmask_5 : tensor<1024xi1> loc(#loc123)
|
| 93 |
+
%tmp14_48 = arith.constant 0.000000e+00 : f32 loc(#loc124)
|
| 94 |
+
%tmp14_49 = arith.constant dense<0.000000e+00> : tensor<1024xf32> loc(#loc124)
|
| 95 |
+
%tmp14_50 = arith.truncf %tmp14_49 : tensor<1024xf32> to tensor<1024xbf16> loc(#loc124)
|
| 96 |
+
%tmp14_51 = tt.load %tmp14_46, %tmp14_47, %tmp14_50 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>> loc(#loc124)
|
| 97 |
+
%tmp14_52 = arith.extf %tmp14_51 : tensor<1024xbf16> to tensor<1024xf32> loc(#loc125)
|
| 98 |
+
%tmp15 = arith.constant 0.000000e+00 : f32 loc(#loc126)
|
| 99 |
+
%tmp15_53 = arith.constant dense<0.000000e+00> : tensor<1024xf32> loc(#loc126)
|
| 100 |
+
%tmp15_54 = arith.subf %tmp15_53, %tmp14_52 : tensor<1024xf32> loc(#loc126)
|
| 101 |
+
%tmp16 = arith.constant 0.000000e+00 : f32 loc(#loc127)
|
| 102 |
+
%tmp16_55 = arith.constant dense<0.000000e+00> : tensor<1024xf32> loc(#loc127)
|
| 103 |
+
%tmp17 = arith.select %tmp13_37, %tmp15_54, %tmp16_55 : tensor<1024xi1>, tensor<1024xf32> loc(#loc128)
|
| 104 |
+
%tmp18 = tt.splat %tmp12_36 : i64 -> tensor<1024xi64> loc(#loc129)
|
| 105 |
+
%tmp18_56 = arith.cmpi sge, %x0_11, %tmp18 : tensor<1024xi64> loc(#loc129)
|
| 106 |
+
%tmp20 = tt.splat %ks3 : i64 -> tensor<1024xi64> loc(#loc130)
|
| 107 |
+
%tmp20_57 = arith.cmpi slt, %x0_11, %tmp20 : tensor<1024xi64> loc(#loc130)
|
| 108 |
+
%tmp21 = tt.splat %ks3 : i64 -> tensor<1024xi64> loc(#loc131)
|
| 109 |
+
%tmp21_58 = arith.muli %tmp21, %x5_13 : tensor<1024xi64> loc(#loc131)
|
| 110 |
+
%tmp21_59 = arith.constant -1 : i32 loc(#loc132)
|
| 111 |
+
%tmp21_60 = arith.constant -1 : i64 loc(#loc132)
|
| 112 |
+
%tmp21_61 = arith.muli %tmp21_60, %ks3 : i64 loc(#loc132)
|
| 113 |
+
%tmp21_62 = tt.splat %tmp21_61 : i64 -> tensor<1024xi64> loc(#loc133)
|
| 114 |
+
%tmp21_63 = arith.addi %x0_11, %tmp21_62 : tensor<1024xi64> loc(#loc133)
|
| 115 |
+
%tmp21_64 = arith.constant 2 : i32 loc(#loc134)
|
| 116 |
+
%tmp21_65 = arith.constant 2 : i64 loc(#loc134)
|
| 117 |
+
%tmp21_66 = arith.divsi %ks3, %tmp21_65 : i64 loc(#loc134)
|
| 118 |
+
%tmp21_67 = tt.splat %tmp21_66 : i64 -> tensor<1024xi64> loc(#loc135)
|
| 119 |
+
%tmp21_68 = arith.addi %tmp21_63, %tmp21_67 : tensor<1024xi64> loc(#loc135)
|
| 120 |
+
%tmp21_69 = arith.addi %tmp21_58, %tmp21_68 : tensor<1024xi64> loc(#loc136)
|
| 121 |
+
%tmp21_70 = tt.splat %in_ptr0 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>> loc(#loc137)
|
| 122 |
+
%tmp21_71 = tt.addptr %tmp21_70, %tmp21_69 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi64> loc(#loc137)
|
| 123 |
+
%tmp21_72 = arith.andi %tmp18_56, %xmask_5 : tensor<1024xi1> loc(#loc138)
|
| 124 |
+
%tmp21_73 = arith.constant 0.000000e+00 : f32 loc(#loc139)
|
| 125 |
+
%tmp21_74 = arith.constant dense<0.000000e+00> : tensor<1024xf32> loc(#loc139)
|
| 126 |
+
%tmp21_75 = arith.truncf %tmp21_74 : tensor<1024xf32> to tensor<1024xbf16> loc(#loc139)
|
| 127 |
+
%tmp21_76 = tt.load %tmp21_71, %tmp21_72, %tmp21_75 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>> loc(#loc139)
|
| 128 |
+
%tmp21_77 = arith.extf %tmp21_76 : tensor<1024xbf16> to tensor<1024xf32> loc(#loc140)
|
| 129 |
+
%tmp22 = arith.select %tmp13_37, %tmp17, %tmp21_77 : tensor<1024xi1>, tensor<1024xf32> loc(#loc141)
|
| 130 |
+
%tmp24 = tt.splat %ks4 : i64 -> tensor<1024xi64> loc(#loc142)
|
| 131 |
+
%tmp24_78 = arith.addi %tmp1_18, %tmp24 : tensor<1024xi64> loc(#loc142)
|
| 132 |
+
%tmp25 = arith.select %tmp4_22, %tmp24_78, %tmp1_18 : tensor<1024xi1>, tensor<1024xi64> loc(#loc143)
|
| 133 |
+
%c0_i32_79 = arith.constant 0 : i32 loc(#loc62)
|
| 134 |
+
%8 = arith.extsi %c0_i32_79 : i32 to i64 loc(#loc62)
|
| 135 |
+
%9 = tt.splat %8 : i64 -> tensor<1024xi64> loc(#loc62)
|
| 136 |
+
%10 = arith.cmpi sle, %9, %tmp25 : tensor<1024xi64> loc(#loc62)
|
| 137 |
+
%11 = tt.splat %ks4 : i64 -> tensor<1024xi64> loc(#loc63)
|
| 138 |
+
%12 = arith.cmpi slt, %tmp25, %11 : tensor<1024xi64> loc(#loc63)
|
| 139 |
+
%13 = arith.andi %10, %12 : tensor<1024xi1> loc(#loc64)
|
| 140 |
+
%true_80 = arith.constant true loc(#loc65)
|
| 141 |
+
%cst_81 = arith.constant dense<true> : tensor<1024xi1> loc(#loc65)
|
| 142 |
+
%14 = arith.xori %xmask_5, %cst_81 : tensor<1024xi1> loc(#loc65)
|
| 143 |
+
%15 = arith.ori %13, %14 : tensor<1024xi1> loc(#loc66)
|
| 144 |
+
tt.assert %15, "index out of bounds: 0 <= tmp25 < ks4" : tensor<1024xi1> loc(#loc67)
|
| 145 |
+
%tmp27 = tt.splat %ks3 : i64 -> tensor<1024xi64> loc(#loc144)
|
| 146 |
+
%tmp27_82 = arith.muli %tmp27, %tmp25 : tensor<1024xi64> loc(#loc144)
|
| 147 |
+
%tmp27_83 = arith.addi %x0_11, %tmp27_82 : tensor<1024xi64> loc(#loc145)
|
| 148 |
+
%tmp27_84 = tt.splat %in_ptr3 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>> loc(#loc146)
|
| 149 |
+
%tmp27_85 = tt.addptr %tmp27_84, %tmp27_83 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi64> loc(#loc146)
|
| 150 |
+
%tmp27_86 = tt.load %tmp27_85, %xmask_5 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>> loc(#loc147)
|
| 151 |
+
%tmp27_87 = arith.extf %tmp27_86 : tensor<1024xbf16> to tensor<1024xf32> loc(#loc148)
|
| 152 |
+
%tmp28 = arith.mulf %tmp22, %tmp27_87 : tensor<1024xf32> loc(#loc149)
|
| 153 |
+
%tmp29 = arith.addf %tmp8, %tmp28 : tensor<1024xf32> loc(#loc150)
|
| 154 |
+
%16 = tt.splat %out_ptr0 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>> loc(#loc75)
|
| 155 |
+
%17 = tt.addptr %16, %xindex_4 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi32> loc(#loc75)
|
| 156 |
+
%18 = arith.truncf %tmp29 : tensor<1024xf32> to tensor<1024xbf16> loc(#loc76)
|
| 157 |
+
tt.store %17, %18, %xmask_5 : tensor<1024x!tt.ptr<bf16>> loc(#loc76)
|
| 158 |
+
tt.return loc(#loc77)
|
| 159 |
+
} loc(#loc)
|
| 160 |
+
} loc(#loc)
|
| 161 |
+
#loc1 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":19:28)
|
| 162 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":19:33)
|
| 163 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":20:36)
|
| 164 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":20:23)
|
| 165 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":21:21)
|
| 166 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":23:21)
|
| 167 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":23:28)
|
| 168 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":24:19)
|
| 169 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":25:19)
|
| 170 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":26:30)
|
| 171 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":26:35)
|
| 172 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":26:75)
|
| 173 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":27:30)
|
| 174 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":27:35)
|
| 175 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":29:18)
|
| 176 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":30:18)
|
| 177 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":31:32)
|
| 178 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:28)
|
| 179 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:44)
|
| 180 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:37)
|
| 181 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:54)
|
| 182 |
+
#loc22 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:52)
|
| 183 |
+
#loc23 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:62)
|
| 184 |
+
#loc24 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:39)
|
| 185 |
+
#loc25 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:35)
|
| 186 |
+
#loc26 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:30)
|
| 187 |
+
#loc27 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:46)
|
| 188 |
+
#loc28 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:86)
|
| 189 |
+
#loc29 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":34:18)
|
| 190 |
+
#loc30 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":36:28)
|
| 191 |
+
#loc31 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":37:20)
|
| 192 |
+
#loc32 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":38:31)
|
| 193 |
+
#loc33 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":38:24)
|
| 194 |
+
#loc34 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":38:18)
|
| 195 |
+
#loc35 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":39:19)
|
| 196 |
+
#loc36 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:35)
|
| 197 |
+
#loc37 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:48)
|
| 198 |
+
#loc38 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:41)
|
| 199 |
+
#loc39 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:54)
|
| 200 |
+
#loc40 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:31)
|
| 201 |
+
#loc41 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:68)
|
| 202 |
+
#loc42 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:60)
|
| 203 |
+
#loc43 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:119)
|
| 204 |
+
#loc44 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":41:13)
|
| 205 |
+
#loc45 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":42:38)
|
| 206 |
+
#loc46 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":43:35)
|
| 207 |
+
#loc47 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":44:20)
|
| 208 |
+
#loc48 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":46:19)
|
| 209 |
+
#loc49 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:35)
|
| 210 |
+
#loc50 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:52)
|
| 211 |
+
#loc51 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:47)
|
| 212 |
+
#loc52 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:67)
|
| 213 |
+
#loc53 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:60)
|
| 214 |
+
#loc54 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:41)
|
| 215 |
+
#loc55 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:31)
|
| 216 |
+
#loc56 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:81)
|
| 217 |
+
#loc57 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:73)
|
| 218 |
+
#loc58 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:132)
|
| 219 |
+
#loc59 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":48:35)
|
| 220 |
+
#loc60 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":50:19)
|
| 221 |
+
#loc61 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":51:34)
|
| 222 |
+
#loc62 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:28)
|
| 223 |
+
#loc63 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:46)
|
| 224 |
+
#loc64 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:38)
|
| 225 |
+
#loc65 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:56)
|
| 226 |
+
#loc66 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:54)
|
| 227 |
+
#loc67 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:64)
|
| 228 |
+
#loc68 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:40)
|
| 229 |
+
#loc69 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:36)
|
| 230 |
+
#loc70 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:31)
|
| 231 |
+
#loc71 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:48)
|
| 232 |
+
#loc72 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:88)
|
| 233 |
+
#loc73 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":54:20)
|
| 234 |
+
#loc74 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":55:19)
|
| 235 |
+
#loc75 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":56:25)
|
| 236 |
+
#loc76 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":56:37)
|
| 237 |
+
#loc77 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":56:4)
|
| 238 |
+
#loc89 = loc("xoffset"(#loc1))
|
| 239 |
+
#loc90 = loc("xoffset"(#loc2))
|
| 240 |
+
#loc91 = loc("xindex"(#loc3))
|
| 241 |
+
#loc92 = loc("xindex"(#loc4))
|
| 242 |
+
#loc93 = loc("xmask"(#loc5))
|
| 243 |
+
#loc94 = loc("x2"(#loc6))
|
| 244 |
+
#loc95 = loc("x2"(#loc7))
|
| 245 |
+
#loc96 = loc("x0"(#loc8))
|
| 246 |
+
#loc97 = loc("x5"(#loc9))
|
| 247 |
+
#loc98 = loc("tmp0"(#loc10))
|
| 248 |
+
#loc99 = loc("tmp0"(#loc11))
|
| 249 |
+
#loc100 = loc("tmp0"(#loc12))
|
| 250 |
+
#loc101 = loc("tmp1"(#loc13))
|
| 251 |
+
#loc102 = loc("tmp1"(#loc14))
|
| 252 |
+
#loc103 = loc("tmp3"(#loc15))
|
| 253 |
+
#loc104 = loc("tmp4"(#loc16))
|
| 254 |
+
#loc105 = loc("tmp5"(#loc17))
|
| 255 |
+
#loc106 = loc("tmp7"(#loc24))
|
| 256 |
+
#loc107 = loc("tmp7"(#loc25))
|
| 257 |
+
#loc108 = loc("tmp7"(#loc26))
|
| 258 |
+
#loc109 = loc("tmp7"(#loc27))
|
| 259 |
+
#loc110 = loc("tmp7"(#loc28))
|
| 260 |
+
#loc111 = loc("tmp8"(#loc29))
|
| 261 |
+
#loc112 = loc("tmp10"(#loc30))
|
| 262 |
+
#loc113 = loc("tmp11"(#loc31))
|
| 263 |
+
#loc114 = loc("tmp12"(#loc32))
|
| 264 |
+
#loc115 = loc("tmp12"(#loc33))
|
| 265 |
+
#loc116 = loc("tmp12"(#loc34))
|
| 266 |
+
#loc117 = loc("tmp13"(#loc35))
|
| 267 |
+
#loc118 = loc("tmp14"(#loc36))
|
| 268 |
+
#loc119 = loc("tmp14"(#loc37))
|
| 269 |
+
#loc120 = loc("tmp14"(#loc38))
|
| 270 |
+
#loc121 = loc("tmp14"(#loc39))
|
| 271 |
+
#loc122 = loc("tmp14"(#loc40))
|
| 272 |
+
#loc123 = loc("tmp14"(#loc41))
|
| 273 |
+
#loc124 = loc("tmp14"(#loc42))
|
| 274 |
+
#loc125 = loc("tmp14"(#loc43))
|
| 275 |
+
#loc126 = loc("tmp15"(#loc44))
|
| 276 |
+
#loc127 = loc("tmp16"(#loc45))
|
| 277 |
+
#loc128 = loc("tmp17"(#loc46))
|
| 278 |
+
#loc129 = loc("tmp18"(#loc47))
|
| 279 |
+
#loc130 = loc("tmp20"(#loc48))
|
| 280 |
+
#loc131 = loc("tmp21"(#loc49))
|
| 281 |
+
#loc132 = loc("tmp21"(#loc50))
|
| 282 |
+
#loc133 = loc("tmp21"(#loc51))
|
| 283 |
+
#loc134 = loc("tmp21"(#loc52))
|
| 284 |
+
#loc135 = loc("tmp21"(#loc53))
|
| 285 |
+
#loc136 = loc("tmp21"(#loc54))
|
| 286 |
+
#loc137 = loc("tmp21"(#loc55))
|
| 287 |
+
#loc138 = loc("tmp21"(#loc56))
|
| 288 |
+
#loc139 = loc("tmp21"(#loc57))
|
| 289 |
+
#loc140 = loc("tmp21"(#loc58))
|
| 290 |
+
#loc141 = loc("tmp22"(#loc59))
|
| 291 |
+
#loc142 = loc("tmp24"(#loc60))
|
| 292 |
+
#loc143 = loc("tmp25"(#loc61))
|
| 293 |
+
#loc144 = loc("tmp27"(#loc68))
|
| 294 |
+
#loc145 = loc("tmp27"(#loc69))
|
| 295 |
+
#loc146 = loc("tmp27"(#loc70))
|
| 296 |
+
#loc147 = loc("tmp27"(#loc71))
|
| 297 |
+
#loc148 = loc("tmp27"(#loc72))
|
| 298 |
+
#loc149 = loc("tmp28"(#loc73))
|
| 299 |
+
#loc150 = loc("tmp29"(#loc74))
|
SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ttgir
ADDED
|
@@ -0,0 +1,232 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#blocked = #ttg.blocked<{sizePerThread = [8], threadsPerWarp = [32], warpsPerCTA = [4], order = [0]}>
|
| 2 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":18:0)
|
| 3 |
+
#loc70 = loc("in_ptr0"(#loc))
|
| 4 |
+
#loc71 = loc("in_ptr1"(#loc))
|
| 5 |
+
#loc72 = loc("in_ptr2"(#loc))
|
| 6 |
+
#loc73 = loc("in_ptr3"(#loc))
|
| 7 |
+
#loc74 = loc("out_ptr0"(#loc))
|
| 8 |
+
#loc75 = loc("ks0"(#loc))
|
| 9 |
+
#loc76 = loc("ks1"(#loc))
|
| 10 |
+
#loc77 = loc("ks2"(#loc))
|
| 11 |
+
#loc78 = loc("ks3"(#loc))
|
| 12 |
+
#loc79 = loc("ks4"(#loc))
|
| 13 |
+
#loc80 = loc("xnumel"(#loc))
|
| 14 |
+
module attributes {"ttg.num-ctas" = 1 : i32, "ttg.num-warps" = 4 : i32, ttg.target = "cuda:90", "ttg.threads-per-warp" = 32 : i32} {
|
| 15 |
+
tt.func public @triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0(%in_ptr0: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %in_ptr1: !tt.ptr<i64> {tt.divisibility = 16 : i32} loc("in_ptr1"(#loc)), %in_ptr2: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("in_ptr2"(#loc)), %in_ptr3: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("in_ptr3"(#loc)), %out_ptr0: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("out_ptr0"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %ks2: i64 loc("ks2"(#loc)), %ks3: i64 loc("ks3"(#loc)), %ks4: i64 loc("ks4"(#loc)), %xnumel: i32 loc("xnumel"(#loc))) attributes {noinline = false} {
|
| 16 |
+
%cst = arith.constant dense<true> : tensor<1024xi1, #blocked> loc(#loc1)
|
| 17 |
+
%c1024_i32 = arith.constant 1024 : i32 loc(#loc1)
|
| 18 |
+
%cst_0 = arith.constant dense<0.000000e+00> : tensor<1024xbf16, #blocked> loc(#loc1)
|
| 19 |
+
%c2_i64 = arith.constant 2 : i64 loc(#loc1)
|
| 20 |
+
%c-1_i64 = arith.constant -1 : i64 loc(#loc1)
|
| 21 |
+
%cst_1 = arith.constant dense<0> : tensor<1024xi64, #blocked> loc(#loc1)
|
| 22 |
+
%cst_2 = arith.constant dense<0.000000e+00> : tensor<1024xf32, #blocked> loc(#loc1)
|
| 23 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc81)
|
| 24 |
+
%xoffset_3 = arith.muli %xoffset, %c1024_i32 : i32 loc(#loc82)
|
| 25 |
+
%xindex = tt.make_range {end = 1024 : i32, start = 0 : i32} : tensor<1024xi32, #blocked> loc(#loc83)
|
| 26 |
+
%xindex_4 = tt.splat %xoffset_3 : i32 -> tensor<1024xi32, #blocked> loc(#loc84)
|
| 27 |
+
%xindex_5 = arith.addi %xindex_4, %xindex : tensor<1024xi32, #blocked> loc(#loc84)
|
| 28 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<1024xi32, #blocked> loc(#loc85)
|
| 29 |
+
%xmask_6 = arith.cmpi slt, %xindex_5, %xmask : tensor<1024xi32, #blocked> loc(#loc85)
|
| 30 |
+
%x2 = arith.extsi %xindex_5 : tensor<1024xi32, #blocked> to tensor<1024xi64, #blocked> loc(#loc86)
|
| 31 |
+
%x2_7 = tt.splat %ks0 : i64 -> tensor<1024xi64, #blocked> loc(#loc86)
|
| 32 |
+
%x2_8 = arith.divsi %x2, %x2_7 : tensor<1024xi64, #blocked> loc(#loc86)
|
| 33 |
+
%x2_9 = tt.splat %ks1 : i64 -> tensor<1024xi64, #blocked> loc(#loc87)
|
| 34 |
+
%x2_10 = arith.remsi %x2_8, %x2_9 : tensor<1024xi64, #blocked> loc(#loc87)
|
| 35 |
+
%x0 = tt.splat %ks3 : i64 -> tensor<1024xi64, #blocked> loc(#loc88)
|
| 36 |
+
%x0_11 = arith.remsi %x2, %x0 : tensor<1024xi64, #blocked> loc(#loc88)
|
| 37 |
+
%x5 = arith.divsi %x2, %x0 : tensor<1024xi64, #blocked> loc(#loc89)
|
| 38 |
+
%tmp0 = tt.splat %in_ptr0 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>, #blocked> loc(#loc90)
|
| 39 |
+
%tmp0_12 = tt.addptr %tmp0, %xindex_5 : tensor<1024x!tt.ptr<bf16>, #blocked>, tensor<1024xi32, #blocked> loc(#loc90)
|
| 40 |
+
%tmp0_13 = tt.load %tmp0_12, %xmask_6 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>, #blocked> loc(#loc91)
|
| 41 |
+
%tmp0_14 = arith.extf %tmp0_13 : tensor<1024xbf16, #blocked> to tensor<1024xf32, #blocked> loc(#loc92)
|
| 42 |
+
%tmp1 = tt.splat %in_ptr1 : !tt.ptr<i64> -> tensor<1024x!tt.ptr<i64>, #blocked> loc(#loc93)
|
| 43 |
+
%tmp1_15 = tt.addptr %tmp1, %x2_10 : tensor<1024x!tt.ptr<i64>, #blocked>, tensor<1024xi64, #blocked> loc(#loc93)
|
| 44 |
+
%tmp1_16 = tt.load %tmp1_15, %xmask_6 evictionPolicy = evict_last : tensor<1024x!tt.ptr<i64>, #blocked> loc(#loc94)
|
| 45 |
+
%tmp3 = tt.splat %ks2 : i64 -> tensor<1024xi64, #blocked> loc(#loc95)
|
| 46 |
+
%tmp3_17 = arith.addi %tmp1_16, %tmp3 : tensor<1024xi64, #blocked> loc(#loc95)
|
| 47 |
+
%tmp4 = arith.cmpi slt, %tmp1_16, %cst_1 : tensor<1024xi64, #blocked> loc(#loc96)
|
| 48 |
+
%tmp5 = arith.select %tmp4, %tmp3_17, %tmp1_16 : tensor<1024xi1, #blocked>, tensor<1024xi64, #blocked> loc(#loc97)
|
| 49 |
+
%0 = arith.cmpi sge, %tmp5, %cst_1 : tensor<1024xi64, #blocked> loc(#loc19)
|
| 50 |
+
%1 = arith.cmpi slt, %tmp5, %tmp3 : tensor<1024xi64, #blocked> loc(#loc20)
|
| 51 |
+
%2 = arith.andi %0, %1 : tensor<1024xi1, #blocked> loc(#loc21)
|
| 52 |
+
%3 = arith.xori %xmask_6, %cst : tensor<1024xi1, #blocked> loc(#loc22)
|
| 53 |
+
%4 = arith.ori %2, %3 : tensor<1024xi1, #blocked> loc(#loc23)
|
| 54 |
+
tt.assert %4, "index out of bounds: 0 <= tmp5 < ks2" : tensor<1024xi1, #blocked> loc(#loc24)
|
| 55 |
+
%tmp7 = arith.muli %x0, %tmp5 : tensor<1024xi64, #blocked> loc(#loc98)
|
| 56 |
+
%tmp7_18 = arith.addi %x0_11, %tmp7 : tensor<1024xi64, #blocked> loc(#loc99)
|
| 57 |
+
%tmp7_19 = tt.splat %in_ptr2 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>, #blocked> loc(#loc100)
|
| 58 |
+
%tmp7_20 = tt.addptr %tmp7_19, %tmp7_18 : tensor<1024x!tt.ptr<bf16>, #blocked>, tensor<1024xi64, #blocked> loc(#loc100)
|
| 59 |
+
%tmp7_21 = tt.load %tmp7_20, %xmask_6 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>, #blocked> loc(#loc101)
|
| 60 |
+
%tmp7_22 = arith.extf %tmp7_21 : tensor<1024xbf16, #blocked> to tensor<1024xf32, #blocked> loc(#loc102)
|
| 61 |
+
%tmp8 = arith.mulf %tmp0_14, %tmp7_22 : tensor<1024xf32, #blocked> loc(#loc103)
|
| 62 |
+
%tmp12 = arith.divsi %ks3, %c2_i64 : i64 loc(#loc104)
|
| 63 |
+
%tmp12_23 = arith.subi %ks3, %tmp12 : i64 loc(#loc105)
|
| 64 |
+
%tmp13 = tt.splat %tmp12_23 : i64 -> tensor<1024xi64, #blocked> loc(#loc106)
|
| 65 |
+
%tmp13_24 = arith.cmpi slt, %x0_11, %tmp13 : tensor<1024xi64, #blocked> loc(#loc106)
|
| 66 |
+
%tmp14 = arith.muli %x0, %x5 : tensor<1024xi64, #blocked> loc(#loc107)
|
| 67 |
+
%tmp14_25 = tt.splat %tmp12 : i64 -> tensor<1024xi64, #blocked> loc(#loc108)
|
| 68 |
+
%tmp14_26 = arith.addi %tmp14, %tmp14_25 : tensor<1024xi64, #blocked> loc(#loc108)
|
| 69 |
+
%tmp14_27 = arith.addi %tmp14_26, %x0_11 : tensor<1024xi64, #blocked> loc(#loc109)
|
| 70 |
+
%tmp14_28 = tt.addptr %tmp0, %tmp14_27 : tensor<1024x!tt.ptr<bf16>, #blocked>, tensor<1024xi64, #blocked> loc(#loc110)
|
| 71 |
+
%tmp14_29 = arith.andi %tmp13_24, %xmask_6 : tensor<1024xi1, #blocked> loc(#loc111)
|
| 72 |
+
%tmp14_30 = tt.load %tmp14_28, %tmp14_29, %cst_0 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>, #blocked> loc(#loc112)
|
| 73 |
+
%tmp14_31 = arith.extf %tmp14_30 : tensor<1024xbf16, #blocked> to tensor<1024xf32, #blocked> loc(#loc113)
|
| 74 |
+
%tmp15 = arith.subf %cst_2, %tmp14_31 : tensor<1024xf32, #blocked> loc(#loc114)
|
| 75 |
+
%tmp18 = arith.cmpi sge, %x0_11, %tmp13 : tensor<1024xi64, #blocked> loc(#loc115)
|
| 76 |
+
%tmp21 = arith.muli %ks3, %c-1_i64 : i64 loc(#loc116)
|
| 77 |
+
%tmp21_32 = tt.splat %tmp21 : i64 -> tensor<1024xi64, #blocked> loc(#loc117)
|
| 78 |
+
%tmp21_33 = arith.addi %x0_11, %tmp21_32 : tensor<1024xi64, #blocked> loc(#loc117)
|
| 79 |
+
%tmp21_34 = arith.addi %tmp21_33, %tmp14_25 : tensor<1024xi64, #blocked> loc(#loc118)
|
| 80 |
+
%tmp21_35 = arith.addi %tmp14, %tmp21_34 : tensor<1024xi64, #blocked> loc(#loc119)
|
| 81 |
+
%tmp21_36 = tt.addptr %tmp0, %tmp21_35 : tensor<1024x!tt.ptr<bf16>, #blocked>, tensor<1024xi64, #blocked> loc(#loc120)
|
| 82 |
+
%tmp21_37 = arith.andi %tmp18, %xmask_6 : tensor<1024xi1, #blocked> loc(#loc121)
|
| 83 |
+
%tmp21_38 = tt.load %tmp21_36, %tmp21_37, %cst_0 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>, #blocked> loc(#loc122)
|
| 84 |
+
%tmp21_39 = arith.extf %tmp21_38 : tensor<1024xbf16, #blocked> to tensor<1024xf32, #blocked> loc(#loc123)
|
| 85 |
+
%tmp22 = arith.select %tmp13_24, %tmp15, %tmp21_39 : tensor<1024xi1, #blocked>, tensor<1024xf32, #blocked> loc(#loc135)
|
| 86 |
+
%tmp24 = tt.splat %ks4 : i64 -> tensor<1024xi64, #blocked> loc(#loc126)
|
| 87 |
+
%tmp24_40 = arith.addi %tmp1_16, %tmp24 : tensor<1024xi64, #blocked> loc(#loc126)
|
| 88 |
+
%tmp25 = arith.select %tmp4, %tmp24_40, %tmp1_16 : tensor<1024xi1, #blocked>, tensor<1024xi64, #blocked> loc(#loc127)
|
| 89 |
+
%5 = arith.cmpi sge, %tmp25, %cst_1 : tensor<1024xi64, #blocked> loc(#loc55)
|
| 90 |
+
%6 = arith.cmpi slt, %tmp25, %tmp24 : tensor<1024xi64, #blocked> loc(#loc56)
|
| 91 |
+
%7 = arith.andi %5, %6 : tensor<1024xi1, #blocked> loc(#loc57)
|
| 92 |
+
%8 = arith.ori %7, %3 : tensor<1024xi1, #blocked> loc(#loc58)
|
| 93 |
+
tt.assert %8, "index out of bounds: 0 <= tmp25 < ks4" : tensor<1024xi1, #blocked> loc(#loc59)
|
| 94 |
+
%tmp27 = arith.muli %x0, %tmp25 : tensor<1024xi64, #blocked> loc(#loc128)
|
| 95 |
+
%tmp27_41 = arith.addi %x0_11, %tmp27 : tensor<1024xi64, #blocked> loc(#loc129)
|
| 96 |
+
%tmp27_42 = tt.splat %in_ptr3 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>, #blocked> loc(#loc130)
|
| 97 |
+
%tmp27_43 = tt.addptr %tmp27_42, %tmp27_41 : tensor<1024x!tt.ptr<bf16>, #blocked>, tensor<1024xi64, #blocked> loc(#loc130)
|
| 98 |
+
%tmp27_44 = tt.load %tmp27_43, %xmask_6 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>, #blocked> loc(#loc131)
|
| 99 |
+
%tmp27_45 = arith.extf %tmp27_44 : tensor<1024xbf16, #blocked> to tensor<1024xf32, #blocked> loc(#loc132)
|
| 100 |
+
%tmp28 = arith.mulf %tmp22, %tmp27_45 : tensor<1024xf32, #blocked> loc(#loc133)
|
| 101 |
+
%tmp29 = arith.addf %tmp8, %tmp28 : tensor<1024xf32, #blocked> loc(#loc134)
|
| 102 |
+
%9 = tt.splat %out_ptr0 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>, #blocked> loc(#loc67)
|
| 103 |
+
%10 = tt.addptr %9, %xindex_5 : tensor<1024x!tt.ptr<bf16>, #blocked>, tensor<1024xi32, #blocked> loc(#loc67)
|
| 104 |
+
%11 = arith.truncf %tmp29 : tensor<1024xf32, #blocked> to tensor<1024xbf16, #blocked> loc(#loc68)
|
| 105 |
+
tt.store %10, %11, %xmask_6 : tensor<1024x!tt.ptr<bf16>, #blocked> loc(#loc68)
|
| 106 |
+
tt.return loc(#loc69)
|
| 107 |
+
} loc(#loc)
|
| 108 |
+
} loc(#loc)
|
| 109 |
+
#loc1 = loc(unknown)
|
| 110 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":19:28)
|
| 111 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":19:33)
|
| 112 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":20:36)
|
| 113 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":20:23)
|
| 114 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":21:21)
|
| 115 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":23:21)
|
| 116 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":23:28)
|
| 117 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":24:19)
|
| 118 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":25:19)
|
| 119 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":26:30)
|
| 120 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":26:35)
|
| 121 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":26:75)
|
| 122 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":27:30)
|
| 123 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":27:35)
|
| 124 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":29:18)
|
| 125 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":30:18)
|
| 126 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":31:32)
|
| 127 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:28)
|
| 128 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:44)
|
| 129 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:37)
|
| 130 |
+
#loc22 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:54)
|
| 131 |
+
#loc23 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:52)
|
| 132 |
+
#loc24 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:62)
|
| 133 |
+
#loc25 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:39)
|
| 134 |
+
#loc26 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:35)
|
| 135 |
+
#loc27 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:30)
|
| 136 |
+
#loc28 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:46)
|
| 137 |
+
#loc29 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:86)
|
| 138 |
+
#loc30 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":34:18)
|
| 139 |
+
#loc31 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":38:31)
|
| 140 |
+
#loc32 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":38:18)
|
| 141 |
+
#loc33 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":39:19)
|
| 142 |
+
#loc34 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:35)
|
| 143 |
+
#loc35 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:41)
|
| 144 |
+
#loc36 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:54)
|
| 145 |
+
#loc37 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:31)
|
| 146 |
+
#loc38 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:68)
|
| 147 |
+
#loc39 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:60)
|
| 148 |
+
#loc40 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:119)
|
| 149 |
+
#loc41 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":41:13)
|
| 150 |
+
#loc42 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":44:20)
|
| 151 |
+
#loc43 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:52)
|
| 152 |
+
#loc44 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:47)
|
| 153 |
+
#loc45 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:60)
|
| 154 |
+
#loc46 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:41)
|
| 155 |
+
#loc47 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:31)
|
| 156 |
+
#loc48 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:81)
|
| 157 |
+
#loc49 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:73)
|
| 158 |
+
#loc50 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:132)
|
| 159 |
+
#loc51 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":48:35)
|
| 160 |
+
#loc52 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":43:35)
|
| 161 |
+
#loc53 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":50:19)
|
| 162 |
+
#loc54 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":51:34)
|
| 163 |
+
#loc55 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:28)
|
| 164 |
+
#loc56 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:46)
|
| 165 |
+
#loc57 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:38)
|
| 166 |
+
#loc58 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:54)
|
| 167 |
+
#loc59 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:64)
|
| 168 |
+
#loc60 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:40)
|
| 169 |
+
#loc61 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:36)
|
| 170 |
+
#loc62 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:31)
|
| 171 |
+
#loc63 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:48)
|
| 172 |
+
#loc64 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:88)
|
| 173 |
+
#loc65 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":54:20)
|
| 174 |
+
#loc66 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":55:19)
|
| 175 |
+
#loc67 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":56:25)
|
| 176 |
+
#loc68 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":56:37)
|
| 177 |
+
#loc69 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":56:4)
|
| 178 |
+
#loc81 = loc("xoffset"(#loc2))
|
| 179 |
+
#loc82 = loc("xoffset"(#loc3))
|
| 180 |
+
#loc83 = loc("xindex"(#loc4))
|
| 181 |
+
#loc84 = loc("xindex"(#loc5))
|
| 182 |
+
#loc85 = loc("xmask"(#loc6))
|
| 183 |
+
#loc86 = loc("x2"(#loc7))
|
| 184 |
+
#loc87 = loc("x2"(#loc8))
|
| 185 |
+
#loc88 = loc("x0"(#loc9))
|
| 186 |
+
#loc89 = loc("x5"(#loc10))
|
| 187 |
+
#loc90 = loc("tmp0"(#loc11))
|
| 188 |
+
#loc91 = loc("tmp0"(#loc12))
|
| 189 |
+
#loc92 = loc("tmp0"(#loc13))
|
| 190 |
+
#loc93 = loc("tmp1"(#loc14))
|
| 191 |
+
#loc94 = loc("tmp1"(#loc15))
|
| 192 |
+
#loc95 = loc("tmp3"(#loc16))
|
| 193 |
+
#loc96 = loc("tmp4"(#loc17))
|
| 194 |
+
#loc97 = loc("tmp5"(#loc18))
|
| 195 |
+
#loc98 = loc("tmp7"(#loc25))
|
| 196 |
+
#loc99 = loc("tmp7"(#loc26))
|
| 197 |
+
#loc100 = loc("tmp7"(#loc27))
|
| 198 |
+
#loc101 = loc("tmp7"(#loc28))
|
| 199 |
+
#loc102 = loc("tmp7"(#loc29))
|
| 200 |
+
#loc103 = loc("tmp8"(#loc30))
|
| 201 |
+
#loc104 = loc("tmp12"(#loc31))
|
| 202 |
+
#loc105 = loc("tmp12"(#loc32))
|
| 203 |
+
#loc106 = loc("tmp13"(#loc33))
|
| 204 |
+
#loc107 = loc("tmp14"(#loc34))
|
| 205 |
+
#loc108 = loc("tmp14"(#loc35))
|
| 206 |
+
#loc109 = loc("tmp14"(#loc36))
|
| 207 |
+
#loc110 = loc("tmp14"(#loc37))
|
| 208 |
+
#loc111 = loc("tmp14"(#loc38))
|
| 209 |
+
#loc112 = loc("tmp14"(#loc39))
|
| 210 |
+
#loc113 = loc("tmp14"(#loc40))
|
| 211 |
+
#loc114 = loc("tmp15"(#loc41))
|
| 212 |
+
#loc115 = loc("tmp18"(#loc42))
|
| 213 |
+
#loc116 = loc("tmp21"(#loc43))
|
| 214 |
+
#loc117 = loc("tmp21"(#loc44))
|
| 215 |
+
#loc118 = loc("tmp21"(#loc45))
|
| 216 |
+
#loc119 = loc("tmp21"(#loc46))
|
| 217 |
+
#loc120 = loc("tmp21"(#loc47))
|
| 218 |
+
#loc121 = loc("tmp21"(#loc48))
|
| 219 |
+
#loc122 = loc("tmp21"(#loc49))
|
| 220 |
+
#loc123 = loc("tmp21"(#loc50))
|
| 221 |
+
#loc124 = loc("tmp22"(#loc51))
|
| 222 |
+
#loc125 = loc("tmp17"(#loc52))
|
| 223 |
+
#loc126 = loc("tmp24"(#loc53))
|
| 224 |
+
#loc127 = loc("tmp25"(#loc54))
|
| 225 |
+
#loc128 = loc("tmp27"(#loc60))
|
| 226 |
+
#loc129 = loc("tmp27"(#loc61))
|
| 227 |
+
#loc130 = loc("tmp27"(#loc62))
|
| 228 |
+
#loc131 = loc("tmp27"(#loc63))
|
| 229 |
+
#loc132 = loc("tmp27"(#loc64))
|
| 230 |
+
#loc133 = loc("tmp28"(#loc65))
|
| 231 |
+
#loc134 = loc("tmp29"(#loc66))
|
| 232 |
+
#loc135 = loc(fused[#loc124, #loc125])
|
SpecForge-ext/cache/compiled_kernels/triton/0/2TU6ZCF6AOXLWQQED5J7FS5ZXMYK7TIOQ6T2MLB767275BROXJCA/triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0.ttir
ADDED
|
@@ -0,0 +1,231 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":18:0)
|
| 2 |
+
#loc70 = loc("in_ptr0"(#loc))
|
| 3 |
+
#loc71 = loc("in_ptr1"(#loc))
|
| 4 |
+
#loc72 = loc("in_ptr2"(#loc))
|
| 5 |
+
#loc73 = loc("in_ptr3"(#loc))
|
| 6 |
+
#loc74 = loc("out_ptr0"(#loc))
|
| 7 |
+
#loc75 = loc("ks0"(#loc))
|
| 8 |
+
#loc76 = loc("ks1"(#loc))
|
| 9 |
+
#loc77 = loc("ks2"(#loc))
|
| 10 |
+
#loc78 = loc("ks3"(#loc))
|
| 11 |
+
#loc79 = loc("ks4"(#loc))
|
| 12 |
+
#loc80 = loc("xnumel"(#loc))
|
| 13 |
+
module {
|
| 14 |
+
tt.func public @triton_poi_fused_add_cat_index_mul_neg_slice_squeeze_unsqueeze_0(%in_ptr0: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %in_ptr1: !tt.ptr<i64> {tt.divisibility = 16 : i32} loc("in_ptr1"(#loc)), %in_ptr2: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("in_ptr2"(#loc)), %in_ptr3: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("in_ptr3"(#loc)), %out_ptr0: !tt.ptr<bf16> {tt.divisibility = 16 : i32} loc("out_ptr0"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %ks2: i64 loc("ks2"(#loc)), %ks3: i64 loc("ks3"(#loc)), %ks4: i64 loc("ks4"(#loc)), %xnumel: i32 loc("xnumel"(#loc))) attributes {noinline = false} {
|
| 15 |
+
%cst = arith.constant dense<0.000000e+00> : tensor<1024xbf16> loc(#loc1)
|
| 16 |
+
%cst_0 = arith.constant dense<0> : tensor<1024xi64> loc(#loc1)
|
| 17 |
+
%cst_1 = arith.constant dense<0.000000e+00> : tensor<1024xf32> loc(#loc1)
|
| 18 |
+
%c-1_i64 = arith.constant -1 : i64 loc(#loc1)
|
| 19 |
+
%c2_i64 = arith.constant 2 : i64 loc(#loc1)
|
| 20 |
+
%cst_2 = arith.constant dense<true> : tensor<1024xi1> loc(#loc1)
|
| 21 |
+
%c1024_i32 = arith.constant 1024 : i32 loc(#loc1)
|
| 22 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc81)
|
| 23 |
+
%xoffset_3 = arith.muli %xoffset, %c1024_i32 : i32 loc(#loc82)
|
| 24 |
+
%xindex = tt.make_range {end = 1024 : i32, start = 0 : i32} : tensor<1024xi32> loc(#loc83)
|
| 25 |
+
%xindex_4 = tt.splat %xoffset_3 : i32 -> tensor<1024xi32> loc(#loc84)
|
| 26 |
+
%xindex_5 = arith.addi %xindex_4, %xindex : tensor<1024xi32> loc(#loc84)
|
| 27 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<1024xi32> loc(#loc85)
|
| 28 |
+
%xmask_6 = arith.cmpi slt, %xindex_5, %xmask : tensor<1024xi32> loc(#loc85)
|
| 29 |
+
%x2 = arith.extsi %xindex_5 : tensor<1024xi32> to tensor<1024xi64> loc(#loc86)
|
| 30 |
+
%x2_7 = tt.splat %ks0 : i64 -> tensor<1024xi64> loc(#loc86)
|
| 31 |
+
%x2_8 = arith.divsi %x2, %x2_7 : tensor<1024xi64> loc(#loc86)
|
| 32 |
+
%x2_9 = tt.splat %ks1 : i64 -> tensor<1024xi64> loc(#loc87)
|
| 33 |
+
%x2_10 = arith.remsi %x2_8, %x2_9 : tensor<1024xi64> loc(#loc87)
|
| 34 |
+
%x0 = tt.splat %ks3 : i64 -> tensor<1024xi64> loc(#loc88)
|
| 35 |
+
%x0_11 = arith.remsi %x2, %x0 : tensor<1024xi64> loc(#loc88)
|
| 36 |
+
%x5 = arith.divsi %x2, %x0 : tensor<1024xi64> loc(#loc89)
|
| 37 |
+
%tmp0 = tt.splat %in_ptr0 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>> loc(#loc90)
|
| 38 |
+
%tmp0_12 = tt.addptr %tmp0, %xindex_5 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi32> loc(#loc90)
|
| 39 |
+
%tmp0_13 = tt.load %tmp0_12, %xmask_6 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>> loc(#loc91)
|
| 40 |
+
%tmp0_14 = arith.extf %tmp0_13 : tensor<1024xbf16> to tensor<1024xf32> loc(#loc92)
|
| 41 |
+
%tmp1 = tt.splat %in_ptr1 : !tt.ptr<i64> -> tensor<1024x!tt.ptr<i64>> loc(#loc93)
|
| 42 |
+
%tmp1_15 = tt.addptr %tmp1, %x2_10 : tensor<1024x!tt.ptr<i64>>, tensor<1024xi64> loc(#loc93)
|
| 43 |
+
%tmp1_16 = tt.load %tmp1_15, %xmask_6 evictionPolicy = evict_last : tensor<1024x!tt.ptr<i64>> loc(#loc94)
|
| 44 |
+
%tmp3 = tt.splat %ks2 : i64 -> tensor<1024xi64> loc(#loc95)
|
| 45 |
+
%tmp3_17 = arith.addi %tmp1_16, %tmp3 : tensor<1024xi64> loc(#loc95)
|
| 46 |
+
%tmp4 = arith.cmpi slt, %tmp1_16, %cst_0 : tensor<1024xi64> loc(#loc96)
|
| 47 |
+
%tmp5 = arith.select %tmp4, %tmp3_17, %tmp1_16 : tensor<1024xi1>, tensor<1024xi64> loc(#loc97)
|
| 48 |
+
%0 = arith.cmpi sge, %tmp5, %cst_0 : tensor<1024xi64> loc(#loc19)
|
| 49 |
+
%1 = arith.cmpi slt, %tmp5, %tmp3 : tensor<1024xi64> loc(#loc20)
|
| 50 |
+
%2 = arith.andi %0, %1 : tensor<1024xi1> loc(#loc21)
|
| 51 |
+
%3 = arith.xori %xmask_6, %cst_2 : tensor<1024xi1> loc(#loc22)
|
| 52 |
+
%4 = arith.ori %2, %3 : tensor<1024xi1> loc(#loc23)
|
| 53 |
+
tt.assert %4, "index out of bounds: 0 <= tmp5 < ks2" : tensor<1024xi1> loc(#loc24)
|
| 54 |
+
%tmp7 = arith.muli %x0, %tmp5 : tensor<1024xi64> loc(#loc98)
|
| 55 |
+
%tmp7_18 = arith.addi %x0_11, %tmp7 : tensor<1024xi64> loc(#loc99)
|
| 56 |
+
%tmp7_19 = tt.splat %in_ptr2 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>> loc(#loc100)
|
| 57 |
+
%tmp7_20 = tt.addptr %tmp7_19, %tmp7_18 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi64> loc(#loc100)
|
| 58 |
+
%tmp7_21 = tt.load %tmp7_20, %xmask_6 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>> loc(#loc101)
|
| 59 |
+
%tmp7_22 = arith.extf %tmp7_21 : tensor<1024xbf16> to tensor<1024xf32> loc(#loc102)
|
| 60 |
+
%tmp8 = arith.mulf %tmp0_14, %tmp7_22 : tensor<1024xf32> loc(#loc103)
|
| 61 |
+
%tmp12 = arith.divsi %ks3, %c2_i64 : i64 loc(#loc104)
|
| 62 |
+
%tmp12_23 = arith.subi %ks3, %tmp12 : i64 loc(#loc105)
|
| 63 |
+
%tmp13 = tt.splat %tmp12_23 : i64 -> tensor<1024xi64> loc(#loc106)
|
| 64 |
+
%tmp13_24 = arith.cmpi slt, %x0_11, %tmp13 : tensor<1024xi64> loc(#loc106)
|
| 65 |
+
%tmp14 = arith.muli %x0, %x5 : tensor<1024xi64> loc(#loc107)
|
| 66 |
+
%tmp14_25 = tt.splat %tmp12 : i64 -> tensor<1024xi64> loc(#loc108)
|
| 67 |
+
%tmp14_26 = arith.addi %tmp14, %tmp14_25 : tensor<1024xi64> loc(#loc108)
|
| 68 |
+
%tmp14_27 = arith.addi %tmp14_26, %x0_11 : tensor<1024xi64> loc(#loc109)
|
| 69 |
+
%tmp14_28 = tt.addptr %tmp0, %tmp14_27 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi64> loc(#loc110)
|
| 70 |
+
%tmp14_29 = arith.andi %tmp13_24, %xmask_6 : tensor<1024xi1> loc(#loc111)
|
| 71 |
+
%tmp14_30 = tt.load %tmp14_28, %tmp14_29, %cst evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>> loc(#loc112)
|
| 72 |
+
%tmp14_31 = arith.extf %tmp14_30 : tensor<1024xbf16> to tensor<1024xf32> loc(#loc113)
|
| 73 |
+
%tmp15 = arith.subf %cst_1, %tmp14_31 : tensor<1024xf32> loc(#loc114)
|
| 74 |
+
%tmp18 = arith.cmpi sge, %x0_11, %tmp13 : tensor<1024xi64> loc(#loc115)
|
| 75 |
+
%tmp21 = arith.muli %ks3, %c-1_i64 : i64 loc(#loc116)
|
| 76 |
+
%tmp21_32 = tt.splat %tmp21 : i64 -> tensor<1024xi64> loc(#loc117)
|
| 77 |
+
%tmp21_33 = arith.addi %x0_11, %tmp21_32 : tensor<1024xi64> loc(#loc117)
|
| 78 |
+
%tmp21_34 = arith.addi %tmp21_33, %tmp14_25 : tensor<1024xi64> loc(#loc118)
|
| 79 |
+
%tmp21_35 = arith.addi %tmp14, %tmp21_34 : tensor<1024xi64> loc(#loc119)
|
| 80 |
+
%tmp21_36 = tt.addptr %tmp0, %tmp21_35 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi64> loc(#loc120)
|
| 81 |
+
%tmp21_37 = arith.andi %tmp18, %xmask_6 : tensor<1024xi1> loc(#loc121)
|
| 82 |
+
%tmp21_38 = tt.load %tmp21_36, %tmp21_37, %cst evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>> loc(#loc122)
|
| 83 |
+
%tmp21_39 = arith.extf %tmp21_38 : tensor<1024xbf16> to tensor<1024xf32> loc(#loc123)
|
| 84 |
+
%tmp22 = arith.select %tmp13_24, %tmp15, %tmp21_39 : tensor<1024xi1>, tensor<1024xf32> loc(#loc135)
|
| 85 |
+
%tmp24 = tt.splat %ks4 : i64 -> tensor<1024xi64> loc(#loc126)
|
| 86 |
+
%tmp24_40 = arith.addi %tmp1_16, %tmp24 : tensor<1024xi64> loc(#loc126)
|
| 87 |
+
%tmp25 = arith.select %tmp4, %tmp24_40, %tmp1_16 : tensor<1024xi1>, tensor<1024xi64> loc(#loc127)
|
| 88 |
+
%5 = arith.cmpi sge, %tmp25, %cst_0 : tensor<1024xi64> loc(#loc55)
|
| 89 |
+
%6 = arith.cmpi slt, %tmp25, %tmp24 : tensor<1024xi64> loc(#loc56)
|
| 90 |
+
%7 = arith.andi %5, %6 : tensor<1024xi1> loc(#loc57)
|
| 91 |
+
%8 = arith.ori %7, %3 : tensor<1024xi1> loc(#loc58)
|
| 92 |
+
tt.assert %8, "index out of bounds: 0 <= tmp25 < ks4" : tensor<1024xi1> loc(#loc59)
|
| 93 |
+
%tmp27 = arith.muli %x0, %tmp25 : tensor<1024xi64> loc(#loc128)
|
| 94 |
+
%tmp27_41 = arith.addi %x0_11, %tmp27 : tensor<1024xi64> loc(#loc129)
|
| 95 |
+
%tmp27_42 = tt.splat %in_ptr3 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>> loc(#loc130)
|
| 96 |
+
%tmp27_43 = tt.addptr %tmp27_42, %tmp27_41 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi64> loc(#loc130)
|
| 97 |
+
%tmp27_44 = tt.load %tmp27_43, %xmask_6 evictionPolicy = evict_last : tensor<1024x!tt.ptr<bf16>> loc(#loc131)
|
| 98 |
+
%tmp27_45 = arith.extf %tmp27_44 : tensor<1024xbf16> to tensor<1024xf32> loc(#loc132)
|
| 99 |
+
%tmp28 = arith.mulf %tmp22, %tmp27_45 : tensor<1024xf32> loc(#loc133)
|
| 100 |
+
%tmp29 = arith.addf %tmp8, %tmp28 : tensor<1024xf32> loc(#loc134)
|
| 101 |
+
%9 = tt.splat %out_ptr0 : !tt.ptr<bf16> -> tensor<1024x!tt.ptr<bf16>> loc(#loc67)
|
| 102 |
+
%10 = tt.addptr %9, %xindex_5 : tensor<1024x!tt.ptr<bf16>>, tensor<1024xi32> loc(#loc67)
|
| 103 |
+
%11 = arith.truncf %tmp29 : tensor<1024xf32> to tensor<1024xbf16> loc(#loc68)
|
| 104 |
+
tt.store %10, %11, %xmask_6 : tensor<1024x!tt.ptr<bf16>> loc(#loc68)
|
| 105 |
+
tt.return loc(#loc69)
|
| 106 |
+
} loc(#loc)
|
| 107 |
+
} loc(#loc)
|
| 108 |
+
#loc1 = loc(unknown)
|
| 109 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":19:28)
|
| 110 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":19:33)
|
| 111 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":20:36)
|
| 112 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":20:23)
|
| 113 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":21:21)
|
| 114 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":23:21)
|
| 115 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":23:28)
|
| 116 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":24:19)
|
| 117 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":25:19)
|
| 118 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":26:30)
|
| 119 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":26:35)
|
| 120 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":26:75)
|
| 121 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":27:30)
|
| 122 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":27:35)
|
| 123 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":29:18)
|
| 124 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":30:18)
|
| 125 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":31:32)
|
| 126 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:28)
|
| 127 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:44)
|
| 128 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:37)
|
| 129 |
+
#loc22 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:54)
|
| 130 |
+
#loc23 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:52)
|
| 131 |
+
#loc24 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":32:62)
|
| 132 |
+
#loc25 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:39)
|
| 133 |
+
#loc26 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:35)
|
| 134 |
+
#loc27 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:30)
|
| 135 |
+
#loc28 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:46)
|
| 136 |
+
#loc29 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":33:86)
|
| 137 |
+
#loc30 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":34:18)
|
| 138 |
+
#loc31 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":38:31)
|
| 139 |
+
#loc32 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":38:18)
|
| 140 |
+
#loc33 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":39:19)
|
| 141 |
+
#loc34 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:35)
|
| 142 |
+
#loc35 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:41)
|
| 143 |
+
#loc36 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:54)
|
| 144 |
+
#loc37 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:31)
|
| 145 |
+
#loc38 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:68)
|
| 146 |
+
#loc39 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:60)
|
| 147 |
+
#loc40 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":40:119)
|
| 148 |
+
#loc41 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":41:13)
|
| 149 |
+
#loc42 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":44:20)
|
| 150 |
+
#loc43 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:52)
|
| 151 |
+
#loc44 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:47)
|
| 152 |
+
#loc45 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:60)
|
| 153 |
+
#loc46 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:41)
|
| 154 |
+
#loc47 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:31)
|
| 155 |
+
#loc48 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:81)
|
| 156 |
+
#loc49 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:73)
|
| 157 |
+
#loc50 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":47:132)
|
| 158 |
+
#loc51 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":48:35)
|
| 159 |
+
#loc52 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":43:35)
|
| 160 |
+
#loc53 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":50:19)
|
| 161 |
+
#loc54 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":51:34)
|
| 162 |
+
#loc55 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:28)
|
| 163 |
+
#loc56 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:46)
|
| 164 |
+
#loc57 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:38)
|
| 165 |
+
#loc58 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:54)
|
| 166 |
+
#loc59 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":52:64)
|
| 167 |
+
#loc60 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:40)
|
| 168 |
+
#loc61 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:36)
|
| 169 |
+
#loc62 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:31)
|
| 170 |
+
#loc63 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:48)
|
| 171 |
+
#loc64 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":53:88)
|
| 172 |
+
#loc65 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":54:20)
|
| 173 |
+
#loc66 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":55:19)
|
| 174 |
+
#loc67 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":56:25)
|
| 175 |
+
#loc68 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":56:37)
|
| 176 |
+
#loc69 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/al/cal2r4tfyw6gic3ggqyud3nufnajx6xau2koieoitx6zg4wsiozm.py":56:4)
|
| 177 |
+
#loc81 = loc("xoffset"(#loc2))
|
| 178 |
+
#loc82 = loc("xoffset"(#loc3))
|
| 179 |
+
#loc83 = loc("xindex"(#loc4))
|
| 180 |
+
#loc84 = loc("xindex"(#loc5))
|
| 181 |
+
#loc85 = loc("xmask"(#loc6))
|
| 182 |
+
#loc86 = loc("x2"(#loc7))
|
| 183 |
+
#loc87 = loc("x2"(#loc8))
|
| 184 |
+
#loc88 = loc("x0"(#loc9))
|
| 185 |
+
#loc89 = loc("x5"(#loc10))
|
| 186 |
+
#loc90 = loc("tmp0"(#loc11))
|
| 187 |
+
#loc91 = loc("tmp0"(#loc12))
|
| 188 |
+
#loc92 = loc("tmp0"(#loc13))
|
| 189 |
+
#loc93 = loc("tmp1"(#loc14))
|
| 190 |
+
#loc94 = loc("tmp1"(#loc15))
|
| 191 |
+
#loc95 = loc("tmp3"(#loc16))
|
| 192 |
+
#loc96 = loc("tmp4"(#loc17))
|
| 193 |
+
#loc97 = loc("tmp5"(#loc18))
|
| 194 |
+
#loc98 = loc("tmp7"(#loc25))
|
| 195 |
+
#loc99 = loc("tmp7"(#loc26))
|
| 196 |
+
#loc100 = loc("tmp7"(#loc27))
|
| 197 |
+
#loc101 = loc("tmp7"(#loc28))
|
| 198 |
+
#loc102 = loc("tmp7"(#loc29))
|
| 199 |
+
#loc103 = loc("tmp8"(#loc30))
|
| 200 |
+
#loc104 = loc("tmp12"(#loc31))
|
| 201 |
+
#loc105 = loc("tmp12"(#loc32))
|
| 202 |
+
#loc106 = loc("tmp13"(#loc33))
|
| 203 |
+
#loc107 = loc("tmp14"(#loc34))
|
| 204 |
+
#loc108 = loc("tmp14"(#loc35))
|
| 205 |
+
#loc109 = loc("tmp14"(#loc36))
|
| 206 |
+
#loc110 = loc("tmp14"(#loc37))
|
| 207 |
+
#loc111 = loc("tmp14"(#loc38))
|
| 208 |
+
#loc112 = loc("tmp14"(#loc39))
|
| 209 |
+
#loc113 = loc("tmp14"(#loc40))
|
| 210 |
+
#loc114 = loc("tmp15"(#loc41))
|
| 211 |
+
#loc115 = loc("tmp18"(#loc42))
|
| 212 |
+
#loc116 = loc("tmp21"(#loc43))
|
| 213 |
+
#loc117 = loc("tmp21"(#loc44))
|
| 214 |
+
#loc118 = loc("tmp21"(#loc45))
|
| 215 |
+
#loc119 = loc("tmp21"(#loc46))
|
| 216 |
+
#loc120 = loc("tmp21"(#loc47))
|
| 217 |
+
#loc121 = loc("tmp21"(#loc48))
|
| 218 |
+
#loc122 = loc("tmp21"(#loc49))
|
| 219 |
+
#loc123 = loc("tmp21"(#loc50))
|
| 220 |
+
#loc124 = loc("tmp22"(#loc51))
|
| 221 |
+
#loc125 = loc("tmp17"(#loc52))
|
| 222 |
+
#loc126 = loc("tmp24"(#loc53))
|
| 223 |
+
#loc127 = loc("tmp25"(#loc54))
|
| 224 |
+
#loc128 = loc("tmp27"(#loc60))
|
| 225 |
+
#loc129 = loc("tmp27"(#loc61))
|
| 226 |
+
#loc130 = loc("tmp27"(#loc62))
|
| 227 |
+
#loc131 = loc("tmp27"(#loc63))
|
| 228 |
+
#loc132 = loc("tmp27"(#loc64))
|
| 229 |
+
#loc133 = loc("tmp28"(#loc65))
|
| 230 |
+
#loc134 = loc("tmp29"(#loc66))
|
| 231 |
+
#loc135 = loc(fused[#loc124, #loc125])
|
SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/__grp__triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"child_paths": {"triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.source": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.source", "triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ttir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ttir", "triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ttgir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ttgir", "triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.llir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.llir", "triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ptx": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ptx", "triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.cubin": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.cubin", "triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.json": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.json"}}
|
SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"hash": "e06ef592ad45788917e8ca2aaf880ce2d9dc1b17df1f92a71a87d94ada7fc9a8", "target": {"backend": "cuda", "arch": 90, "warp_size": 32}, "num_warps": 8, "num_ctas": 1, "num_stages": 1, "warp_size": 32, "maxnreg": null, "cluster_dims": [1, 1, 1], "ptx_version": null, "ptx_options": null, "ir_override": null, "enable_fp_fusion": true, "launch_cooperative_grid": false, "launch_pdl": false, "supported_fp8_dtypes": ["fp8e4b15", "fp8e4nv", "fp8e5"], "deprecated_fp8_dot_operand_dtypes": ["fp8e4b15"], "default_dot_input_precision": "tf32", "allowed_dot_input_precisions": ["tf32", "tf32x3", "ieee"], "max_num_imprecise_acc_default": 1073741824, "extern_libs": [["libdevice", "/workspace/specforge/lib/python3.11/site-packages/triton/backends/nvidia/lib/libdevice.10.bc"]], "debug": true, "backend_name": "cuda", "sanitize_overflow": false, "arch": "sm90", "instrumentation_mode": "", "triton_version": "3.5.1", "tensordesc_meta": [], "shared": 16384, "tmem_size": 0, "global_scratch_size": 0, "global_scratch_align": 1, "profile_scratch_size": 0, "profile_scratch_align": 1, "name": "triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2"}
|
SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.llir
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ptx
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.source
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ttgir
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/4BXPLEVNIV4ISF7IZIVK7CAM4LM5YGYX34PZFJY2Q7MUVWT7ZGUA/triton_per_fused__to_copy_arange_bitwise_and_eq_gt_index_put_lt_new_zeros_scalar_tensor_sort_sum_unsqueeze_view_where_2.ttir
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/__grp__triton_red_fused__to_copy_clone_slice_sum_transpose_5.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"child_paths": {"triton_red_fused__to_copy_clone_slice_sum_transpose_5.source": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.source", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttir", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttgir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttgir", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.llir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.llir", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.ptx": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ptx", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.cubin": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.cubin", "triton_red_fused__to_copy_clone_slice_sum_transpose_5.json": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.json"}}
|
SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.cubin
ADDED
|
Binary file (17.3 kB). View file
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"hash": "e2867fc252660bf62700f11f1d82815dae162f391a3711d09922588be2264244", "target": {"backend": "cuda", "arch": 90, "warp_size": 32}, "num_warps": 8, "num_ctas": 1, "num_stages": 1, "warp_size": 32, "maxnreg": null, "cluster_dims": [1, 1, 1], "ptx_version": null, "ptx_options": null, "ir_override": null, "enable_fp_fusion": true, "launch_cooperative_grid": false, "launch_pdl": false, "supported_fp8_dtypes": ["fp8e4b15", "fp8e4nv", "fp8e5"], "deprecated_fp8_dot_operand_dtypes": ["fp8e4b15"], "default_dot_input_precision": "tf32", "allowed_dot_input_precisions": ["tf32", "tf32x3", "ieee"], "max_num_imprecise_acc_default": 1073741824, "extern_libs": [["libdevice", "/workspace/specforge/lib/python3.11/site-packages/triton/backends/nvidia/lib/libdevice.10.bc"]], "debug": true, "backend_name": "cuda", "sanitize_overflow": false, "arch": "sm90", "instrumentation_mode": "", "triton_version": "3.5.1", "tensordesc_meta": [], "shared": 2048, "tmem_size": 0, "global_scratch_size": 0, "global_scratch_align": 1, "profile_scratch_size": 0, "profile_scratch_align": 1, "name": "triton_red_fused__to_copy_clone_slice_sum_transpose_5"}
|
SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.llir
ADDED
|
@@ -0,0 +1,205 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
; ModuleID = 'LLVMDialectModule'
|
| 2 |
+
source_filename = "LLVMDialectModule"
|
| 3 |
+
target datalayout = "e-p3:32:32-p4:32:32-p5:32:32-p6:32:32-p7:32:32-i64:64-i128:128-v16:16-v32:32-n16:32:64"
|
| 4 |
+
|
| 5 |
+
@global_smem = external addrspace(3) global [0 x i8], align 16
|
| 6 |
+
|
| 7 |
+
; Function Attrs: nounwind
|
| 8 |
+
define ptx_kernel void @triton_red_fused__to_copy_clone_slice_sum_transpose_5(ptr addrspace(1) %0, ptr addrspace(1) %1, i64 %2, i64 %3, i32 %4, i32 %5, ptr addrspace(1) readnone captures(none) %6, ptr addrspace(1) readnone captures(none) %7) local_unnamed_addr #0 !dbg !4 {
|
| 9 |
+
%9 = tail call i32 @llvm.nvvm.read.ptx.sreg.ctaid.x(), !dbg !7
|
| 10 |
+
%10 = shl i32 %9, 6, !dbg !8
|
| 11 |
+
%11 = tail call i32 @llvm.nvvm.read.ptx.sreg.tid.x(), !dbg !9
|
| 12 |
+
%12 = and i32 %11, 63, !dbg !9
|
| 13 |
+
%13 = or disjoint i32 %10, %12, !dbg !10
|
| 14 |
+
%14 = icmp slt i32 %13, %4, !dbg !11
|
| 15 |
+
%15 = lshr i32 %11, 6, !dbg !12
|
| 16 |
+
%16 = and i32 %15, 3, !dbg !12
|
| 17 |
+
%17 = sext i32 %13 to i64, !dbg !13
|
| 18 |
+
%.frozen = freeze i64 %2, !dbg !14
|
| 19 |
+
%18 = sdiv i64 %17, %.frozen, !dbg !14
|
| 20 |
+
%19 = mul i64 %18, %.frozen, !dbg !13
|
| 21 |
+
%.decomposed = sub i64 %17, %19, !dbg !13
|
| 22 |
+
%20 = icmp sgt i32 %5, 0, !dbg !15
|
| 23 |
+
br i1 %20, label %.lr.ph, label %._crit_edge, !dbg !15
|
| 24 |
+
|
| 25 |
+
.lr.ph: ; preds = %8
|
| 26 |
+
%21 = mul i64 %3, %2, !dbg !16
|
| 27 |
+
%22 = mul i64 %21, %18, !dbg !17
|
| 28 |
+
%23 = getelementptr i32, ptr addrspace(1) %0, i64 %.decomposed
|
| 29 |
+
%invariant.gep = getelementptr i32, ptr addrspace(1) %23, i64 %22, !dbg !15
|
| 30 |
+
%invariant.op = or i32 %15, 12, !dbg !15
|
| 31 |
+
%24 = insertelement <4 x i1> poison, i1 %14, i64 0, !dbg !18
|
| 32 |
+
%25 = shufflevector <4 x i1> %24, <4 x i1> poison, <4 x i32> zeroinitializer, !dbg !18
|
| 33 |
+
%26 = insertelement <4 x i32> poison, i32 %5, i64 0, !dbg !19
|
| 34 |
+
%27 = shufflevector <4 x i32> %26, <4 x i32> poison, <4 x i32> zeroinitializer, !dbg !19
|
| 35 |
+
br label %28, !dbg !15
|
| 36 |
+
|
| 37 |
+
28: ; preds = %.lr.ph, %28
|
| 38 |
+
%29 = phi i32 [ 0, %.lr.ph ], [ %67, %28 ]
|
| 39 |
+
%30 = phi <4 x i64> [ zeroinitializer, %.lr.ph ], [ %66, %28 ]
|
| 40 |
+
%31 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #5, !dbg !20
|
| 41 |
+
%32 = or disjoint i32 %29, %16, !dbg !21
|
| 42 |
+
%33 = or disjoint i32 %32, 4, !dbg !21
|
| 43 |
+
%34 = or disjoint i32 %32, 8, !dbg !21
|
| 44 |
+
%.reass = or disjoint i32 %29, %invariant.op
|
| 45 |
+
%35 = insertelement <4 x i32> poison, i32 %32, i64 0, !dbg !19
|
| 46 |
+
%36 = insertelement <4 x i32> %35, i32 %33, i64 1, !dbg !19
|
| 47 |
+
%37 = insertelement <4 x i32> %36, i32 %34, i64 2, !dbg !19
|
| 48 |
+
%38 = insertelement <4 x i32> %37, i32 %.reass, i64 3, !dbg !19
|
| 49 |
+
%39 = icmp slt <4 x i32> %38, %27, !dbg !19
|
| 50 |
+
%40 = sext i32 %32 to i64, !dbg !22
|
| 51 |
+
%41 = sext i32 %33 to i64, !dbg !22
|
| 52 |
+
%42 = sext i32 %34 to i64, !dbg !22
|
| 53 |
+
%43 = sext i32 %.reass to i64, !dbg !22
|
| 54 |
+
%44 = mul i64 %2, %40, !dbg !22
|
| 55 |
+
%45 = mul i64 %2, %41, !dbg !22
|
| 56 |
+
%46 = mul i64 %2, %42, !dbg !22
|
| 57 |
+
%47 = mul i64 %2, %43, !dbg !22
|
| 58 |
+
%gep = getelementptr i32, ptr addrspace(1) %invariant.gep, i64 %44, !dbg !23
|
| 59 |
+
%gep4 = getelementptr i32, ptr addrspace(1) %invariant.gep, i64 %45, !dbg !23
|
| 60 |
+
%gep6 = getelementptr i32, ptr addrspace(1) %invariant.gep, i64 %46, !dbg !23
|
| 61 |
+
%gep8 = getelementptr i32, ptr addrspace(1) %invariant.gep, i64 %47, !dbg !23
|
| 62 |
+
%48 = and <4 x i1> %25, %39, !dbg !18
|
| 63 |
+
%49 = extractelement <4 x i1> %48, i64 0, !dbg !20
|
| 64 |
+
%50 = tail call i32 asm sideeffect "mov.u32 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b32 { $0 }, [ $1 + 0 ], $2;", "=r,l,l,b"(ptr addrspace(1) %gep, i64 %31, i1 %49) #5, !dbg !20
|
| 65 |
+
%51 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #5, !dbg !20
|
| 66 |
+
%52 = extractelement <4 x i1> %48, i64 1, !dbg !20
|
| 67 |
+
%53 = tail call i32 asm sideeffect "mov.u32 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b32 { $0 }, [ $1 + 0 ], $2;", "=r,l,l,b"(ptr addrspace(1) %gep4, i64 %51, i1 %52) #5, !dbg !20
|
| 68 |
+
%54 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #5, !dbg !20
|
| 69 |
+
%55 = extractelement <4 x i1> %48, i64 2, !dbg !20
|
| 70 |
+
%56 = tail call i32 asm sideeffect "mov.u32 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b32 { $0 }, [ $1 + 0 ], $2;", "=r,l,l,b"(ptr addrspace(1) %gep6, i64 %54, i1 %55) #5, !dbg !20
|
| 71 |
+
%57 = tail call i64 asm sideeffect "mov.u64 $0, 0x0;\0A\09createpolicy.fractional.L2::evict_last.b64 $0, 1.0;", "=l"() #5, !dbg !20
|
| 72 |
+
%58 = extractelement <4 x i1> %48, i64 3, !dbg !20
|
| 73 |
+
%59 = tail call i32 asm sideeffect "mov.u32 $0, 0x0;\0A\09@$3 ld.global.L1::evict_last.L2::cache_hint.b32 { $0 }, [ $1 + 0 ], $2;", "=r,l,l,b"(ptr addrspace(1) %gep8, i64 %57, i1 %58) #5, !dbg !20
|
| 74 |
+
%60 = insertelement <4 x i32> poison, i32 %50, i64 0, !dbg !24
|
| 75 |
+
%61 = insertelement <4 x i32> %60, i32 %53, i64 1, !dbg !24
|
| 76 |
+
%62 = insertelement <4 x i32> %61, i32 %56, i64 2, !dbg !24
|
| 77 |
+
%63 = insertelement <4 x i32> %62, i32 %59, i64 3, !dbg !24
|
| 78 |
+
%64 = sext <4 x i32> %63 to <4 x i64>, !dbg !24
|
| 79 |
+
%65 = select <4 x i1> %48, <4 x i64> %64, <4 x i64> zeroinitializer, !dbg !25
|
| 80 |
+
%66 = add <4 x i64> %65, %30, !dbg !25
|
| 81 |
+
%67 = add i32 %29, 16, !dbg !15
|
| 82 |
+
%68 = icmp slt i32 %67, %5, !dbg !15
|
| 83 |
+
br i1 %68, label %28, label %._crit_edge.loopexit, !dbg !15
|
| 84 |
+
|
| 85 |
+
._crit_edge.loopexit: ; preds = %28
|
| 86 |
+
%69 = tail call i64 @llvm.vector.reduce.add.v4i64(<4 x i64> %66), !dbg !26
|
| 87 |
+
br label %._crit_edge, !dbg !26
|
| 88 |
+
|
| 89 |
+
._crit_edge: ; preds = %._crit_edge.loopexit, %8
|
| 90 |
+
%70 = phi i64 [ 0, %8 ], [ %69, %._crit_edge.loopexit ], !dbg !26
|
| 91 |
+
%.idx = shl nuw nsw i32 %12, 5, !dbg !30
|
| 92 |
+
%71 = getelementptr i8, ptr addrspace(3) @global_smem, i32 %.idx, !dbg !30
|
| 93 |
+
%72 = getelementptr i64, ptr addrspace(3) %71, i32 %16, !dbg !30
|
| 94 |
+
%73 = insertelement <1 x i64> poison, i64 %70, i64 0, !dbg !30
|
| 95 |
+
tail call void asm sideeffect "@$2 st.shared.b64 [ $0 + 0 ], $1;", "r,l,b"(ptr addrspace(3) %72, <1 x i64> %73, i1 true) #5, !dbg !30
|
| 96 |
+
tail call void @llvm.nvvm.barrier.cta.sync.aligned.all(i32 0), !dbg !30
|
| 97 |
+
%74 = icmp samesign ult i32 %11, 256, !dbg !30
|
| 98 |
+
%75 = getelementptr i64, ptr addrspace(3) @global_smem, i32 %11, !dbg !30
|
| 99 |
+
%76 = tail call i64 asm sideeffect "@$2 ld.shared.b64 $0, [ $1 + 0 ];", "=l,r,b"(ptr addrspace(3) %75, i1 %74) #5, !dbg !30
|
| 100 |
+
%extelt.offset = lshr i64 %76, 32, !dbg !30
|
| 101 |
+
%77 = trunc nuw i64 %extelt.offset to i32, !dbg !30
|
| 102 |
+
%78 = trunc i64 %76 to i32, !dbg !30
|
| 103 |
+
%79 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %78, i32 2, i32 31), !dbg !30
|
| 104 |
+
%80 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %77, i32 2, i32 31), !dbg !30
|
| 105 |
+
%81 = insertelement <2 x i32> poison, i32 %79, i64 0, !dbg !30
|
| 106 |
+
%82 = insertelement <2 x i32> %81, i32 %80, i64 1, !dbg !30
|
| 107 |
+
%83 = bitcast <2 x i32> %82 to i64, !dbg !30
|
| 108 |
+
%84 = add i64 %76, %83, !dbg !26
|
| 109 |
+
%extelt.offset2 = lshr i64 %84, 32, !dbg !30
|
| 110 |
+
%85 = trunc nuw i64 %extelt.offset2 to i32, !dbg !30
|
| 111 |
+
%86 = trunc i64 %84 to i32, !dbg !30
|
| 112 |
+
%87 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %86, i32 1, i32 31), !dbg !30
|
| 113 |
+
%88 = tail call i32 @llvm.nvvm.shfl.sync.bfly.i32(i32 -1, i32 %85, i32 1, i32 31), !dbg !30
|
| 114 |
+
%89 = insertelement <2 x i32> poison, i32 %87, i64 0, !dbg !30
|
| 115 |
+
%90 = insertelement <2 x i32> %89, i32 %88, i64 1, !dbg !30
|
| 116 |
+
%91 = bitcast <2 x i32> %90 to i64, !dbg !30
|
| 117 |
+
%92 = add i64 %84, %91, !dbg !26
|
| 118 |
+
%93 = and i32 %11, 771, !dbg !30
|
| 119 |
+
%94 = icmp eq i32 %93, 0, !dbg !30
|
| 120 |
+
%95 = insertelement <1 x i64> poison, i64 %92, i64 0, !dbg !30
|
| 121 |
+
tail call void asm sideeffect "@$2 st.shared.b64 [ $0 + 0 ], $1;", "r,l,b"(ptr addrspace(3) %75, <1 x i64> %95, i1 %94) #5, !dbg !30
|
| 122 |
+
tail call void @llvm.nvvm.barrier.cta.sync.aligned.all(i32 0), !dbg !30
|
| 123 |
+
%96 = load i64, ptr addrspace(3) %71, align 16, !dbg !30
|
| 124 |
+
%97 = trunc i64 %96 to i32, !dbg !31
|
| 125 |
+
%98 = icmp slt i64 %2, 2, !dbg !32
|
| 126 |
+
%99 = icmp sgt i64 %2, 1, !dbg !33
|
| 127 |
+
%100 = select i1 %99, i64 %2, i64 0, !dbg !34
|
| 128 |
+
%101 = zext i1 %98 to i64, !dbg !35
|
| 129 |
+
%102 = add i64 %100, %101, !dbg !36
|
| 130 |
+
%103 = mul i64 %18, %102, !dbg !37
|
| 131 |
+
%104 = getelementptr i32, ptr addrspace(1) %1, i64 %.decomposed, !dbg !38
|
| 132 |
+
%105 = getelementptr i32, ptr addrspace(1) %104, i64 %103, !dbg !38
|
| 133 |
+
%106 = and i32 %11, 192, !dbg !39
|
| 134 |
+
%107 = icmp eq i32 %106, 0, !dbg !39
|
| 135 |
+
%108 = and i1 %107, %14, !dbg !39
|
| 136 |
+
tail call void asm sideeffect "@$2 st.global.b32 [ $1 + 0 ], { $0 };", "r,l,b"(i32 %97, ptr addrspace(1) %105, i1 %108) #5, !dbg !39
|
| 137 |
+
ret void, !dbg !40
|
| 138 |
+
}
|
| 139 |
+
|
| 140 |
+
; Function Attrs: mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none)
|
| 141 |
+
declare noundef range(i32 0, 2147483647) i32 @llvm.nvvm.read.ptx.sreg.ctaid.x() #1
|
| 142 |
+
|
| 143 |
+
; Function Attrs: mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none)
|
| 144 |
+
declare noundef range(i32 0, 1024) i32 @llvm.nvvm.read.ptx.sreg.tid.x() #1
|
| 145 |
+
|
| 146 |
+
; Function Attrs: convergent nocallback nounwind
|
| 147 |
+
declare void @llvm.nvvm.barrier.cta.sync.aligned.all(i32) #2
|
| 148 |
+
|
| 149 |
+
; Function Attrs: convergent nocallback nounwind memory(inaccessiblemem: readwrite)
|
| 150 |
+
declare i32 @llvm.nvvm.shfl.sync.bfly.i32(i32, i32, i32, i32) #3
|
| 151 |
+
|
| 152 |
+
; Function Attrs: nocallback nofree nosync nounwind speculatable willreturn memory(none)
|
| 153 |
+
declare i64 @llvm.vector.reduce.add.v4i64(<4 x i64>) #4
|
| 154 |
+
|
| 155 |
+
attributes #0 = { nounwind "nvvm.reqntid"="256" }
|
| 156 |
+
attributes #1 = { mustprogress nocallback nofree nosync nounwind speculatable willreturn memory(none) }
|
| 157 |
+
attributes #2 = { convergent nocallback nounwind }
|
| 158 |
+
attributes #3 = { convergent nocallback nounwind memory(inaccessiblemem: readwrite) }
|
| 159 |
+
attributes #4 = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }
|
| 160 |
+
attributes #5 = { nounwind }
|
| 161 |
+
|
| 162 |
+
!llvm.dbg.cu = !{!0}
|
| 163 |
+
!llvm.module.flags = !{!2, !3}
|
| 164 |
+
|
| 165 |
+
!0 = distinct !DICompileUnit(language: DW_LANG_C, file: !1, producer: "triton", isOptimized: true, runtimeVersion: 0, emissionKind: LineTablesOnly)
|
| 166 |
+
!1 = !DIFile(filename: "cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py", directory: "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh")
|
| 167 |
+
!2 = !{i32 2, !"Debug Info Version", i32 3}
|
| 168 |
+
!3 = !{i32 4, !"nvvm-reflect-ftz", i32 1}
|
| 169 |
+
!4 = distinct !DISubprogram(name: "triton_red_fused__to_copy_clone_slice_sum_transpose_5", linkageName: "triton_red_fused__to_copy_clone_slice_sum_transpose_5", scope: !1, file: !1, line: 18, type: !5, scopeLine: 18, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
|
| 170 |
+
!5 = !DISubroutineType(cc: DW_CC_normal, types: !6)
|
| 171 |
+
!6 = !{}
|
| 172 |
+
!7 = !DILocation(line: 21, column: 28, scope: !4)
|
| 173 |
+
!8 = !DILocation(line: 21, column: 33, scope: !4)
|
| 174 |
+
!9 = !DILocation(line: 22, column: 44, scope: !4)
|
| 175 |
+
!10 = !DILocation(line: 22, column: 23, scope: !4)
|
| 176 |
+
!11 = !DILocation(line: 23, column: 21, scope: !4)
|
| 177 |
+
!12 = !DILocation(line: 24, column: 37, scope: !4)
|
| 178 |
+
!13 = !DILocation(line: 26, column: 19, scope: !4)
|
| 179 |
+
!14 = !DILocation(line: 27, column: 19, scope: !4)
|
| 180 |
+
!15 = !DILocation(line: 30, column: 40, scope: !4)
|
| 181 |
+
!16 = !DILocation(line: 36, column: 54, scope: !4)
|
| 182 |
+
!17 = !DILocation(line: 36, column: 58, scope: !4)
|
| 183 |
+
!18 = !DILocation(line: 36, column: 73, scope: !4)
|
| 184 |
+
!19 = !DILocation(line: 32, column: 29, scope: !4)
|
| 185 |
+
!20 = !DILocation(line: 36, column: 63, scope: !4)
|
| 186 |
+
!21 = !DILocation(line: 31, column: 31, scope: !4)
|
| 187 |
+
!22 = !DILocation(line: 36, column: 43, scope: !4)
|
| 188 |
+
!23 = !DILocation(line: 36, column: 34, scope: !4)
|
| 189 |
+
!24 = !DILocation(line: 37, column: 23, scope: !4)
|
| 190 |
+
!25 = !DILocation(line: 40, column: 48, scope: !4)
|
| 191 |
+
!26 = !DILocation(line: 261, column: 15, scope: !27, inlinedAt: !29)
|
| 192 |
+
!27 = distinct !DILexicalBlockFile(scope: !4, file: !28, discriminator: 0)
|
| 193 |
+
!28 = !DIFile(filename: "standard.py", directory: "/workspace/specforge/lib/python3.11/site-packages/triton/language")
|
| 194 |
+
!29 = !DILocation(line: 41, column: 25, scope: !4)
|
| 195 |
+
!30 = !DILocation(line: 291, column: 36, scope: !27, inlinedAt: !29)
|
| 196 |
+
!31 = !DILocation(line: 42, column: 19, scope: !4)
|
| 197 |
+
!32 = !DILocation(line: 43, column: 49, scope: !4)
|
| 198 |
+
!33 = !DILocation(line: 43, column: 75, scope: !4)
|
| 199 |
+
!34 = !DILocation(line: 43, column: 66, scope: !4)
|
| 200 |
+
!35 = !DILocation(line: 43, scope: !4)
|
| 201 |
+
!36 = !DILocation(line: 43, column: 57, scope: !4)
|
| 202 |
+
!37 = !DILocation(line: 43, column: 34, scope: !4)
|
| 203 |
+
!38 = !DILocation(line: 43, column: 25, scope: !4)
|
| 204 |
+
!39 = !DILocation(line: 43, column: 88, scope: !4)
|
| 205 |
+
!40 = !DILocation(line: 43, column: 4, scope: !4)
|
SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ptx
ADDED
|
@@ -0,0 +1,527 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
//
|
| 2 |
+
// Generated by LLVM NVPTX Back-End
|
| 3 |
+
//
|
| 4 |
+
|
| 5 |
+
.version 8.7
|
| 6 |
+
.target sm_90a
|
| 7 |
+
.address_size 64
|
| 8 |
+
|
| 9 |
+
// .globl triton_red_fused__to_copy_clone_slice_sum_transpose_5 // -- Begin function triton_red_fused__to_copy_clone_slice_sum_transpose_5
|
| 10 |
+
.extern .shared .align 16 .b8 global_smem[];
|
| 11 |
+
// @triton_red_fused__to_copy_clone_slice_sum_transpose_5
|
| 12 |
+
.visible .entry triton_red_fused__to_copy_clone_slice_sum_transpose_5(
|
| 13 |
+
.param .u64 .ptr .global .align 1 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_0,
|
| 14 |
+
.param .u64 .ptr .global .align 1 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_1,
|
| 15 |
+
.param .u64 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_2,
|
| 16 |
+
.param .u64 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_3,
|
| 17 |
+
.param .u32 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_4,
|
| 18 |
+
.param .u32 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_5,
|
| 19 |
+
.param .u64 .ptr .global .align 1 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_6,
|
| 20 |
+
.param .u64 .ptr .global .align 1 triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_7
|
| 21 |
+
)
|
| 22 |
+
.reqntid 256
|
| 23 |
+
{
|
| 24 |
+
.reg .pred %p<24>;
|
| 25 |
+
.reg .b32 %r<53>;
|
| 26 |
+
.reg .b64 %rd<97>;
|
| 27 |
+
.loc 1 18 0 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:18:0
|
| 28 |
+
$L__func_begin0:
|
| 29 |
+
.loc 1 18 0 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:18:0
|
| 30 |
+
|
| 31 |
+
// %bb.0:
|
| 32 |
+
ld.param.b32 %r13, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_5];
|
| 33 |
+
ld.param.b64 %rd20, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_2];
|
| 34 |
+
$L__tmp0:
|
| 35 |
+
.loc 1 21 28 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:21:28
|
| 36 |
+
mov.u32 %r14, %ctaid.x;
|
| 37 |
+
.loc 1 21 33 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:21:33
|
| 38 |
+
shl.b32 %r15, %r14, 6;
|
| 39 |
+
.loc 1 22 44 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:22:44
|
| 40 |
+
mov.u32 %r1, %tid.x;
|
| 41 |
+
and.b32 %r2, %r1, 63;
|
| 42 |
+
.loc 1 22 23 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:22:23
|
| 43 |
+
or.b32 %r16, %r15, %r2;
|
| 44 |
+
.loc 1 26 19 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:26:19
|
| 45 |
+
cvt.s64.s32 %rd1, %r16;
|
| 46 |
+
.loc 1 27 19 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:27:19
|
| 47 |
+
or.b64 %rd23, %rd1, %rd20;
|
| 48 |
+
and.b64 %rd24, %rd23, -4294967296;
|
| 49 |
+
setp.ne.b64 %p5, %rd24, 0;
|
| 50 |
+
cvt.u32.u64 %r51, %rd1;
|
| 51 |
+
@%p5 bra $L__BB0_2;
|
| 52 |
+
bra.uni $L__BB0_1;
|
| 53 |
+
$L__BB0_2:
|
| 54 |
+
div.s64 %rd91, %rd1, %rd20;
|
| 55 |
+
bra.uni $L__BB0_3;
|
| 56 |
+
$L__BB0_1:
|
| 57 |
+
cvt.u32.u64 %r17, %rd20;
|
| 58 |
+
div.u32 %r19, %r51, %r17;
|
| 59 |
+
cvt.u64.u32 %rd91, %r19;
|
| 60 |
+
$L__BB0_3:
|
| 61 |
+
.loc 1 0 19 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:0:19
|
| 62 |
+
ld.param.b32 %r12, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_4];
|
| 63 |
+
ld.param.b64 %rd19, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_1];
|
| 64 |
+
bfe.u32 %r4, %r1, 6, 2;
|
| 65 |
+
.loc 1 26 19 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:26:19
|
| 66 |
+
mul.lo.s64 %rd26, %rd91, %rd20;
|
| 67 |
+
sub.s64 %rd6, %rd1, %rd26;
|
| 68 |
+
.loc 1 30 40 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:30:40
|
| 69 |
+
setp.lt.s32 %p6, %r13, 1;
|
| 70 |
+
mov.b64 %rd96, 0;
|
| 71 |
+
shl.b64 %rd90, %rd6, 2;
|
| 72 |
+
@%p6 bra $L__BB0_7;
|
| 73 |
+
// %bb.4: // %.lr.ph
|
| 74 |
+
.loc 1 0 40 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:0:40
|
| 75 |
+
ld.param.b64 %rd21, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_3];
|
| 76 |
+
ld.param.b64 %rd18, [triton_red_fused__to_copy_clone_slice_sum_transpose_5_param_0];
|
| 77 |
+
shr.u32 %r3, %r1, 6;
|
| 78 |
+
.loc 1 23 21 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:23:21
|
| 79 |
+
setp.lt.s32 %p1, %r51, %r12;
|
| 80 |
+
.loc 1 36 54 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:36:54
|
| 81 |
+
mul.lo.s64 %rd31, %rd21, %rd20;
|
| 82 |
+
.loc 1 36 58 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:36:58
|
| 83 |
+
mul.lo.s64 %rd32, %rd31, %rd91;
|
| 84 |
+
add.s64 %rd34, %rd18, %rd90;
|
| 85 |
+
.loc 1 30 40 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:30:40
|
| 86 |
+
shl.b64 %rd35, %rd32, 2;
|
| 87 |
+
add.s64 %rd7, %rd34, %rd35;
|
| 88 |
+
or.b32 %r5, %r3, 12;
|
| 89 |
+
mov.b64 %rd92, 0;
|
| 90 |
+
mov.b32 %r52, 0;
|
| 91 |
+
mov.b64 %rd93, %rd92;
|
| 92 |
+
mov.b64 %rd94, %rd92;
|
| 93 |
+
mov.b64 %rd95, %rd92;
|
| 94 |
+
$L__BB0_5: // =>This Inner Loop Header: Depth=1
|
| 95 |
+
.loc 1 36 63 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:36:63
|
| 96 |
+
// begin inline asm
|
| 97 |
+
mov.u64 %rd36, 0x0;
|
| 98 |
+
createpolicy.fractional.L2::evict_last.b64 %rd36, 1.0;
|
| 99 |
+
// end inline asm
|
| 100 |
+
.loc 1 31 31 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:31:31
|
| 101 |
+
add.s32 %r26, %r4, %r52;
|
| 102 |
+
add.s32 %r27, %r26, 4;
|
| 103 |
+
add.s32 %r28, %r26, 8;
|
| 104 |
+
.loc 1 32 29 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:32:29
|
| 105 |
+
add.s32 %r29, %r5, %r52;
|
| 106 |
+
setp.lt.s32 %p11, %r26, %r13;
|
| 107 |
+
setp.lt.s32 %p12, %r27, %r13;
|
| 108 |
+
setp.lt.s32 %p13, %r28, %r13;
|
| 109 |
+
setp.lt.s32 %p14, %r29, %r13;
|
| 110 |
+
.loc 1 36 43 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:36:43
|
| 111 |
+
cvt.s64.s32 %rd48, %r26;
|
| 112 |
+
cvt.s64.s32 %rd49, %r27;
|
| 113 |
+
cvt.s64.s32 %rd50, %r28;
|
| 114 |
+
cvt.s64.s32 %rd51, %r29;
|
| 115 |
+
mul.lo.s64 %rd52, %rd20, %rd48;
|
| 116 |
+
mul.lo.s64 %rd53, %rd20, %rd49;
|
| 117 |
+
mul.lo.s64 %rd54, %rd20, %rd50;
|
| 118 |
+
mul.lo.s64 %rd55, %rd20, %rd51;
|
| 119 |
+
.loc 1 36 34 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:36:34
|
| 120 |
+
shl.b64 %rd56, %rd52, 2;
|
| 121 |
+
add.s64 %rd37, %rd7, %rd56;
|
| 122 |
+
shl.b64 %rd57, %rd53, 2;
|
| 123 |
+
add.s64 %rd40, %rd7, %rd57;
|
| 124 |
+
shl.b64 %rd58, %rd54, 2;
|
| 125 |
+
add.s64 %rd43, %rd7, %rd58;
|
| 126 |
+
shl.b64 %rd59, %rd55, 2;
|
| 127 |
+
add.s64 %rd46, %rd7, %rd59;
|
| 128 |
+
.loc 1 36 73 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:36:73
|
| 129 |
+
and.pred %p10, %p1, %p14;
|
| 130 |
+
and.pred %p9, %p1, %p13;
|
| 131 |
+
and.pred %p8, %p1, %p12;
|
| 132 |
+
and.pred %p7, %p1, %p11;
|
| 133 |
+
.loc 1 36 63 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:36:63
|
| 134 |
+
// begin inline asm
|
| 135 |
+
mov.u32 %r22, 0x0;
|
| 136 |
+
@%p7 ld.global.L1::evict_last.L2::cache_hint.b32 { %r22 }, [ %rd37 + 0 ], %rd36;
|
| 137 |
+
// end inline asm
|
| 138 |
+
// begin inline asm
|
| 139 |
+
mov.u64 %rd39, 0x0;
|
| 140 |
+
createpolicy.fractional.L2::evict_last.b64 %rd39, 1.0;
|
| 141 |
+
// end inline asm
|
| 142 |
+
// begin inline asm
|
| 143 |
+
mov.u32 %r23, 0x0;
|
| 144 |
+
@%p8 ld.global.L1::evict_last.L2::cache_hint.b32 { %r23 }, [ %rd40 + 0 ], %rd39;
|
| 145 |
+
// end inline asm
|
| 146 |
+
// begin inline asm
|
| 147 |
+
mov.u64 %rd42, 0x0;
|
| 148 |
+
createpolicy.fractional.L2::evict_last.b64 %rd42, 1.0;
|
| 149 |
+
// end inline asm
|
| 150 |
+
// begin inline asm
|
| 151 |
+
mov.u32 %r24, 0x0;
|
| 152 |
+
@%p9 ld.global.L1::evict_last.L2::cache_hint.b32 { %r24 }, [ %rd43 + 0 ], %rd42;
|
| 153 |
+
// end inline asm
|
| 154 |
+
// begin inline asm
|
| 155 |
+
mov.u64 %rd45, 0x0;
|
| 156 |
+
createpolicy.fractional.L2::evict_last.b64 %rd45, 1.0;
|
| 157 |
+
// end inline asm
|
| 158 |
+
// begin inline asm
|
| 159 |
+
mov.u32 %r25, 0x0;
|
| 160 |
+
@%p10 ld.global.L1::evict_last.L2::cache_hint.b32 { %r25 }, [ %rd46 + 0 ], %rd45;
|
| 161 |
+
// end inline asm
|
| 162 |
+
.loc 1 37 23 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:37:23
|
| 163 |
+
cvt.s64.s32 %rd60, %r22;
|
| 164 |
+
cvt.s64.s32 %rd61, %r23;
|
| 165 |
+
cvt.s64.s32 %rd62, %r24;
|
| 166 |
+
cvt.s64.s32 %rd63, %r25;
|
| 167 |
+
.loc 1 40 48 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:40:48
|
| 168 |
+
selp.b64 %rd64, %rd63, 0, %p10;
|
| 169 |
+
selp.b64 %rd65, %rd62, 0, %p9;
|
| 170 |
+
selp.b64 %rd66, %rd61, 0, %p8;
|
| 171 |
+
selp.b64 %rd67, %rd60, 0, %p7;
|
| 172 |
+
add.s64 %rd92, %rd67, %rd92;
|
| 173 |
+
add.s64 %rd93, %rd66, %rd93;
|
| 174 |
+
add.s64 %rd94, %rd65, %rd94;
|
| 175 |
+
add.s64 %rd95, %rd64, %rd95;
|
| 176 |
+
.loc 1 30 40 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:30:40
|
| 177 |
+
add.s32 %r52, %r52, 16;
|
| 178 |
+
setp.lt.s32 %p15, %r52, %r13;
|
| 179 |
+
@%p15 bra $L__BB0_5;
|
| 180 |
+
// %bb.6: // %._crit_edge.loopexit
|
| 181 |
+
$L__tmp1:
|
| 182 |
+
.loc 2 261 15 // standard.py:261:15 @[ cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:41:25 ]
|
| 183 |
+
add.s64 %rd68, %rd92, %rd94;
|
| 184 |
+
add.s64 %rd69, %rd93, %rd95;
|
| 185 |
+
add.s64 %rd96, %rd68, %rd69;
|
| 186 |
+
$L__tmp2:
|
| 187 |
+
$L__BB0_7: // %._crit_edge
|
| 188 |
+
.loc 1 23 21 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:23:21
|
| 189 |
+
setp.lt.s32 %p20, %r51, %r12;
|
| 190 |
+
$L__tmp3:
|
| 191 |
+
.loc 2 291 36 // standard.py:291:36 @[ cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:41:25 ]
|
| 192 |
+
shl.b32 %r35, %r2, 5;
|
| 193 |
+
mov.b32 %r36, global_smem;
|
| 194 |
+
add.s32 %r37, %r36, %r35;
|
| 195 |
+
shl.b32 %r38, %r4, 3;
|
| 196 |
+
add.s32 %r30, %r37, %r38;
|
| 197 |
+
mov.pred %p16, -1;
|
| 198 |
+
// begin inline asm
|
| 199 |
+
@%p16 st.shared.b64 [ %r30 + 0 ], %rd96;
|
| 200 |
+
// end inline asm
|
| 201 |
+
bar.sync 0;
|
| 202 |
+
setp.lt.u32 %p17, %r1, 256;
|
| 203 |
+
shl.b32 %r39, %r1, 3;
|
| 204 |
+
add.s32 %r31, %r36, %r39;
|
| 205 |
+
// begin inline asm
|
| 206 |
+
@%p17 ld.shared.b64 %rd71, [ %r31 + 0 ];
|
| 207 |
+
// end inline asm
|
| 208 |
+
mov.b64 {_, %r40}, %rd71;
|
| 209 |
+
cvt.u32.u64 %r41, %rd71;
|
| 210 |
+
shfl.sync.bfly.b32 %r42, %r41, 2, 31, -1;
|
| 211 |
+
shfl.sync.bfly.b32 %r43, %r40, 2, 31, -1;
|
| 212 |
+
cvt.u64.u32 %rd74, %r42;
|
| 213 |
+
cvt.u64.u32 %rd75, %r43;
|
| 214 |
+
shl.b64 %rd76, %rd75, 32;
|
| 215 |
+
or.b64 %rd77, %rd74, %rd76;
|
| 216 |
+
.loc 2 261 15 // standard.py:261:15 @[ cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:41:25 ]
|
| 217 |
+
add.s64 %rd78, %rd71, %rd77;
|
| 218 |
+
.loc 2 291 36 // standard.py:291:36 @[ cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:41:25 ]
|
| 219 |
+
mov.b64 {_, %r44}, %rd78;
|
| 220 |
+
cvt.u32.u64 %r45, %rd78;
|
| 221 |
+
shfl.sync.bfly.b32 %r46, %r45, 1, 31, -1;
|
| 222 |
+
shfl.sync.bfly.b32 %r47, %r44, 1, 31, -1;
|
| 223 |
+
cvt.u64.u32 %rd79, %r46;
|
| 224 |
+
cvt.u64.u32 %rd80, %r47;
|
| 225 |
+
shl.b64 %rd81, %rd80, 32;
|
| 226 |
+
or.b64 %rd82, %rd79, %rd81;
|
| 227 |
+
.loc 2 261 15 // standard.py:261:15 @[ cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:41:25 ]
|
| 228 |
+
add.s64 %rd72, %rd78, %rd82;
|
| 229 |
+
.loc 2 291 36 // standard.py:291:36 @[ cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:41:25 ]
|
| 230 |
+
and.b32 %r48, %r1, 771;
|
| 231 |
+
setp.eq.b32 %p18, %r48, 0;
|
| 232 |
+
// begin inline asm
|
| 233 |
+
@%p18 st.shared.b64 [ %r31 + 0 ], %rd72;
|
| 234 |
+
// end inline asm
|
| 235 |
+
bar.sync 0;
|
| 236 |
+
ld.shared.b32 %r33, [%r37];
|
| 237 |
+
$L__tmp4:
|
| 238 |
+
.loc 1 43 49 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:43:49
|
| 239 |
+
setp.lt.s64 %p21, %rd20, 2;
|
| 240 |
+
.loc 1 43 75 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:43:75
|
| 241 |
+
setp.gt.s64 %p22, %rd20, 1;
|
| 242 |
+
.loc 1 43 66 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:43:66
|
| 243 |
+
selp.b64 %rd83, %rd20, 0, %p22;
|
| 244 |
+
.loc 1 43 0 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:43
|
| 245 |
+
selp.b64 %rd84, 1, 0, %p21;
|
| 246 |
+
.loc 1 43 57 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:43:57
|
| 247 |
+
add.s64 %rd85, %rd83, %rd84;
|
| 248 |
+
.loc 1 43 34 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:43:34
|
| 249 |
+
mul.lo.s64 %rd86, %rd91, %rd85;
|
| 250 |
+
.loc 1 43 25 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:43:25
|
| 251 |
+
add.s64 %rd88, %rd19, %rd90;
|
| 252 |
+
shl.b64 %rd89, %rd86, 2;
|
| 253 |
+
add.s64 %rd73, %rd88, %rd89;
|
| 254 |
+
.loc 1 43 88 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:43:88
|
| 255 |
+
and.b32 %r49, %r1, 192;
|
| 256 |
+
setp.eq.b32 %p23, %r49, 0;
|
| 257 |
+
and.pred %p19, %p23, %p20;
|
| 258 |
+
// begin inline asm
|
| 259 |
+
@%p19 st.global.b32 [ %rd73 + 0 ], { %r33 };
|
| 260 |
+
// end inline asm
|
| 261 |
+
.loc 1 43 4 // cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py:43:4
|
| 262 |
+
ret;
|
| 263 |
+
$L__tmp5:
|
| 264 |
+
$L__func_end0:
|
| 265 |
+
// -- End function
|
| 266 |
+
}
|
| 267 |
+
.file 1 "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py"
|
| 268 |
+
.file 2 "/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py"
|
| 269 |
+
.section .debug_abbrev
|
| 270 |
+
{
|
| 271 |
+
.b8 1 // Abbreviation Code
|
| 272 |
+
.b8 17 // DW_TAG_compile_unit
|
| 273 |
+
.b8 1 // DW_CHILDREN_yes
|
| 274 |
+
.b8 37 // DW_AT_producer
|
| 275 |
+
.b8 8 // DW_FORM_string
|
| 276 |
+
.b8 19 // DW_AT_language
|
| 277 |
+
.b8 5 // DW_FORM_data2
|
| 278 |
+
.b8 3 // DW_AT_name
|
| 279 |
+
.b8 8 // DW_FORM_string
|
| 280 |
+
.b8 16 // DW_AT_stmt_list
|
| 281 |
+
.b8 6 // DW_FORM_data4
|
| 282 |
+
.b8 27 // DW_AT_comp_dir
|
| 283 |
+
.b8 8 // DW_FORM_string
|
| 284 |
+
.b8 0 // EOM(1)
|
| 285 |
+
.b8 0 // EOM(2)
|
| 286 |
+
.b8 2 // Abbreviation Code
|
| 287 |
+
.b8 46 // DW_TAG_subprogram
|
| 288 |
+
.b8 0 // DW_CHILDREN_no
|
| 289 |
+
.b8 3 // DW_AT_name
|
| 290 |
+
.b8 8 // DW_FORM_string
|
| 291 |
+
.b8 32 // DW_AT_inline
|
| 292 |
+
.b8 11 // DW_FORM_data1
|
| 293 |
+
.b8 0 // EOM(1)
|
| 294 |
+
.b8 0 // EOM(2)
|
| 295 |
+
.b8 3 // Abbreviation Code
|
| 296 |
+
.b8 46 // DW_TAG_subprogram
|
| 297 |
+
.b8 1 // DW_CHILDREN_yes
|
| 298 |
+
.b8 17 // DW_AT_low_pc
|
| 299 |
+
.b8 1 // DW_FORM_addr
|
| 300 |
+
.b8 18 // DW_AT_high_pc
|
| 301 |
+
.b8 1 // DW_FORM_addr
|
| 302 |
+
.b8 49 // DW_AT_abstract_origin
|
| 303 |
+
.b8 19 // DW_FORM_ref4
|
| 304 |
+
.b8 0 // EOM(1)
|
| 305 |
+
.b8 0 // EOM(2)
|
| 306 |
+
.b8 4 // Abbreviation Code
|
| 307 |
+
.b8 29 // DW_TAG_inlined_subroutine
|
| 308 |
+
.b8 0 // DW_CHILDREN_no
|
| 309 |
+
.b8 49 // DW_AT_abstract_origin
|
| 310 |
+
.b8 19 // DW_FORM_ref4
|
| 311 |
+
.b8 17 // DW_AT_low_pc
|
| 312 |
+
.b8 1 // DW_FORM_addr
|
| 313 |
+
.b8 18 // DW_AT_high_pc
|
| 314 |
+
.b8 1 // DW_FORM_addr
|
| 315 |
+
.b8 88 // DW_AT_call_file
|
| 316 |
+
.b8 11 // DW_FORM_data1
|
| 317 |
+
.b8 89 // DW_AT_call_line
|
| 318 |
+
.b8 11 // DW_FORM_data1
|
| 319 |
+
.b8 87 // DW_AT_call_column
|
| 320 |
+
.b8 11 // DW_FORM_data1
|
| 321 |
+
.b8 0 // EOM(1)
|
| 322 |
+
.b8 0 // EOM(2)
|
| 323 |
+
.b8 0 // EOM(3)
|
| 324 |
+
}
|
| 325 |
+
.section .debug_info
|
| 326 |
+
{
|
| 327 |
+
.b32 238 // Length of Unit
|
| 328 |
+
.b8 2 // DWARF version number
|
| 329 |
+
.b8 0
|
| 330 |
+
.b32 .debug_abbrev // Offset Into Abbrev. Section
|
| 331 |
+
.b8 8 // Address Size (in bytes)
|
| 332 |
+
.b8 1 // Abbrev [1] 0xb:0xe7 DW_TAG_compile_unit
|
| 333 |
+
.b8 116 // DW_AT_producer
|
| 334 |
+
.b8 114
|
| 335 |
+
.b8 105
|
| 336 |
+
.b8 116
|
| 337 |
+
.b8 111
|
| 338 |
+
.b8 110
|
| 339 |
+
.b8 0
|
| 340 |
+
.b8 2 // DW_AT_language
|
| 341 |
+
.b8 0
|
| 342 |
+
.b8 99 // DW_AT_name
|
| 343 |
+
.b8 119
|
| 344 |
+
.b8 104
|
| 345 |
+
.b8 103
|
| 346 |
+
.b8 117
|
| 347 |
+
.b8 50
|
| 348 |
+
.b8 98
|
| 349 |
+
.b8 122
|
| 350 |
+
.b8 102
|
| 351 |
+
.b8 107
|
| 352 |
+
.b8 115
|
| 353 |
+
.b8 101
|
| 354 |
+
.b8 99
|
| 355 |
+
.b8 98
|
| 356 |
+
.b8 109
|
| 357 |
+
.b8 52
|
| 358 |
+
.b8 105
|
| 359 |
+
.b8 110
|
| 360 |
+
.b8 116
|
| 361 |
+
.b8 51
|
| 362 |
+
.b8 111
|
| 363 |
+
.b8 107
|
| 364 |
+
.b8 107
|
| 365 |
+
.b8 104
|
| 366 |
+
.b8 51
|
| 367 |
+
.b8 112
|
| 368 |
+
.b8 104
|
| 369 |
+
.b8 53
|
| 370 |
+
.b8 52
|
| 371 |
+
.b8 107
|
| 372 |
+
.b8 100
|
| 373 |
+
.b8 109
|
| 374 |
+
.b8 115
|
| 375 |
+
.b8 54
|
| 376 |
+
.b8 122
|
| 377 |
+
.b8 102
|
| 378 |
+
.b8 97
|
| 379 |
+
.b8 100
|
| 380 |
+
.b8 55
|
| 381 |
+
.b8 110
|
| 382 |
+
.b8 103
|
| 383 |
+
.b8 112
|
| 384 |
+
.b8 102
|
| 385 |
+
.b8 122
|
| 386 |
+
.b8 113
|
| 387 |
+
.b8 54
|
| 388 |
+
.b8 104
|
| 389 |
+
.b8 50
|
| 390 |
+
.b8 111
|
| 391 |
+
.b8 119
|
| 392 |
+
.b8 119
|
| 393 |
+
.b8 119
|
| 394 |
+
.b8 46
|
| 395 |
+
.b8 112
|
| 396 |
+
.b8 121
|
| 397 |
+
.b8 0
|
| 398 |
+
.b32 .debug_line // DW_AT_stmt_list
|
| 399 |
+
.b8 47 // DW_AT_comp_dir
|
| 400 |
+
.b8 119
|
| 401 |
+
.b8 111
|
| 402 |
+
.b8 114
|
| 403 |
+
.b8 107
|
| 404 |
+
.b8 115
|
| 405 |
+
.b8 112
|
| 406 |
+
.b8 97
|
| 407 |
+
.b8 99
|
| 408 |
+
.b8 101
|
| 409 |
+
.b8 47
|
| 410 |
+
.b8 104
|
| 411 |
+
.b8 97
|
| 412 |
+
.b8 110
|
| 413 |
+
.b8 114
|
| 414 |
+
.b8 117
|
| 415 |
+
.b8 105
|
| 416 |
+
.b8 47
|
| 417 |
+
.b8 83
|
| 418 |
+
.b8 112
|
| 419 |
+
.b8 101
|
| 420 |
+
.b8 99
|
| 421 |
+
.b8 70
|
| 422 |
+
.b8 111
|
| 423 |
+
.b8 114
|
| 424 |
+
.b8 103
|
| 425 |
+
.b8 101
|
| 426 |
+
.b8 45
|
| 427 |
+
.b8 101
|
| 428 |
+
.b8 120
|
| 429 |
+
.b8 116
|
| 430 |
+
.b8 47
|
| 431 |
+
.b8 99
|
| 432 |
+
.b8 97
|
| 433 |
+
.b8 99
|
| 434 |
+
.b8 104
|
| 435 |
+
.b8 101
|
| 436 |
+
.b8 47
|
| 437 |
+
.b8 99
|
| 438 |
+
.b8 111
|
| 439 |
+
.b8 109
|
| 440 |
+
.b8 112
|
| 441 |
+
.b8 105
|
| 442 |
+
.b8 108
|
| 443 |
+
.b8 101
|
| 444 |
+
.b8 100
|
| 445 |
+
.b8 95
|
| 446 |
+
.b8 107
|
| 447 |
+
.b8 101
|
| 448 |
+
.b8 114
|
| 449 |
+
.b8 110
|
| 450 |
+
.b8 101
|
| 451 |
+
.b8 108
|
| 452 |
+
.b8 115
|
| 453 |
+
.b8 47
|
| 454 |
+
.b8 119
|
| 455 |
+
.b8 104
|
| 456 |
+
.b8 0
|
| 457 |
+
.b8 2 // Abbrev [2] 0x8b:0x38 DW_TAG_subprogram
|
| 458 |
+
.b8 116 // DW_AT_name
|
| 459 |
+
.b8 114
|
| 460 |
+
.b8 105
|
| 461 |
+
.b8 116
|
| 462 |
+
.b8 111
|
| 463 |
+
.b8 110
|
| 464 |
+
.b8 95
|
| 465 |
+
.b8 114
|
| 466 |
+
.b8 101
|
| 467 |
+
.b8 100
|
| 468 |
+
.b8 95
|
| 469 |
+
.b8 102
|
| 470 |
+
.b8 117
|
| 471 |
+
.b8 115
|
| 472 |
+
.b8 101
|
| 473 |
+
.b8 100
|
| 474 |
+
.b8 95
|
| 475 |
+
.b8 95
|
| 476 |
+
.b8 116
|
| 477 |
+
.b8 111
|
| 478 |
+
.b8 95
|
| 479 |
+
.b8 99
|
| 480 |
+
.b8 111
|
| 481 |
+
.b8 112
|
| 482 |
+
.b8 121
|
| 483 |
+
.b8 95
|
| 484 |
+
.b8 99
|
| 485 |
+
.b8 108
|
| 486 |
+
.b8 111
|
| 487 |
+
.b8 110
|
| 488 |
+
.b8 101
|
| 489 |
+
.b8 95
|
| 490 |
+
.b8 115
|
| 491 |
+
.b8 108
|
| 492 |
+
.b8 105
|
| 493 |
+
.b8 99
|
| 494 |
+
.b8 101
|
| 495 |
+
.b8 95
|
| 496 |
+
.b8 115
|
| 497 |
+
.b8 117
|
| 498 |
+
.b8 109
|
| 499 |
+
.b8 95
|
| 500 |
+
.b8 116
|
| 501 |
+
.b8 114
|
| 502 |
+
.b8 97
|
| 503 |
+
.b8 110
|
| 504 |
+
.b8 115
|
| 505 |
+
.b8 112
|
| 506 |
+
.b8 111
|
| 507 |
+
.b8 115
|
| 508 |
+
.b8 101
|
| 509 |
+
.b8 95
|
| 510 |
+
.b8 53
|
| 511 |
+
.b8 0
|
| 512 |
+
.b8 1 // DW_AT_inline
|
| 513 |
+
.b8 3 // Abbrev [3] 0xc3:0x2e DW_TAG_subprogram
|
| 514 |
+
.b64 $L__func_begin0 // DW_AT_low_pc
|
| 515 |
+
.b64 $L__func_end0 // DW_AT_high_pc
|
| 516 |
+
.b32 139 // DW_AT_abstract_origin
|
| 517 |
+
.b8 4 // Abbrev [4] 0xd8:0x18 DW_TAG_inlined_subroutine
|
| 518 |
+
.b32 139 // DW_AT_abstract_origin
|
| 519 |
+
.b64 $L__tmp1 // DW_AT_low_pc
|
| 520 |
+
.b64 $L__tmp4 // DW_AT_high_pc
|
| 521 |
+
.b8 1 // DW_AT_call_file
|
| 522 |
+
.b8 41 // DW_AT_call_line
|
| 523 |
+
.b8 25 // DW_AT_call_column
|
| 524 |
+
.b8 0 // End Of Children Mark
|
| 525 |
+
.b8 0 // End Of Children Mark
|
| 526 |
+
}
|
| 527 |
+
.section .debug_macinfo { }
|
SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.source
ADDED
|
@@ -0,0 +1,193 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":18:0)
|
| 2 |
+
#loc41 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":285:0)
|
| 3 |
+
#loc43 = loc(unknown)
|
| 4 |
+
#loc46 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":260:0)
|
| 5 |
+
#loc50 = loc("in_ptr0"(#loc))
|
| 6 |
+
#loc51 = loc("out_ptr1"(#loc))
|
| 7 |
+
#loc52 = loc("ks0"(#loc))
|
| 8 |
+
#loc53 = loc("ks1"(#loc))
|
| 9 |
+
#loc54 = loc("xnumel"(#loc))
|
| 10 |
+
#loc55 = loc("r0_numel"(#loc))
|
| 11 |
+
#loc85 = loc("input"(#loc41))
|
| 12 |
+
#loc86 = loc("a"(#loc46))
|
| 13 |
+
#loc87 = loc("b"(#loc46))
|
| 14 |
+
module {
|
| 15 |
+
tt.func public @triton_red_fused__to_copy_clone_slice_sum_transpose_5(%in_ptr0: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %out_ptr1: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("out_ptr1"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %xnumel: i32 loc("xnumel"(#loc)), %r0_numel: i32 loc("r0_numel"(#loc))) attributes {noinline = false} {
|
| 16 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc56)
|
| 17 |
+
%xoffset_0 = arith.constant 64 : i32 loc(#loc57)
|
| 18 |
+
%xoffset_1 = arith.constant 64 : i32 loc(#loc57)
|
| 19 |
+
%xoffset_2 = arith.muli %xoffset, %xoffset_1 : i32 loc(#loc57)
|
| 20 |
+
%xindex = tt.make_range {end = 64 : i32, start = 0 : i32} : tensor<64xi32> loc(#loc58)
|
| 21 |
+
%xindex_3 = tt.expand_dims %xindex {axis = 1 : i32} : tensor<64xi32> -> tensor<64x1xi32> loc(#loc59)
|
| 22 |
+
%xindex_4 = tt.splat %xoffset_2 : i32 -> tensor<64x1xi32> loc(#loc60)
|
| 23 |
+
%xindex_5 = arith.addi %xindex_4, %xindex_3 : tensor<64x1xi32> loc(#loc60)
|
| 24 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<64x1xi32> loc(#loc61)
|
| 25 |
+
%xmask_6 = arith.cmpi slt, %xindex_5, %xmask : tensor<64x1xi32> loc(#loc61)
|
| 26 |
+
%r0_base = tt.make_range {end = 16 : i32, start = 0 : i32} : tensor<16xi32> loc(#loc62)
|
| 27 |
+
%r0_base_7 = tt.expand_dims %r0_base {axis = 0 : i32} : tensor<16xi32> -> tensor<1x16xi32> loc(#loc63)
|
| 28 |
+
%x0 = arith.extsi %xindex_5 : tensor<64x1xi32> to tensor<64x1xi64> loc(#loc64)
|
| 29 |
+
%x0_8 = tt.splat %ks0 : i64 -> tensor<64x1xi64> loc(#loc64)
|
| 30 |
+
%x0_9 = arith.remsi %x0, %x0_8 : tensor<64x1xi64> loc(#loc64)
|
| 31 |
+
%x1 = arith.extsi %xindex_5 : tensor<64x1xi32> to tensor<64x1xi64> loc(#loc65)
|
| 32 |
+
%x1_10 = tt.splat %ks0 : i64 -> tensor<64x1xi64> loc(#loc65)
|
| 33 |
+
%x1_11 = arith.divsi %x1, %x1_10 : tensor<64x1xi64> loc(#loc65)
|
| 34 |
+
%_tmp3 = arith.constant 0 : i64 loc(#loc66)
|
| 35 |
+
%_tmp3_12 = arith.constant dense<0> : tensor<64x16xi64> loc(#loc66)
|
| 36 |
+
%c0_i32 = arith.constant 0 : i32 loc(#loc12)
|
| 37 |
+
%c16_i32 = arith.constant 16 : i32 loc(#loc12)
|
| 38 |
+
%0 = arith.bitcast %c0_i32 : i32 to i32 loc(#loc12)
|
| 39 |
+
%1 = arith.bitcast %r0_numel : i32 to i32 loc(#loc12)
|
| 40 |
+
%2 = arith.bitcast %c16_i32 : i32 to i32 loc(#loc12)
|
| 41 |
+
%3 = ub.poison : i32 loc(#loc12)
|
| 42 |
+
%_tmp3_13 = scf.for %r0_offset = %0 to %1 step %2 iter_args(%_tmp3_18 = %_tmp3_12) -> (tensor<64x16xi64>) : i32 {
|
| 43 |
+
%r0_index = tt.splat %r0_offset : i32 -> tensor<1x16xi32> loc(#loc68)
|
| 44 |
+
%r0_index_19 = arith.addi %r0_index, %r0_base_7 : tensor<1x16xi32> loc(#loc68)
|
| 45 |
+
%r0_mask = tt.splat %r0_numel : i32 -> tensor<1x16xi32> loc(#loc69)
|
| 46 |
+
%r0_mask_20 = arith.cmpi slt, %r0_index_19, %r0_mask : tensor<1x16xi32> loc(#loc69)
|
| 47 |
+
%tmp0 = arith.extsi %r0_index_19 : tensor<1x16xi32> to tensor<1x16xi64> loc(#loc70)
|
| 48 |
+
%tmp0_21 = tt.splat %ks0 : i64 -> tensor<1x16xi64> loc(#loc70)
|
| 49 |
+
%tmp0_22 = arith.muli %tmp0_21, %tmp0 : tensor<1x16xi64> loc(#loc70)
|
| 50 |
+
%tmp0_23 = tt.broadcast %x0_9 : tensor<64x1xi64> -> tensor<64x16xi64> loc(#loc71)
|
| 51 |
+
%tmp0_24 = tt.broadcast %tmp0_22 : tensor<1x16xi64> -> tensor<64x16xi64> loc(#loc71)
|
| 52 |
+
%tmp0_25 = arith.addi %tmp0_23, %tmp0_24 : tensor<64x16xi64> loc(#loc71)
|
| 53 |
+
%tmp0_26 = arith.muli %ks0, %ks1 : i64 loc(#loc72)
|
| 54 |
+
%tmp0_27 = tt.splat %tmp0_26 : i64 -> tensor<64x1xi64> loc(#loc73)
|
| 55 |
+
%tmp0_28 = arith.muli %tmp0_27, %x1_11 : tensor<64x1xi64> loc(#loc73)
|
| 56 |
+
%tmp0_29 = tt.broadcast %tmp0_28 : tensor<64x1xi64> -> tensor<64x16xi64> loc(#loc74)
|
| 57 |
+
%tmp0_30 = arith.addi %tmp0_25, %tmp0_29 : tensor<64x16xi64> loc(#loc74)
|
| 58 |
+
%tmp0_31 = tt.splat %in_ptr0 : !tt.ptr<i32> -> tensor<64x16x!tt.ptr<i32>> loc(#loc75)
|
| 59 |
+
%tmp0_32 = tt.addptr %tmp0_31, %tmp0_30 : tensor<64x16x!tt.ptr<i32>>, tensor<64x16xi64> loc(#loc75)
|
| 60 |
+
%tmp0_33 = tt.broadcast %r0_mask_20 : tensor<1x16xi1> -> tensor<64x16xi1> loc(#loc76)
|
| 61 |
+
%tmp0_34 = tt.broadcast %xmask_6 : tensor<64x1xi1> -> tensor<64x16xi1> loc(#loc76)
|
| 62 |
+
%tmp0_35 = arith.andi %tmp0_33, %tmp0_34 : tensor<64x16xi1> loc(#loc76)
|
| 63 |
+
%tmp0_36 = arith.constant 0.000000e+00 : f32 loc(#loc77)
|
| 64 |
+
%tmp0_37 = arith.constant dense<0.000000e+00> : tensor<64x16xf32> loc(#loc77)
|
| 65 |
+
%tmp0_38 = arith.fptosi %tmp0_37 : tensor<64x16xf32> to tensor<64x16xi32> loc(#loc77)
|
| 66 |
+
%tmp0_39 = tt.load %tmp0_32, %tmp0_35, %tmp0_38 evictionPolicy = evict_last : tensor<64x16x!tt.ptr<i32>> loc(#loc77)
|
| 67 |
+
%tmp1 = arith.extsi %tmp0_39 : tensor<64x16xi32> to tensor<64x16xi64> loc(#loc78)
|
| 68 |
+
%tmp4 = arith.addi %_tmp3_18, %tmp1 : tensor<64x16xi64> loc(#loc79)
|
| 69 |
+
%_tmp3_40 = tt.broadcast %r0_mask_20 : tensor<1x16xi1> -> tensor<64x16xi1> loc(#loc80)
|
| 70 |
+
%_tmp3_41 = tt.broadcast %xmask_6 : tensor<64x1xi1> -> tensor<64x16xi1> loc(#loc80)
|
| 71 |
+
%_tmp3_42 = arith.andi %_tmp3_40, %_tmp3_41 : tensor<64x16xi1> loc(#loc80)
|
| 72 |
+
%_tmp3_43 = arith.select %_tmp3_42, %tmp4, %_tmp3_18 : tensor<64x16xi1>, tensor<64x16xi64> loc(#loc81)
|
| 73 |
+
scf.yield %_tmp3_43 : tensor<64x16xi64> loc(#loc27)
|
| 74 |
+
} loc(#loc67)
|
| 75 |
+
%tmp3 = tt.call @"triton.language.standard.sum__i64S64_16S__(1,)cconstexpr_1__(2,)cconstexpr_False__(3,)cNone"(%_tmp3_13) : (tensor<64x16xi64>) -> tensor<64xi64> loc(#loc82)
|
| 76 |
+
%tmp3_14 = tt.expand_dims %tmp3 {axis = 1 : i32} : tensor<64xi64> -> tensor<64x1xi64> loc(#loc83)
|
| 77 |
+
%tmp5 = arith.trunci %tmp3_14 : tensor<64x1xi64> to tensor<64x1xi32> loc(#loc84)
|
| 78 |
+
%c1_i32 = arith.constant 1 : i32 loc(#loc31)
|
| 79 |
+
%4 = arith.extsi %c1_i32 : i32 to i64 loc(#loc31)
|
| 80 |
+
%5 = arith.cmpi sge, %4, %ks0 : i64 loc(#loc31)
|
| 81 |
+
%c1_i32_15 = arith.constant 1 : i32 loc(#loc32)
|
| 82 |
+
%c1_i32_16 = arith.constant 1 : i32 loc(#loc32)
|
| 83 |
+
%6 = arith.extui %5 : i1 to i32 loc(#loc32)
|
| 84 |
+
%7 = arith.muli %c1_i32_16, %6 : i32 loc(#loc32)
|
| 85 |
+
%c1_i32_17 = arith.constant 1 : i32 loc(#loc33)
|
| 86 |
+
%8 = arith.extsi %c1_i32_17 : i32 to i64 loc(#loc33)
|
| 87 |
+
%9 = arith.cmpi sgt, %ks0, %8 : i64 loc(#loc33)
|
| 88 |
+
%10 = arith.extui %9 : i1 to i64 loc(#loc34)
|
| 89 |
+
%11 = arith.muli %ks0, %10 : i64 loc(#loc34)
|
| 90 |
+
%12 = arith.extsi %7 : i32 to i64 loc(#loc35)
|
| 91 |
+
%13 = arith.addi %12, %11 : i64 loc(#loc35)
|
| 92 |
+
%14 = tt.splat %13 : i64 -> tensor<64x1xi64> loc(#loc36)
|
| 93 |
+
%15 = arith.muli %x1_11, %14 : tensor<64x1xi64> loc(#loc36)
|
| 94 |
+
%16 = arith.addi %x0_9, %15 : tensor<64x1xi64> loc(#loc37)
|
| 95 |
+
%17 = tt.splat %out_ptr1 : !tt.ptr<i32> -> tensor<64x1x!tt.ptr<i32>> loc(#loc38)
|
| 96 |
+
%18 = tt.addptr %17, %16 : tensor<64x1x!tt.ptr<i32>>, tensor<64x1xi64> loc(#loc38)
|
| 97 |
+
tt.store %18, %tmp5, %xmask_6 : tensor<64x1x!tt.ptr<i32>> loc(#loc39)
|
| 98 |
+
tt.return loc(#loc40)
|
| 99 |
+
} loc(#loc)
|
| 100 |
+
tt.func private @"triton.language.standard.sum__i64S64_16S__(1,)cconstexpr_1__(2,)cconstexpr_False__(3,)cNone"(%input: tensor<64x16xi64> loc("input"(#loc41))) -> tensor<64xi64> attributes {noinline = false} {
|
| 101 |
+
%0 = "tt.reduce"(%input) <{axis = 1 : i32}> ({
|
| 102 |
+
^bb0(%arg1: i64 loc(unknown), %arg2: i64 loc(unknown)):
|
| 103 |
+
%2 = tt.call @triton.language.standard._sum_combine__i64_i64__(%arg1, %arg2) : (i64, i64) -> i64 loc(#loc42)
|
| 104 |
+
tt.reduce.return %2 : i64 loc(#loc42)
|
| 105 |
+
}) : (tensor<64x16xi64>) -> tensor<64xi64> loc(#loc42)
|
| 106 |
+
tt.return %0 : tensor<64xi64> loc(#loc44)
|
| 107 |
+
^bb1: // no predecessors
|
| 108 |
+
%1 = ub.poison : tensor<64xi64> loc(#loc45)
|
| 109 |
+
tt.return %1 : tensor<64xi64> loc(#loc45)
|
| 110 |
+
} loc(#loc41)
|
| 111 |
+
tt.func private @triton.language.standard._sum_combine__i64_i64__(%a: i64 loc("a"(#loc46)), %b: i64 loc("b"(#loc46))) -> i64 attributes {noinline = false} {
|
| 112 |
+
%0 = arith.addi %a, %b : i64 loc(#loc47)
|
| 113 |
+
tt.return %0 : i64 loc(#loc48)
|
| 114 |
+
^bb1: // no predecessors
|
| 115 |
+
%1 = ub.poison : i64 loc(#loc49)
|
| 116 |
+
tt.return %1 : i64 loc(#loc49)
|
| 117 |
+
} loc(#loc46)
|
| 118 |
+
} loc(#loc)
|
| 119 |
+
#loc1 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":21:28)
|
| 120 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":21:33)
|
| 121 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":22:36)
|
| 122 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":22:44)
|
| 123 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":22:23)
|
| 124 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":23:21)
|
| 125 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":24:27)
|
| 126 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":24:37)
|
| 127 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":26:19)
|
| 128 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":27:19)
|
| 129 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":28:43)
|
| 130 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":30:40)
|
| 131 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":31:31)
|
| 132 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":32:29)
|
| 133 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:43)
|
| 134 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:39)
|
| 135 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:54)
|
| 136 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:58)
|
| 137 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:50)
|
| 138 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:34)
|
| 139 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:73)
|
| 140 |
+
#loc22 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:63)
|
| 141 |
+
#loc23 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":37:23)
|
| 142 |
+
#loc24 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":39:23)
|
| 143 |
+
#loc25 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":40:35)
|
| 144 |
+
#loc26 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":40:48)
|
| 145 |
+
#loc27 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":40:8)
|
| 146 |
+
#loc28 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":41:25)
|
| 147 |
+
#loc29 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":41:28)
|
| 148 |
+
#loc30 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":42:19)
|
| 149 |
+
#loc31 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:49)
|
| 150 |
+
#loc32 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:41)
|
| 151 |
+
#loc33 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:75)
|
| 152 |
+
#loc34 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:66)
|
| 153 |
+
#loc35 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:57)
|
| 154 |
+
#loc36 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:34)
|
| 155 |
+
#loc37 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:30)
|
| 156 |
+
#loc38 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:25)
|
| 157 |
+
#loc39 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:88)
|
| 158 |
+
#loc40 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:4)
|
| 159 |
+
#loc42 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:36)
|
| 160 |
+
#loc44 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:11)
|
| 161 |
+
#loc45 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:4)
|
| 162 |
+
#loc47 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:15)
|
| 163 |
+
#loc48 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:11)
|
| 164 |
+
#loc49 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:4)
|
| 165 |
+
#loc56 = loc("xoffset"(#loc1))
|
| 166 |
+
#loc57 = loc("xoffset"(#loc2))
|
| 167 |
+
#loc58 = loc("xindex"(#loc3))
|
| 168 |
+
#loc59 = loc("xindex"(#loc4))
|
| 169 |
+
#loc60 = loc("xindex"(#loc5))
|
| 170 |
+
#loc61 = loc("xmask"(#loc6))
|
| 171 |
+
#loc62 = loc("r0_base"(#loc7))
|
| 172 |
+
#loc63 = loc("r0_base"(#loc8))
|
| 173 |
+
#loc64 = loc("x0"(#loc9))
|
| 174 |
+
#loc65 = loc("x1"(#loc10))
|
| 175 |
+
#loc66 = loc("_tmp3"(#loc11))
|
| 176 |
+
#loc67 = loc("_tmp3"(#loc12))
|
| 177 |
+
#loc68 = loc("r0_index"(#loc13))
|
| 178 |
+
#loc69 = loc("r0_mask"(#loc14))
|
| 179 |
+
#loc70 = loc("tmp0"(#loc15))
|
| 180 |
+
#loc71 = loc("tmp0"(#loc16))
|
| 181 |
+
#loc72 = loc("tmp0"(#loc17))
|
| 182 |
+
#loc73 = loc("tmp0"(#loc18))
|
| 183 |
+
#loc74 = loc("tmp0"(#loc19))
|
| 184 |
+
#loc75 = loc("tmp0"(#loc20))
|
| 185 |
+
#loc76 = loc("tmp0"(#loc21))
|
| 186 |
+
#loc77 = loc("tmp0"(#loc22))
|
| 187 |
+
#loc78 = loc("tmp1"(#loc23))
|
| 188 |
+
#loc79 = loc("tmp4"(#loc24))
|
| 189 |
+
#loc80 = loc("_tmp3"(#loc25))
|
| 190 |
+
#loc81 = loc("_tmp3"(#loc26))
|
| 191 |
+
#loc82 = loc("tmp3"(#loc28))
|
| 192 |
+
#loc83 = loc("tmp3"(#loc29))
|
| 193 |
+
#loc84 = loc("tmp5"(#loc30))
|
SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttgir
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#blocked = #ttg.blocked<{sizePerThread = [1, 1], threadsPerWarp = [32, 1], warpsPerCTA = [2, 4], order = [0, 1]}>
|
| 2 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":18:0)
|
| 3 |
+
#loc1 = loc(unknown)
|
| 4 |
+
#loc26 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":41:25)
|
| 5 |
+
#loc40 = loc("in_ptr0"(#loc))
|
| 6 |
+
#loc41 = loc("out_ptr1"(#loc))
|
| 7 |
+
#loc42 = loc("ks0"(#loc))
|
| 8 |
+
#loc43 = loc("ks1"(#loc))
|
| 9 |
+
#loc44 = loc("xnumel"(#loc))
|
| 10 |
+
#loc45 = loc("r0_numel"(#loc))
|
| 11 |
+
#loc68 = loc("tmp3"(#loc26))
|
| 12 |
+
#loc73 = loc(callsite(#loc1 at #loc68))
|
| 13 |
+
module attributes {"ttg.num-ctas" = 1 : i32, "ttg.num-warps" = 8 : i32, ttg.target = "cuda:90", "ttg.threads-per-warp" = 32 : i32} {
|
| 14 |
+
tt.func public @triton_red_fused__to_copy_clone_slice_sum_transpose_5(%in_ptr0: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %out_ptr1: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("out_ptr1"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %xnumel: i32 loc("xnumel"(#loc)), %r0_numel: i32 loc("r0_numel"(#loc))) attributes {noinline = false} {
|
| 15 |
+
%cst = arith.constant dense<0> : tensor<64x16xi64, #blocked> loc(#loc1)
|
| 16 |
+
%c0_i32 = arith.constant 0 : i32 loc(#loc1)
|
| 17 |
+
%c16_i32 = arith.constant 16 : i32 loc(#loc1)
|
| 18 |
+
%c1_i64 = arith.constant 1 : i64 loc(#loc1)
|
| 19 |
+
%c64_i32 = arith.constant 64 : i32 loc(#loc1)
|
| 20 |
+
%cst_0 = arith.constant dense<0> : tensor<64x16xi32, #blocked> loc(#loc1)
|
| 21 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc46)
|
| 22 |
+
%xoffset_1 = arith.muli %xoffset, %c64_i32 : i32 loc(#loc47)
|
| 23 |
+
%xindex = tt.make_range {end = 64 : i32, start = 0 : i32} : tensor<64xi32, #ttg.slice<{dim = 1, parent = #blocked}>> loc(#loc48)
|
| 24 |
+
%xindex_2 = tt.expand_dims %xindex {axis = 1 : i32} : tensor<64xi32, #ttg.slice<{dim = 1, parent = #blocked}>> -> tensor<64x1xi32, #blocked> loc(#loc48)
|
| 25 |
+
%xindex_3 = tt.splat %xoffset_1 : i32 -> tensor<64x1xi32, #blocked> loc(#loc49)
|
| 26 |
+
%xindex_4 = arith.addi %xindex_3, %xindex_2 : tensor<64x1xi32, #blocked> loc(#loc49)
|
| 27 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<64x1xi32, #blocked> loc(#loc50)
|
| 28 |
+
%xmask_5 = arith.cmpi slt, %xindex_4, %xmask : tensor<64x1xi32, #blocked> loc(#loc50)
|
| 29 |
+
%r0_base = tt.make_range {end = 16 : i32, start = 0 : i32} : tensor<16xi32, #ttg.slice<{dim = 0, parent = #blocked}>> loc(#loc51)
|
| 30 |
+
%r0_base_6 = tt.expand_dims %r0_base {axis = 0 : i32} : tensor<16xi32, #ttg.slice<{dim = 0, parent = #blocked}>> -> tensor<1x16xi32, #blocked> loc(#loc51)
|
| 31 |
+
%x0 = arith.extsi %xindex_4 : tensor<64x1xi32, #blocked> to tensor<64x1xi64, #blocked> loc(#loc52)
|
| 32 |
+
%x0_7 = tt.splat %ks0 : i64 -> tensor<64x1xi64, #blocked> loc(#loc52)
|
| 33 |
+
%x0_8 = arith.remsi %x0, %x0_7 : tensor<64x1xi64, #blocked> loc(#loc52)
|
| 34 |
+
%x1 = arith.divsi %x0, %x0_7 : tensor<64x1xi64, #blocked> loc(#loc53)
|
| 35 |
+
%r0_mask = tt.splat %r0_numel : i32 -> tensor<1x16xi32, #blocked> loc(#loc54)
|
| 36 |
+
%tmp0 = tt.splat %ks0 : i64 -> tensor<1x16xi64, #blocked> loc(#loc55)
|
| 37 |
+
%tmp0_9 = tt.broadcast %x0_8 : tensor<64x1xi64, #blocked> -> tensor<64x16xi64, #blocked> loc(#loc56)
|
| 38 |
+
%tmp0_10 = arith.muli %ks0, %ks1 : i64 loc(#loc57)
|
| 39 |
+
%tmp0_11 = tt.splat %tmp0_10 : i64 -> tensor<64x1xi64, #blocked> loc(#loc58)
|
| 40 |
+
%tmp0_12 = arith.muli %tmp0_11, %x1 : tensor<64x1xi64, #blocked> loc(#loc58)
|
| 41 |
+
%tmp0_13 = tt.broadcast %tmp0_12 : tensor<64x1xi64, #blocked> -> tensor<64x16xi64, #blocked> loc(#loc59)
|
| 42 |
+
%tmp0_14 = tt.splat %in_ptr0 : !tt.ptr<i32> -> tensor<64x16x!tt.ptr<i32>, #blocked> loc(#loc60)
|
| 43 |
+
%tmp0_15 = tt.broadcast %xmask_5 : tensor<64x1xi1, #blocked> -> tensor<64x16xi1, #blocked> loc(#loc61)
|
| 44 |
+
%_tmp3 = scf.for %_tmp3_17 = %c0_i32 to %r0_numel step %c16_i32 iter_args(%_tmp3_18 = %cst) -> (tensor<64x16xi64, #blocked>) : i32 {
|
| 45 |
+
%r0_index = tt.splat %_tmp3_17 : i32 -> tensor<1x16xi32, #blocked> loc(#loc63)
|
| 46 |
+
%r0_index_19 = arith.addi %r0_index, %r0_base_6 : tensor<1x16xi32, #blocked> loc(#loc63)
|
| 47 |
+
%r0_mask_20 = arith.cmpi slt, %r0_index_19, %r0_mask : tensor<1x16xi32, #blocked> loc(#loc54)
|
| 48 |
+
%tmp0_21 = arith.extsi %r0_index_19 : tensor<1x16xi32, #blocked> to tensor<1x16xi64, #blocked> loc(#loc55)
|
| 49 |
+
%tmp0_22 = arith.muli %tmp0, %tmp0_21 : tensor<1x16xi64, #blocked> loc(#loc55)
|
| 50 |
+
%tmp0_23 = tt.broadcast %tmp0_22 : tensor<1x16xi64, #blocked> -> tensor<64x16xi64, #blocked> loc(#loc56)
|
| 51 |
+
%tmp0_24 = arith.addi %tmp0_9, %tmp0_23 : tensor<64x16xi64, #blocked> loc(#loc56)
|
| 52 |
+
%tmp0_25 = arith.addi %tmp0_24, %tmp0_13 : tensor<64x16xi64, #blocked> loc(#loc59)
|
| 53 |
+
%tmp0_26 = tt.addptr %tmp0_14, %tmp0_25 : tensor<64x16x!tt.ptr<i32>, #blocked>, tensor<64x16xi64, #blocked> loc(#loc60)
|
| 54 |
+
%tmp0_27 = tt.broadcast %r0_mask_20 : tensor<1x16xi1, #blocked> -> tensor<64x16xi1, #blocked> loc(#loc61)
|
| 55 |
+
%tmp0_28 = arith.andi %tmp0_27, %tmp0_15 : tensor<64x16xi1, #blocked> loc(#loc61)
|
| 56 |
+
%tmp0_29 = tt.load %tmp0_26, %tmp0_28, %cst_0 evictionPolicy = evict_last : tensor<64x16x!tt.ptr<i32>, #blocked> loc(#loc64)
|
| 57 |
+
%tmp1 = arith.extsi %tmp0_29 : tensor<64x16xi32, #blocked> to tensor<64x16xi64, #blocked> loc(#loc65)
|
| 58 |
+
%tmp4 = arith.addi %_tmp3_18, %tmp1 : tensor<64x16xi64, #blocked> loc(#loc66)
|
| 59 |
+
%_tmp3_30 = arith.select %tmp0_28, %tmp4, %_tmp3_18 : tensor<64x16xi1, #blocked>, tensor<64x16xi64, #blocked> loc(#loc67)
|
| 60 |
+
scf.yield %_tmp3_30 : tensor<64x16xi64, #blocked> loc(#loc24)
|
| 61 |
+
} loc(#loc62)
|
| 62 |
+
%tmp3 = "tt.reduce"(%_tmp3) <{axis = 1 : i32}> ({
|
| 63 |
+
^bb0(%tmp3_17: i64 loc(callsite(#loc1 at #loc68)), %tmp3_18: i64 loc(callsite(#loc1 at #loc68))):
|
| 64 |
+
%tmp3_19 = arith.addi %tmp3_17, %tmp3_18 : i64 loc(#loc74)
|
| 65 |
+
tt.reduce.return %tmp3_19 : i64 loc(#loc72)
|
| 66 |
+
}) : (tensor<64x16xi64, #blocked>) -> tensor<64xi64, #ttg.slice<{dim = 1, parent = #blocked}>> loc(#loc72)
|
| 67 |
+
%tmp3_16 = tt.expand_dims %tmp3 {axis = 1 : i32} : tensor<64xi64, #ttg.slice<{dim = 1, parent = #blocked}>> -> tensor<64x1xi64, #blocked> loc(#loc69)
|
| 68 |
+
%tmp5 = arith.trunci %tmp3_16 : tensor<64x1xi64, #blocked> to tensor<64x1xi32, #blocked> loc(#loc70)
|
| 69 |
+
%0 = arith.cmpi sle, %ks0, %c1_i64 : i64 loc(#loc30)
|
| 70 |
+
%1 = arith.cmpi sgt, %ks0, %c1_i64 : i64 loc(#loc31)
|
| 71 |
+
%2 = arith.extui %1 : i1 to i64 loc(#loc32)
|
| 72 |
+
%3 = arith.muli %ks0, %2 : i64 loc(#loc32)
|
| 73 |
+
%4 = arith.extui %0 : i1 to i64 loc(#loc71)
|
| 74 |
+
%5 = arith.addi %4, %3 : i64 loc(#loc33)
|
| 75 |
+
%6 = tt.splat %5 : i64 -> tensor<64x1xi64, #blocked> loc(#loc35)
|
| 76 |
+
%7 = arith.muli %x1, %6 : tensor<64x1xi64, #blocked> loc(#loc35)
|
| 77 |
+
%8 = arith.addi %x0_8, %7 : tensor<64x1xi64, #blocked> loc(#loc36)
|
| 78 |
+
%9 = tt.splat %out_ptr1 : !tt.ptr<i32> -> tensor<64x1x!tt.ptr<i32>, #blocked> loc(#loc37)
|
| 79 |
+
%10 = tt.addptr %9, %8 : tensor<64x1x!tt.ptr<i32>, #blocked>, tensor<64x1xi64, #blocked> loc(#loc37)
|
| 80 |
+
tt.store %10, %tmp5, %xmask_5 : tensor<64x1x!tt.ptr<i32>, #blocked> loc(#loc38)
|
| 81 |
+
tt.return loc(#loc39)
|
| 82 |
+
} loc(#loc)
|
| 83 |
+
} loc(#loc)
|
| 84 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":21:28)
|
| 85 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":21:33)
|
| 86 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":22:44)
|
| 87 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":22:23)
|
| 88 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":23:21)
|
| 89 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":24:37)
|
| 90 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":26:19)
|
| 91 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":27:19)
|
| 92 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":32:29)
|
| 93 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:43)
|
| 94 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:39)
|
| 95 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:54)
|
| 96 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:58)
|
| 97 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:50)
|
| 98 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:34)
|
| 99 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:73)
|
| 100 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":30:40)
|
| 101 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":31:31)
|
| 102 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:63)
|
| 103 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":37:23)
|
| 104 |
+
#loc22 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":39:23)
|
| 105 |
+
#loc23 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":40:48)
|
| 106 |
+
#loc24 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":40:8)
|
| 107 |
+
#loc25 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:36)
|
| 108 |
+
#loc27 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:15)
|
| 109 |
+
#loc28 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":41:28)
|
| 110 |
+
#loc29 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":42:19)
|
| 111 |
+
#loc30 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:49)
|
| 112 |
+
#loc31 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:75)
|
| 113 |
+
#loc32 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:66)
|
| 114 |
+
#loc33 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:57)
|
| 115 |
+
#loc34 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:41)
|
| 116 |
+
#loc35 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:34)
|
| 117 |
+
#loc36 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:30)
|
| 118 |
+
#loc37 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:25)
|
| 119 |
+
#loc38 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:88)
|
| 120 |
+
#loc39 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:4)
|
| 121 |
+
#loc46 = loc("xoffset"(#loc2))
|
| 122 |
+
#loc47 = loc("xoffset"(#loc3))
|
| 123 |
+
#loc48 = loc("xindex"(#loc4))
|
| 124 |
+
#loc49 = loc("xindex"(#loc5))
|
| 125 |
+
#loc50 = loc("xmask"(#loc6))
|
| 126 |
+
#loc51 = loc("r0_base"(#loc7))
|
| 127 |
+
#loc52 = loc("x0"(#loc8))
|
| 128 |
+
#loc53 = loc("x1"(#loc9))
|
| 129 |
+
#loc54 = loc("r0_mask"(#loc10))
|
| 130 |
+
#loc55 = loc("tmp0"(#loc11))
|
| 131 |
+
#loc56 = loc("tmp0"(#loc12))
|
| 132 |
+
#loc57 = loc("tmp0"(#loc13))
|
| 133 |
+
#loc58 = loc("tmp0"(#loc14))
|
| 134 |
+
#loc59 = loc("tmp0"(#loc15))
|
| 135 |
+
#loc60 = loc("tmp0"(#loc16))
|
| 136 |
+
#loc61 = loc("tmp0"(#loc17))
|
| 137 |
+
#loc62 = loc("_tmp3"(#loc18))
|
| 138 |
+
#loc63 = loc("r0_index"(#loc19))
|
| 139 |
+
#loc64 = loc("tmp0"(#loc20))
|
| 140 |
+
#loc65 = loc("tmp1"(#loc21))
|
| 141 |
+
#loc66 = loc("tmp4"(#loc22))
|
| 142 |
+
#loc67 = loc("_tmp3"(#loc23))
|
| 143 |
+
#loc69 = loc("tmp3"(#loc28))
|
| 144 |
+
#loc70 = loc("tmp5"(#loc29))
|
| 145 |
+
#loc71 = loc(fused[#loc33, #loc34])
|
| 146 |
+
#loc72 = loc(callsite(#loc25 at #loc68))
|
| 147 |
+
#loc74 = loc(callsite(#loc27 at #loc72))
|
SpecForge-ext/cache/compiled_kernels/triton/0/4KDH7QSSMYF7MJYA6EPR3AUBLWXBMLZZDI3RDUEZEJMIXYRGIJCA/triton_red_fused__to_copy_clone_slice_sum_transpose_5.ttir
ADDED
|
@@ -0,0 +1,152 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":18:0)
|
| 2 |
+
#loc1 = loc(unknown)
|
| 3 |
+
#loc29 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":41:25)
|
| 4 |
+
#loc43 = loc("in_ptr0"(#loc))
|
| 5 |
+
#loc44 = loc("out_ptr1"(#loc))
|
| 6 |
+
#loc45 = loc("ks0"(#loc))
|
| 7 |
+
#loc46 = loc("ks1"(#loc))
|
| 8 |
+
#loc47 = loc("xnumel"(#loc))
|
| 9 |
+
#loc48 = loc("r0_numel"(#loc))
|
| 10 |
+
#loc74 = loc("tmp3"(#loc29))
|
| 11 |
+
#loc79 = loc(callsite(#loc1 at #loc74))
|
| 12 |
+
module {
|
| 13 |
+
tt.func public @triton_red_fused__to_copy_clone_slice_sum_transpose_5(%in_ptr0: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %out_ptr1: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("out_ptr1"(#loc)), %ks0: i64 loc("ks0"(#loc)), %ks1: i64 loc("ks1"(#loc)), %xnumel: i32 loc("xnumel"(#loc)), %r0_numel: i32 loc("r0_numel"(#loc))) attributes {noinline = false} {
|
| 14 |
+
%c1_i64 = arith.constant 1 : i64 loc(#loc1)
|
| 15 |
+
%cst = arith.constant dense<0> : tensor<64x16xi32> loc(#loc1)
|
| 16 |
+
%c16_i32 = arith.constant 16 : i32 loc(#loc2)
|
| 17 |
+
%c0_i32 = arith.constant 0 : i32 loc(#loc2)
|
| 18 |
+
%_tmp3 = arith.constant dense<0> : tensor<64x16xi64> loc(#loc49)
|
| 19 |
+
%c64_i32 = arith.constant 64 : i32 loc(#loc1)
|
| 20 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc50)
|
| 21 |
+
%xoffset_0 = arith.muli %xoffset, %c64_i32 : i32 loc(#loc51)
|
| 22 |
+
%xindex = tt.make_range {end = 64 : i32, start = 0 : i32} : tensor<64xi32> loc(#loc52)
|
| 23 |
+
%xindex_1 = tt.expand_dims %xindex {axis = 1 : i32} : tensor<64xi32> -> tensor<64x1xi32> loc(#loc53)
|
| 24 |
+
%xindex_2 = tt.splat %xoffset_0 : i32 -> tensor<64x1xi32> loc(#loc54)
|
| 25 |
+
%xindex_3 = arith.addi %xindex_2, %xindex_1 : tensor<64x1xi32> loc(#loc54)
|
| 26 |
+
%xmask = tt.splat %xnumel : i32 -> tensor<64x1xi32> loc(#loc55)
|
| 27 |
+
%xmask_4 = arith.cmpi slt, %xindex_3, %xmask : tensor<64x1xi32> loc(#loc55)
|
| 28 |
+
%r0_base = tt.make_range {end = 16 : i32, start = 0 : i32} : tensor<16xi32> loc(#loc56)
|
| 29 |
+
%r0_base_5 = tt.expand_dims %r0_base {axis = 0 : i32} : tensor<16xi32> -> tensor<1x16xi32> loc(#loc57)
|
| 30 |
+
%x0 = arith.extsi %xindex_3 : tensor<64x1xi32> to tensor<64x1xi64> loc(#loc58)
|
| 31 |
+
%x0_6 = tt.splat %ks0 : i64 -> tensor<64x1xi64> loc(#loc58)
|
| 32 |
+
%x0_7 = arith.remsi %x0, %x0_6 : tensor<64x1xi64> loc(#loc58)
|
| 33 |
+
%x1 = arith.divsi %x0, %x0_6 : tensor<64x1xi64> loc(#loc59)
|
| 34 |
+
%_tmp3_8 = scf.for %r0_offset = %c0_i32 to %r0_numel step %c16_i32 iter_args(%_tmp3_10 = %_tmp3) -> (tensor<64x16xi64>) : i32 {
|
| 35 |
+
%r0_index = tt.splat %r0_offset : i32 -> tensor<1x16xi32> loc(#loc61)
|
| 36 |
+
%r0_index_11 = arith.addi %r0_index, %r0_base_5 : tensor<1x16xi32> loc(#loc61)
|
| 37 |
+
%r0_mask = tt.splat %r0_numel : i32 -> tensor<1x16xi32> loc(#loc62)
|
| 38 |
+
%r0_mask_12 = arith.cmpi slt, %r0_index_11, %r0_mask : tensor<1x16xi32> loc(#loc62)
|
| 39 |
+
%tmp0 = arith.extsi %r0_index_11 : tensor<1x16xi32> to tensor<1x16xi64> loc(#loc63)
|
| 40 |
+
%tmp0_13 = tt.splat %ks0 : i64 -> tensor<1x16xi64> loc(#loc63)
|
| 41 |
+
%tmp0_14 = arith.muli %tmp0_13, %tmp0 : tensor<1x16xi64> loc(#loc63)
|
| 42 |
+
%tmp0_15 = tt.broadcast %x0_7 : tensor<64x1xi64> -> tensor<64x16xi64> loc(#loc64)
|
| 43 |
+
%tmp0_16 = tt.broadcast %tmp0_14 : tensor<1x16xi64> -> tensor<64x16xi64> loc(#loc64)
|
| 44 |
+
%tmp0_17 = arith.addi %tmp0_15, %tmp0_16 : tensor<64x16xi64> loc(#loc64)
|
| 45 |
+
%tmp0_18 = arith.muli %ks0, %ks1 : i64 loc(#loc65)
|
| 46 |
+
%tmp0_19 = tt.splat %tmp0_18 : i64 -> tensor<64x1xi64> loc(#loc66)
|
| 47 |
+
%tmp0_20 = arith.muli %tmp0_19, %x1 : tensor<64x1xi64> loc(#loc66)
|
| 48 |
+
%tmp0_21 = tt.broadcast %tmp0_20 : tensor<64x1xi64> -> tensor<64x16xi64> loc(#loc67)
|
| 49 |
+
%tmp0_22 = arith.addi %tmp0_17, %tmp0_21 : tensor<64x16xi64> loc(#loc67)
|
| 50 |
+
%tmp0_23 = tt.splat %in_ptr0 : !tt.ptr<i32> -> tensor<64x16x!tt.ptr<i32>> loc(#loc68)
|
| 51 |
+
%tmp0_24 = tt.addptr %tmp0_23, %tmp0_22 : tensor<64x16x!tt.ptr<i32>>, tensor<64x16xi64> loc(#loc68)
|
| 52 |
+
%tmp0_25 = tt.broadcast %r0_mask_12 : tensor<1x16xi1> -> tensor<64x16xi1> loc(#loc69)
|
| 53 |
+
%tmp0_26 = tt.broadcast %xmask_4 : tensor<64x1xi1> -> tensor<64x16xi1> loc(#loc69)
|
| 54 |
+
%tmp0_27 = arith.andi %tmp0_25, %tmp0_26 : tensor<64x16xi1> loc(#loc69)
|
| 55 |
+
%tmp0_28 = tt.load %tmp0_24, %tmp0_27, %cst evictionPolicy = evict_last : tensor<64x16x!tt.ptr<i32>> loc(#loc70)
|
| 56 |
+
%tmp1 = arith.extsi %tmp0_28 : tensor<64x16xi32> to tensor<64x16xi64> loc(#loc71)
|
| 57 |
+
%tmp4 = arith.addi %_tmp3_10, %tmp1 : tensor<64x16xi64> loc(#loc72)
|
| 58 |
+
%_tmp3_29 = arith.select %tmp0_27, %tmp4, %_tmp3_10 : tensor<64x16xi1>, tensor<64x16xi64> loc(#loc73)
|
| 59 |
+
scf.yield %_tmp3_29 : tensor<64x16xi64> loc(#loc27)
|
| 60 |
+
} loc(#loc60)
|
| 61 |
+
%tmp3 = "tt.reduce"(%_tmp3_8) <{axis = 1 : i32}> ({
|
| 62 |
+
^bb0(%tmp3_10: i64 loc(callsite(#loc1 at #loc74)), %tmp3_11: i64 loc(callsite(#loc1 at #loc74))):
|
| 63 |
+
%tmp3_12 = arith.addi %tmp3_10, %tmp3_11 : i64 loc(#loc80)
|
| 64 |
+
tt.reduce.return %tmp3_12 : i64 loc(#loc78)
|
| 65 |
+
}) : (tensor<64x16xi64>) -> tensor<64xi64> loc(#loc78)
|
| 66 |
+
%tmp3_9 = tt.expand_dims %tmp3 {axis = 1 : i32} : tensor<64xi64> -> tensor<64x1xi64> loc(#loc75)
|
| 67 |
+
%tmp5 = arith.trunci %tmp3_9 : tensor<64x1xi64> to tensor<64x1xi32> loc(#loc76)
|
| 68 |
+
%0 = arith.cmpi sle, %ks0, %c1_i64 : i64 loc(#loc33)
|
| 69 |
+
%1 = arith.cmpi sgt, %ks0, %c1_i64 : i64 loc(#loc34)
|
| 70 |
+
%2 = arith.extui %1 : i1 to i64 loc(#loc35)
|
| 71 |
+
%3 = arith.muli %ks0, %2 : i64 loc(#loc35)
|
| 72 |
+
%4 = arith.extui %0 : i1 to i64 loc(#loc77)
|
| 73 |
+
%5 = arith.addi %4, %3 : i64 loc(#loc36)
|
| 74 |
+
%6 = tt.splat %5 : i64 -> tensor<64x1xi64> loc(#loc38)
|
| 75 |
+
%7 = arith.muli %x1, %6 : tensor<64x1xi64> loc(#loc38)
|
| 76 |
+
%8 = arith.addi %x0_7, %7 : tensor<64x1xi64> loc(#loc39)
|
| 77 |
+
%9 = tt.splat %out_ptr1 : !tt.ptr<i32> -> tensor<64x1x!tt.ptr<i32>> loc(#loc40)
|
| 78 |
+
%10 = tt.addptr %9, %8 : tensor<64x1x!tt.ptr<i32>>, tensor<64x1xi64> loc(#loc40)
|
| 79 |
+
tt.store %10, %tmp5, %xmask_4 : tensor<64x1x!tt.ptr<i32>> loc(#loc41)
|
| 80 |
+
tt.return loc(#loc42)
|
| 81 |
+
} loc(#loc)
|
| 82 |
+
} loc(#loc)
|
| 83 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":30:40)
|
| 84 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":28:43)
|
| 85 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":21:28)
|
| 86 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":21:33)
|
| 87 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":22:36)
|
| 88 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":22:44)
|
| 89 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":22:23)
|
| 90 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":23:21)
|
| 91 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":24:27)
|
| 92 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":24:37)
|
| 93 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":26:19)
|
| 94 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":27:19)
|
| 95 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":31:31)
|
| 96 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":32:29)
|
| 97 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:43)
|
| 98 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:39)
|
| 99 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:54)
|
| 100 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:58)
|
| 101 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:50)
|
| 102 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:34)
|
| 103 |
+
#loc22 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:73)
|
| 104 |
+
#loc23 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":36:63)
|
| 105 |
+
#loc24 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":37:23)
|
| 106 |
+
#loc25 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":39:23)
|
| 107 |
+
#loc26 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":40:48)
|
| 108 |
+
#loc27 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":40:8)
|
| 109 |
+
#loc28 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:36)
|
| 110 |
+
#loc30 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:15)
|
| 111 |
+
#loc31 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":41:28)
|
| 112 |
+
#loc32 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":42:19)
|
| 113 |
+
#loc33 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:49)
|
| 114 |
+
#loc34 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:75)
|
| 115 |
+
#loc35 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:66)
|
| 116 |
+
#loc36 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:57)
|
| 117 |
+
#loc37 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:41)
|
| 118 |
+
#loc38 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:34)
|
| 119 |
+
#loc39 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:30)
|
| 120 |
+
#loc40 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:25)
|
| 121 |
+
#loc41 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:88)
|
| 122 |
+
#loc42 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/wh/cwhgu2bzfksecbm4int3okkh3ph54kdms6zfad7ngpfzq6h2owww.py":43:4)
|
| 123 |
+
#loc49 = loc("_tmp3"(#loc3))
|
| 124 |
+
#loc50 = loc("xoffset"(#loc4))
|
| 125 |
+
#loc51 = loc("xoffset"(#loc5))
|
| 126 |
+
#loc52 = loc("xindex"(#loc6))
|
| 127 |
+
#loc53 = loc("xindex"(#loc7))
|
| 128 |
+
#loc54 = loc("xindex"(#loc8))
|
| 129 |
+
#loc55 = loc("xmask"(#loc9))
|
| 130 |
+
#loc56 = loc("r0_base"(#loc10))
|
| 131 |
+
#loc57 = loc("r0_base"(#loc11))
|
| 132 |
+
#loc58 = loc("x0"(#loc12))
|
| 133 |
+
#loc59 = loc("x1"(#loc13))
|
| 134 |
+
#loc60 = loc("_tmp3"(#loc2))
|
| 135 |
+
#loc61 = loc("r0_index"(#loc14))
|
| 136 |
+
#loc62 = loc("r0_mask"(#loc15))
|
| 137 |
+
#loc63 = loc("tmp0"(#loc16))
|
| 138 |
+
#loc64 = loc("tmp0"(#loc17))
|
| 139 |
+
#loc65 = loc("tmp0"(#loc18))
|
| 140 |
+
#loc66 = loc("tmp0"(#loc19))
|
| 141 |
+
#loc67 = loc("tmp0"(#loc20))
|
| 142 |
+
#loc68 = loc("tmp0"(#loc21))
|
| 143 |
+
#loc69 = loc("tmp0"(#loc22))
|
| 144 |
+
#loc70 = loc("tmp0"(#loc23))
|
| 145 |
+
#loc71 = loc("tmp1"(#loc24))
|
| 146 |
+
#loc72 = loc("tmp4"(#loc25))
|
| 147 |
+
#loc73 = loc("_tmp3"(#loc26))
|
| 148 |
+
#loc75 = loc("tmp3"(#loc31))
|
| 149 |
+
#loc76 = loc("tmp5"(#loc32))
|
| 150 |
+
#loc77 = loc(fused[#loc36, #loc37])
|
| 151 |
+
#loc78 = loc(callsite(#loc28 at #loc74))
|
| 152 |
+
#loc80 = loc(callsite(#loc30 at #loc78))
|
SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/__grp__triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"child_paths": {"triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.source": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.source", "triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ttir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ttir", "triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ttgir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ttgir", "triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.llir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.llir", "triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ptx": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ptx", "triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.cubin": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.cubin", "triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.json": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.json"}}
|
SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.cubin
ADDED
|
Binary file (86.4 kB). View file
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"hash": "e50c6a987142dbb9ad0869ff13cb48e2cad8442b31ebedd676f284f006fc5f0a", "target": {"backend": "cuda", "arch": 90, "warp_size": 32}, "num_warps": 4, "num_ctas": 1, "num_stages": 1, "warp_size": 32, "maxnreg": null, "cluster_dims": [1, 1, 1], "ptx_version": null, "ptx_options": null, "ir_override": null, "enable_fp_fusion": true, "launch_cooperative_grid": false, "launch_pdl": false, "supported_fp8_dtypes": ["fp8e4b15", "fp8e4nv", "fp8e5"], "deprecated_fp8_dot_operand_dtypes": ["fp8e4b15"], "default_dot_input_precision": "tf32", "allowed_dot_input_precisions": ["tf32", "tf32x3", "ieee"], "max_num_imprecise_acc_default": 1073741824, "extern_libs": [["libdevice", "/workspace/specforge/lib/python3.11/site-packages/triton/backends/nvidia/lib/libdevice.10.bc"]], "debug": true, "backend_name": "cuda", "sanitize_overflow": false, "arch": "sm90", "instrumentation_mode": "", "triton_version": "3.5.1", "tensordesc_meta": [], "shared": 2048, "tmem_size": 0, "global_scratch_size": 0, "global_scratch_align": 1, "profile_scratch_size": 0, "profile_scratch_align": 1, "name": "triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3"}
|
SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.llir
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ptx
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.source
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ttgir
ADDED
|
@@ -0,0 +1,841 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#blocked = #ttg.blocked<{sizePerThread = [1, 1], threadsPerWarp = [32, 1], warpsPerCTA = [1, 4], order = [0, 1]}>
|
| 2 |
+
#blocked1 = #ttg.blocked<{sizePerThread = [1, 4], threadsPerWarp = [8, 4], warpsPerCTA = [4, 1], order = [1, 0]}>
|
| 3 |
+
#linear = #ttg.linear<{register = [[0, 4], [0, 8]], lane = [[1, 0], [2, 0], [4, 0], [8, 0], [16, 0]], warp = [[0, 1], [0, 2]], block = []}>
|
| 4 |
+
#linear1 = #ttg.linear<{register = [[2, 0, 0], [4, 0, 0]], lane = [[8, 0, 0], [16, 0, 0], [32, 0, 0], [64, 0, 0], [128, 0, 0]], warp = [[0, 1, 0], [1, 0, 0]], block = []}>
|
| 5 |
+
#linear2 = #ttg.linear<{register = [[1, 0, 0], [2, 0, 0]], lane = [[4, 0, 0], [8, 0, 0], [16, 0, 0], [32, 0, 0], [64, 0, 0]], warp = [[0, 0, 1], [0, 1, 0]], block = []}>
|
| 6 |
+
#linear3 = #ttg.linear<{register = [[0, 1, 0], [1, 0, 0]], lane = [[2, 0, 0], [4, 0, 0], [8, 0, 0], [16, 0, 0], [32, 0, 0]], warp = [[0, 0, 1], [0, 0, 2]], block = []}>
|
| 7 |
+
#linear4 = #ttg.linear<{register = [[0, 0, 4], [0, 1, 0]], lane = [[1, 0, 0], [2, 0, 0], [4, 0, 0], [8, 0, 0], [16, 0, 0]], warp = [[0, 0, 1], [0, 0, 2]], block = []}>
|
| 8 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":18:0)
|
| 9 |
+
#loc1 = loc(unknown)
|
| 10 |
+
#loc19 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":662:12)
|
| 11 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":41:67)
|
| 12 |
+
#loc24 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":634:73)
|
| 13 |
+
#loc28 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":538:51)
|
| 14 |
+
#loc33 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":539:53)
|
| 15 |
+
#loc42 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":548:50)
|
| 16 |
+
#loc47 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":551:51)
|
| 17 |
+
#loc67 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":45:26)
|
| 18 |
+
#loc77 = loc("in_ptr0"(#loc))
|
| 19 |
+
#loc78 = loc("out_ptr2"(#loc))
|
| 20 |
+
#loc79 = loc("out_ptr3"(#loc))
|
| 21 |
+
#loc80 = loc("xnumel"(#loc))
|
| 22 |
+
#loc81 = loc("r0_numel"(#loc))
|
| 23 |
+
#loc99 = loc(callsite(#loc19 at #loc20))
|
| 24 |
+
#loc105 = loc("ileft"(#loc28))
|
| 25 |
+
#loc109 = loc("iright"(#loc33))
|
| 26 |
+
#loc118 = loc("left_idx"(#loc42))
|
| 27 |
+
#loc123 = loc("right_idx"(#loc47))
|
| 28 |
+
#loc143 = loc("tmp11"(#loc67))
|
| 29 |
+
#loc149 = loc(callsite(#loc24 at #loc99))
|
| 30 |
+
#loc153 = loc(callsite(#loc1 at #loc143))
|
| 31 |
+
#loc157 = loc(callsite(#loc105 at #loc149))
|
| 32 |
+
#loc161 = loc(callsite(#loc109 at #loc149))
|
| 33 |
+
#loc169 = loc(callsite(#loc118 at #loc149))
|
| 34 |
+
#loc174 = loc(callsite(#loc123 at #loc149))
|
| 35 |
+
#loc194 = loc(callsite(#loc1 at #loc157))
|
| 36 |
+
#loc196 = loc(callsite(#loc1 at #loc161))
|
| 37 |
+
#loc199 = loc(callsite(#loc1 at #loc169))
|
| 38 |
+
#loc202 = loc(callsite(#loc1 at #loc174))
|
| 39 |
+
module attributes {"ttg.num-ctas" = 1 : i32, "ttg.num-warps" = 4 : i32, ttg.target = "cuda:90", "ttg.threads-per-warp" = 32 : i32} {
|
| 40 |
+
tt.func public @triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3(%in_ptr0: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %out_ptr2: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("out_ptr2"(#loc)), %out_ptr3: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("out_ptr3"(#loc)), %xnumel: i32 {tt.divisibility = 16 : i32} loc("xnumel"(#loc)), %r0_numel: i32 {tt.divisibility = 16 : i32} loc("r0_numel"(#loc))) attributes {noinline = false} {
|
| 41 |
+
%cst = arith.constant dense<0> : tensor<32x16xi32, #linear> loc(#loc1)
|
| 42 |
+
%cst_0 = arith.constant dense<0> : tensor<32x16xi64, #blocked> loc(#loc1)
|
| 43 |
+
%c32_i32 = arith.constant 32 : i32 loc(#loc1)
|
| 44 |
+
%cst_1 = arith.constant dense<128> : tensor<32x1xi32, #blocked> loc(#loc1)
|
| 45 |
+
%cst_2 = arith.constant dense<128> : tensor<32x1xi32, #blocked1> loc(#loc1)
|
| 46 |
+
%cst_3 = arith.constant dense<16> : tensor<32x1xi32, #blocked> loc(#loc1)
|
| 47 |
+
%cst_4 = arith.constant dense<16> : tensor<32x1xi32, #blocked1> loc(#loc1)
|
| 48 |
+
%cst_5 = arith.constant dense<17> : tensor<1x16xi32, #blocked> loc(#loc1)
|
| 49 |
+
%cst_6 = arith.constant dense<272> : tensor<32x1xi32, #blocked> loc(#loc1)
|
| 50 |
+
%cst_7 = arith.constant dense<1> : tensor<1x2x1xi32, #linear1> loc(#loc1)
|
| 51 |
+
%cst_8 = arith.constant dense<1> : tensor<1x2x1xi32, #linear2> loc(#loc1)
|
| 52 |
+
%cst_9 = arith.constant dense<1> : tensor<1x2x1xi32, #linear3> loc(#loc1)
|
| 53 |
+
%cst_10 = arith.constant dense<1> : tensor<1x2x1xi32, #linear4> loc(#loc1)
|
| 54 |
+
%cst_11 = arith.constant dense<0> : tensor<32x16xi32, #blocked> loc(#loc1)
|
| 55 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc82)
|
| 56 |
+
%xoffset_12 = arith.muli %xoffset, %c32_i32 : i32 loc(#loc83)
|
| 57 |
+
%xindex = tt.make_range {end = 32 : i32, start = 0 : i32} : tensor<32xi32, #ttg.slice<{dim = 1, parent = #blocked}>> loc(#loc84)
|
| 58 |
+
%xindex_13 = tt.make_range {end = 32 : i32, start = 0 : i32} : tensor<32xi32, #ttg.slice<{dim = 1, parent = #blocked1}>> loc(#loc84)
|
| 59 |
+
%xindex_14 = tt.expand_dims %xindex {axis = 1 : i32} : tensor<32xi32, #ttg.slice<{dim = 1, parent = #blocked}>> -> tensor<32x1xi32, #blocked> loc(#loc84)
|
| 60 |
+
%xindex_15 = tt.expand_dims %xindex_13 {axis = 1 : i32} : tensor<32xi32, #ttg.slice<{dim = 1, parent = #blocked1}>> -> tensor<32x1xi32, #blocked1> loc(#loc84)
|
| 61 |
+
%xindex_16 = tt.splat %xoffset_12 : i32 -> tensor<32x1xi32, #blocked> loc(#loc85)
|
| 62 |
+
%xindex_17 = tt.splat %xoffset_12 : i32 -> tensor<32x1xi32, #blocked1> loc(#loc85)
|
| 63 |
+
%xindex_18 = arith.addi %xindex_16, %xindex_14 : tensor<32x1xi32, #blocked> loc(#loc85)
|
| 64 |
+
%xindex_19 = arith.addi %xindex_17, %xindex_15 : tensor<32x1xi32, #blocked1> loc(#loc85)
|
| 65 |
+
%xmask = arith.cmpi slt, %xindex_18, %cst_1 : tensor<32x1xi32, #blocked> loc(#loc86)
|
| 66 |
+
%xmask_20 = arith.cmpi slt, %xindex_19, %cst_2 : tensor<32x1xi32, #blocked1> loc(#loc86)
|
| 67 |
+
%r0_index = tt.make_range {end = 16 : i32, start = 0 : i32} : tensor<16xi32, #ttg.slice<{dim = 0, parent = #blocked}>> loc(#loc87)
|
| 68 |
+
%r0_index_21 = tt.make_range {end = 16 : i32, start = 0 : i32} : tensor<16xi32, #ttg.slice<{dim = 0, parent = #linear}>> loc(#loc87)
|
| 69 |
+
%r0_index_22 = tt.make_range {end = 16 : i32, start = 0 : i32} : tensor<16xi32, #ttg.slice<{dim = 0, parent = #blocked1}>> loc(#loc87)
|
| 70 |
+
%r0_index_23 = tt.expand_dims %r0_index {axis = 0 : i32} : tensor<16xi32, #ttg.slice<{dim = 0, parent = #blocked}>> -> tensor<1x16xi32, #blocked> loc(#loc87)
|
| 71 |
+
%r0_index_24 = tt.expand_dims %r0_index_21 {axis = 0 : i32} : tensor<16xi32, #ttg.slice<{dim = 0, parent = #linear}>> -> tensor<1x16xi32, #linear> loc(#loc87)
|
| 72 |
+
%r0_index_25 = tt.expand_dims %r0_index_22 {axis = 0 : i32} : tensor<16xi32, #ttg.slice<{dim = 0, parent = #blocked1}>> -> tensor<1x16xi32, #blocked1> loc(#loc87)
|
| 73 |
+
%x0 = arith.remsi %xindex_18, %cst_3 : tensor<32x1xi32, #blocked> loc(#loc88)
|
| 74 |
+
%x1 = arith.divsi %xindex_18, %cst_3 : tensor<32x1xi32, #blocked> loc(#loc89)
|
| 75 |
+
%tmp0 = arith.muli %r0_index_23, %cst_5 : tensor<1x16xi32, #blocked> loc(#loc90)
|
| 76 |
+
%tmp0_26 = tt.broadcast %x0 : tensor<32x1xi32, #blocked> -> tensor<32x16xi32, #blocked> loc(#loc91)
|
| 77 |
+
%tmp0_27 = tt.broadcast %tmp0 : tensor<1x16xi32, #blocked> -> tensor<32x16xi32, #blocked> loc(#loc91)
|
| 78 |
+
%tmp0_28 = arith.addi %tmp0_26, %tmp0_27 : tensor<32x16xi32, #blocked> loc(#loc91)
|
| 79 |
+
%tmp0_29 = arith.muli %x1, %cst_6 : tensor<32x1xi32, #blocked> loc(#loc92)
|
| 80 |
+
%tmp0_30 = tt.broadcast %tmp0_29 : tensor<32x1xi32, #blocked> -> tensor<32x16xi32, #blocked> loc(#loc93)
|
| 81 |
+
%tmp0_31 = arith.addi %tmp0_28, %tmp0_30 : tensor<32x16xi32, #blocked> loc(#loc93)
|
| 82 |
+
%tmp0_32 = tt.splat %in_ptr0 : !tt.ptr<i32> -> tensor<32x16x!tt.ptr<i32>, #blocked> loc(#loc94)
|
| 83 |
+
%tmp0_33 = tt.addptr %tmp0_32, %tmp0_31 : tensor<32x16x!tt.ptr<i32>, #blocked>, tensor<32x16xi32, #blocked> loc(#loc94)
|
| 84 |
+
%tmp0_34 = tt.broadcast %xmask : tensor<32x1xi1, #blocked> -> tensor<32x16xi1, #blocked> loc(#loc95)
|
| 85 |
+
%tmp0_35 = tt.broadcast %xmask_20 : tensor<32x1xi1, #blocked1> -> tensor<32x16xi1, #blocked1> loc(#loc95)
|
| 86 |
+
%tmp0_36 = tt.load %tmp0_33, %tmp0_34, %cst_11 : tensor<32x16x!tt.ptr<i32>, #blocked> loc(#loc95)
|
| 87 |
+
%tmp2 = arith.trunci %r0_index_24 : tensor<1x16xi32, #linear> to tensor<1x16xi16, #linear> loc(#loc96)
|
| 88 |
+
%tmp4 = tt.broadcast %tmp2 : tensor<1x16xi16, #linear> -> tensor<32x16xi16, #linear> loc(#loc97)
|
| 89 |
+
%flip = tt.make_range {end = 2 : i32, start = 0 : i32} : tensor<2xi32, #ttg.slice<{dim = 0, parent = #ttg.slice<{dim = 2, parent = #linear2}>}>> loc(#loc146)
|
| 90 |
+
%flip_37 = tt.make_range {end = 2 : i32, start = 0 : i32} : tensor<2xi32, #ttg.slice<{dim = 0, parent = #ttg.slice<{dim = 2, parent = #linear1}>}>> loc(#loc146)
|
| 91 |
+
%flip_38 = tt.make_range {end = 2 : i32, start = 0 : i32} : tensor<2xi32, #ttg.slice<{dim = 0, parent = #ttg.slice<{dim = 2, parent = #linear3}>}>> loc(#loc146)
|
| 92 |
+
%flip_39 = tt.make_range {end = 2 : i32, start = 0 : i32} : tensor<2xi32, #ttg.slice<{dim = 0, parent = #ttg.slice<{dim = 2, parent = #linear4}>}>> loc(#loc146)
|
| 93 |
+
%flip_40 = tt.expand_dims %flip {axis = 0 : i32} : tensor<2xi32, #ttg.slice<{dim = 0, parent = #ttg.slice<{dim = 2, parent = #linear2}>}>> -> tensor<1x2xi32, #ttg.slice<{dim = 2, parent = #linear2}>> loc(#loc146)
|
| 94 |
+
%flip_41 = tt.expand_dims %flip_37 {axis = 0 : i32} : tensor<2xi32, #ttg.slice<{dim = 0, parent = #ttg.slice<{dim = 2, parent = #linear1}>}>> -> tensor<1x2xi32, #ttg.slice<{dim = 2, parent = #linear1}>> loc(#loc146)
|
| 95 |
+
%flip_42 = tt.expand_dims %flip_38 {axis = 0 : i32} : tensor<2xi32, #ttg.slice<{dim = 0, parent = #ttg.slice<{dim = 2, parent = #linear3}>}>> -> tensor<1x2xi32, #ttg.slice<{dim = 2, parent = #linear3}>> loc(#loc146)
|
| 96 |
+
%flip_43 = tt.expand_dims %flip_39 {axis = 0 : i32} : tensor<2xi32, #ttg.slice<{dim = 0, parent = #ttg.slice<{dim = 2, parent = #linear4}>}>> -> tensor<1x2xi32, #ttg.slice<{dim = 2, parent = #linear4}>> loc(#loc146)
|
| 97 |
+
%flip_44 = tt.expand_dims %flip_40 {axis = 2 : i32} : tensor<1x2xi32, #ttg.slice<{dim = 2, parent = #linear2}>> -> tensor<1x2x1xi32, #linear2> loc(#loc146)
|
| 98 |
+
%flip_45 = tt.expand_dims %flip_41 {axis = 2 : i32} : tensor<1x2xi32, #ttg.slice<{dim = 2, parent = #linear1}>> -> tensor<1x2x1xi32, #linear1> loc(#loc146)
|
| 99 |
+
%flip_46 = tt.expand_dims %flip_42 {axis = 2 : i32} : tensor<1x2xi32, #ttg.slice<{dim = 2, parent = #linear3}>> -> tensor<1x2x1xi32, #linear3> loc(#loc146)
|
| 100 |
+
%flip_47 = tt.expand_dims %flip_43 {axis = 2 : i32} : tensor<1x2xi32, #ttg.slice<{dim = 2, parent = #linear4}>> -> tensor<1x2x1xi32, #linear4> loc(#loc146)
|
| 101 |
+
%flip_48 = tt.broadcast %flip_44 : tensor<1x2x1xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc147)
|
| 102 |
+
%flip_49 = tt.reshape %flip_48 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #blocked> loc(#loc148)
|
| 103 |
+
%flip_50 = tt.reshape %flip_48 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc148)
|
| 104 |
+
%y = tt.reshape %tmp0_36 : tensor<32x16xi32, #blocked> -> tensor<256x2x1xi32, #linear1> loc(#loc154)
|
| 105 |
+
%left_mask = arith.subi %cst_7, %flip_45 : tensor<1x2x1xi32, #linear1> loc(#loc155)
|
| 106 |
+
%left_mask_51 = arith.subi %cst_8, %flip_44 : tensor<1x2x1xi32, #linear2> loc(#loc155)
|
| 107 |
+
%left_mask_52 = arith.subi %cst_9, %flip_46 : tensor<1x2x1xi32, #linear3> loc(#loc155)
|
| 108 |
+
%left_mask_53 = arith.subi %cst_10, %flip_47 : tensor<1x2x1xi32, #linear4> loc(#loc155)
|
| 109 |
+
%ileft = tt.broadcast %left_mask : tensor<1x2x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc156)
|
| 110 |
+
%ileft_54 = arith.muli %y, %ileft : tensor<256x2x1xi32, #linear1> loc(#loc156)
|
| 111 |
+
%ileft_55 = "tt.reduce"(%ileft_54) <{axis = 1 : i32}> ({
|
| 112 |
+
^bb0(%ileft_419: i32 loc(callsite(#loc1 at #loc157)), %ileft_420: i32 loc(callsite(#loc1 at #loc157))):
|
| 113 |
+
%ileft_421 = arith.addi %ileft_419, %ileft_420 : i32 loc(#loc203)
|
| 114 |
+
tt.reduce.return %ileft_421 : i32 loc(#loc193)
|
| 115 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc193)
|
| 116 |
+
%ileft_56 = tt.expand_dims %ileft_55 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc158)
|
| 117 |
+
%ileft_57 = tt.broadcast %ileft_56 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc159)
|
| 118 |
+
%iright = tt.broadcast %flip_45 : tensor<1x2x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc160)
|
| 119 |
+
%iright_58 = arith.muli %y, %iright : tensor<256x2x1xi32, #linear1> loc(#loc160)
|
| 120 |
+
%iright_59 = "tt.reduce"(%iright_58) <{axis = 1 : i32}> ({
|
| 121 |
+
^bb0(%iright_419: i32 loc(callsite(#loc1 at #loc161)), %iright_420: i32 loc(callsite(#loc1 at #loc161))):
|
| 122 |
+
%iright_421 = arith.addi %iright_419, %iright_420 : i32 loc(#loc204)
|
| 123 |
+
tt.reduce.return %iright_421 : i32 loc(#loc195)
|
| 124 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc195)
|
| 125 |
+
%iright_60 = tt.expand_dims %iright_59 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc162)
|
| 126 |
+
%iright_61 = tt.broadcast %iright_60 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc163)
|
| 127 |
+
%ileft_62 = tt.reshape %ileft_57 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #blocked> loc(#loc164)
|
| 128 |
+
%ileft_63 = tt.reshape %ileft_57 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc164)
|
| 129 |
+
%iright_64 = tt.reshape %iright_61 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #blocked> loc(#loc165)
|
| 130 |
+
%iright_65 = tt.reshape %iright_61 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc165)
|
| 131 |
+
%y_idx = tt.reshape %tmp4 : tensor<32x16xi16, #linear> -> tensor<256x2x1xi16, #linear1> loc(#loc166)
|
| 132 |
+
%left_idx = arith.trunci %left_mask : tensor<1x2x1xi32, #linear1> to tensor<1x2x1xi16, #linear1> loc(#loc167)
|
| 133 |
+
%left_idx_66 = tt.broadcast %left_idx : tensor<1x2x1xi16, #linear1> -> tensor<256x2x1xi16, #linear1> loc(#loc168)
|
| 134 |
+
%left_idx_67 = arith.muli %y_idx, %left_idx_66 : tensor<256x2x1xi16, #linear1> loc(#loc168)
|
| 135 |
+
%input = arith.extsi %left_idx_67 : tensor<256x2x1xi16, #linear1> to tensor<256x2x1xi32, #linear1> loc(#loc197)
|
| 136 |
+
%left_idx_68 = "tt.reduce"(%input) <{axis = 1 : i32}> ({
|
| 137 |
+
^bb0(%left_idx_419: i32 loc(callsite(#loc1 at #loc169)), %left_idx_420: i32 loc(callsite(#loc1 at #loc169))):
|
| 138 |
+
%left_idx_421 = arith.addi %left_idx_419, %left_idx_420 : i32 loc(#loc205)
|
| 139 |
+
tt.reduce.return %left_idx_421 : i32 loc(#loc198)
|
| 140 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc198)
|
| 141 |
+
%left_idx_69 = tt.expand_dims %left_idx_68 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc170)
|
| 142 |
+
%left_idx_70 = tt.broadcast %left_idx_69 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc171)
|
| 143 |
+
%right_idx = arith.trunci %flip_45 : tensor<1x2x1xi32, #linear1> to tensor<1x2x1xi16, #linear1> loc(#loc172)
|
| 144 |
+
%right_idx_71 = tt.broadcast %right_idx : tensor<1x2x1xi16, #linear1> -> tensor<256x2x1xi16, #linear1> loc(#loc173)
|
| 145 |
+
%right_idx_72 = arith.muli %y_idx, %right_idx_71 : tensor<256x2x1xi16, #linear1> loc(#loc173)
|
| 146 |
+
%input_73 = arith.extsi %right_idx_72 : tensor<256x2x1xi16, #linear1> to tensor<256x2x1xi32, #linear1> loc(#loc200)
|
| 147 |
+
%right_idx_74 = "tt.reduce"(%input_73) <{axis = 1 : i32}> ({
|
| 148 |
+
^bb0(%right_idx_419: i32 loc(callsite(#loc1 at #loc174)), %right_idx_420: i32 loc(callsite(#loc1 at #loc174))):
|
| 149 |
+
%right_idx_421 = arith.addi %right_idx_419, %right_idx_420 : i32 loc(#loc206)
|
| 150 |
+
tt.reduce.return %right_idx_421 : i32 loc(#loc201)
|
| 151 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc201)
|
| 152 |
+
%right_idx_75 = tt.expand_dims %right_idx_74 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc175)
|
| 153 |
+
%right_idx_76 = tt.broadcast %right_idx_75 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc176)
|
| 154 |
+
%left_idx_77 = tt.reshape %left_idx_70 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #blocked> loc(#loc177)
|
| 155 |
+
%left_idx_78 = tt.reshape %left_idx_70 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc177)
|
| 156 |
+
%right_idx_79 = tt.reshape %right_idx_76 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #blocked> loc(#loc178)
|
| 157 |
+
%right_idx_80 = tt.reshape %right_idx_76 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc178)
|
| 158 |
+
%cond = arith.cmpi slt, %ileft_62, %iright_64 : tensor<32x16xi32, #blocked> loc(#loc179)
|
| 159 |
+
%cond_81 = arith.cmpi slt, %ileft_63, %iright_65 : tensor<32x16xi32, #linear> loc(#loc179)
|
| 160 |
+
%eq = arith.cmpi eq, %ileft_62, %iright_64 : tensor<32x16xi32, #blocked> loc(#loc180)
|
| 161 |
+
%eq_82 = arith.cmpi eq, %ileft_63, %iright_65 : tensor<32x16xi32, #linear> loc(#loc180)
|
| 162 |
+
%cond_83 = arith.cmpi sgt, %left_idx_77, %right_idx_79 : tensor<32x16xi32, #blocked> loc(#loc181)
|
| 163 |
+
%cond_84 = arith.cmpi sgt, %left_idx_78, %right_idx_80 : tensor<32x16xi32, #linear> loc(#loc181)
|
| 164 |
+
%cond_85 = arith.andi %eq, %cond_83 : tensor<32x16xi1, #blocked> loc(#loc182)
|
| 165 |
+
%cond_86 = arith.andi %eq_82, %cond_84 : tensor<32x16xi1, #linear> loc(#loc182)
|
| 166 |
+
%cond_87 = arith.ori %cond, %cond_85 : tensor<32x16xi1, #blocked> loc(#loc183)
|
| 167 |
+
%cond_88 = arith.ori %cond_81, %cond_86 : tensor<32x16xi1, #linear> loc(#loc183)
|
| 168 |
+
%cond_89 = arith.extui %cond_87 : tensor<32x16xi1, #blocked> to tensor<32x16xi32, #blocked> loc(#loc184)
|
| 169 |
+
%cond_90 = arith.extui %cond_88 : tensor<32x16xi1, #linear> to tensor<32x16xi32, #linear> loc(#loc184)
|
| 170 |
+
%cond_91 = arith.xori %cond_89, %flip_49 : tensor<32x16xi32, #blocked> loc(#loc184)
|
| 171 |
+
%cond_92 = arith.xori %cond_90, %flip_50 : tensor<32x16xi32, #linear> loc(#loc184)
|
| 172 |
+
%cond_93 = arith.cmpi ne, %cond_91, %cst_11 : tensor<32x16xi32, #blocked> loc(#loc185)
|
| 173 |
+
%cond_94 = arith.cmpi ne, %cond_92, %cst : tensor<32x16xi32, #linear> loc(#loc185)
|
| 174 |
+
%ret = arith.xori %ileft_62, %iright_64 : tensor<32x16xi32, #blocked> loc(#loc186)
|
| 175 |
+
%ret_95 = arith.select %cond_93, %ret, %cst_11 : tensor<32x16xi1, #blocked>, tensor<32x16xi32, #blocked> loc(#loc187)
|
| 176 |
+
%ret_96 = arith.xori %tmp0_36, %ret_95 : tensor<32x16xi32, #blocked> loc(#loc188)
|
| 177 |
+
%ret_97 = ttg.convert_layout %ret_96 : tensor<32x16xi32, #blocked> -> tensor<32x16xi32, #linear> loc(#loc188)
|
| 178 |
+
%new_idxs = arith.xori %left_idx_78, %right_idx_80 : tensor<32x16xi32, #linear> loc(#loc189)
|
| 179 |
+
%new_idxs_98 = arith.select %cond_94, %new_idxs, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc190)
|
| 180 |
+
%new_idxs_99 = arith.extsi %tmp2 : tensor<1x16xi16, #linear> to tensor<1x16xi32, #linear> loc(#loc191)
|
| 181 |
+
%new_idxs_100 = tt.broadcast %new_idxs_99 : tensor<1x16xi32, #linear> -> tensor<32x16xi32, #linear> loc(#loc191)
|
| 182 |
+
%new_idxs_101 = arith.xori %new_idxs_100, %new_idxs_98 : tensor<32x16xi32, #linear> loc(#loc191)
|
| 183 |
+
%flip_102 = tt.broadcast %flip_46 : tensor<1x2x1xi32, #linear3> -> tensor<64x2x4xi32, #linear3> loc(#loc147)
|
| 184 |
+
%flip_103 = tt.reshape %flip_102 : tensor<64x2x4xi32, #linear3> -> tensor<32x16xi32, #linear> loc(#loc148)
|
| 185 |
+
%y_104 = tt.reshape %ret_96 : tensor<32x16xi32, #blocked> -> tensor<128x2x2xi32, #linear2> loc(#loc154)
|
| 186 |
+
%ileft_105 = tt.broadcast %left_mask_51 : tensor<1x2x1xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc156)
|
| 187 |
+
%ileft_106 = arith.muli %y_104, %ileft_105 : tensor<128x2x2xi32, #linear2> loc(#loc156)
|
| 188 |
+
%ileft_107 = "tt.reduce"(%ileft_106) <{axis = 1 : i32}> ({
|
| 189 |
+
^bb0(%ileft_419: i32 loc(callsite(#loc1 at #loc157)), %ileft_420: i32 loc(callsite(#loc1 at #loc157))):
|
| 190 |
+
%ileft_421 = arith.addi %ileft_419, %ileft_420 : i32 loc(#loc203)
|
| 191 |
+
tt.reduce.return %ileft_421 : i32 loc(#loc193)
|
| 192 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc193)
|
| 193 |
+
%ileft_108 = tt.expand_dims %ileft_107 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc158)
|
| 194 |
+
%ileft_109 = tt.broadcast %ileft_108 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc159)
|
| 195 |
+
%iright_110 = arith.muli %y_104, %flip_48 : tensor<128x2x2xi32, #linear2> loc(#loc160)
|
| 196 |
+
%iright_111 = "tt.reduce"(%iright_110) <{axis = 1 : i32}> ({
|
| 197 |
+
^bb0(%iright_419: i32 loc(callsite(#loc1 at #loc161)), %iright_420: i32 loc(callsite(#loc1 at #loc161))):
|
| 198 |
+
%iright_421 = arith.addi %iright_419, %iright_420 : i32 loc(#loc204)
|
| 199 |
+
tt.reduce.return %iright_421 : i32 loc(#loc195)
|
| 200 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc195)
|
| 201 |
+
%iright_112 = tt.expand_dims %iright_111 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc162)
|
| 202 |
+
%iright_113 = tt.broadcast %iright_112 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc163)
|
| 203 |
+
%ileft_114 = tt.reshape %ileft_109 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc164)
|
| 204 |
+
%iright_115 = tt.reshape %iright_113 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc165)
|
| 205 |
+
%y_idx_116 = tt.reshape %new_idxs_101 : tensor<32x16xi32, #linear> -> tensor<128x2x2xi32, #linear2> loc(#loc166)
|
| 206 |
+
%left_idx_117 = arith.muli %y_idx_116, %ileft_105 : tensor<128x2x2xi32, #linear2> loc(#loc168)
|
| 207 |
+
%left_idx_118 = "tt.reduce"(%left_idx_117) <{axis = 1 : i32}> ({
|
| 208 |
+
^bb0(%left_idx_419: i32 loc(callsite(#loc1 at #loc169)), %left_idx_420: i32 loc(callsite(#loc1 at #loc169))):
|
| 209 |
+
%left_idx_421 = arith.addi %left_idx_419, %left_idx_420 : i32 loc(#loc205)
|
| 210 |
+
tt.reduce.return %left_idx_421 : i32 loc(#loc198)
|
| 211 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc198)
|
| 212 |
+
%left_idx_119 = tt.expand_dims %left_idx_118 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc170)
|
| 213 |
+
%left_idx_120 = tt.broadcast %left_idx_119 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc171)
|
| 214 |
+
%right_idx_121 = arith.muli %y_idx_116, %flip_48 : tensor<128x2x2xi32, #linear2> loc(#loc173)
|
| 215 |
+
%right_idx_122 = "tt.reduce"(%right_idx_121) <{axis = 1 : i32}> ({
|
| 216 |
+
^bb0(%right_idx_419: i32 loc(callsite(#loc1 at #loc174)), %right_idx_420: i32 loc(callsite(#loc1 at #loc174))):
|
| 217 |
+
%right_idx_421 = arith.addi %right_idx_419, %right_idx_420 : i32 loc(#loc206)
|
| 218 |
+
tt.reduce.return %right_idx_421 : i32 loc(#loc201)
|
| 219 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc201)
|
| 220 |
+
%right_idx_123 = tt.expand_dims %right_idx_122 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc175)
|
| 221 |
+
%right_idx_124 = tt.broadcast %right_idx_123 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc176)
|
| 222 |
+
%left_idx_125 = tt.reshape %left_idx_120 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc177)
|
| 223 |
+
%right_idx_126 = tt.reshape %right_idx_124 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc178)
|
| 224 |
+
%cond_127 = arith.cmpi slt, %ileft_114, %iright_115 : tensor<32x16xi32, #linear> loc(#loc179)
|
| 225 |
+
%eq_128 = arith.cmpi eq, %ileft_114, %iright_115 : tensor<32x16xi32, #linear> loc(#loc180)
|
| 226 |
+
%cond_129 = arith.cmpi sgt, %left_idx_125, %right_idx_126 : tensor<32x16xi32, #linear> loc(#loc181)
|
| 227 |
+
%cond_130 = arith.andi %eq_128, %cond_129 : tensor<32x16xi1, #linear> loc(#loc182)
|
| 228 |
+
%cond_131 = arith.ori %cond_127, %cond_130 : tensor<32x16xi1, #linear> loc(#loc183)
|
| 229 |
+
%cond_132 = arith.extui %cond_131 : tensor<32x16xi1, #linear> to tensor<32x16xi32, #linear> loc(#loc184)
|
| 230 |
+
%cond_133 = arith.xori %cond_132, %flip_103 : tensor<32x16xi32, #linear> loc(#loc184)
|
| 231 |
+
%cond_134 = arith.cmpi ne, %cond_133, %cst : tensor<32x16xi32, #linear> loc(#loc185)
|
| 232 |
+
%ret_135 = arith.xori %ileft_114, %iright_115 : tensor<32x16xi32, #linear> loc(#loc186)
|
| 233 |
+
%ret_136 = arith.select %cond_134, %ret_135, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc187)
|
| 234 |
+
%ret_137 = arith.xori %ret_97, %ret_136 : tensor<32x16xi32, #linear> loc(#loc188)
|
| 235 |
+
%new_idxs_138 = arith.xori %left_idx_125, %right_idx_126 : tensor<32x16xi32, #linear> loc(#loc189)
|
| 236 |
+
%new_idxs_139 = arith.select %cond_134, %new_idxs_138, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc190)
|
| 237 |
+
%new_idxs_140 = arith.xori %new_idxs_101, %new_idxs_139 : tensor<32x16xi32, #linear> loc(#loc191)
|
| 238 |
+
%y_141 = tt.reshape %ret_137 : tensor<32x16xi32, #linear> -> tensor<256x2x1xi32, #linear1> loc(#loc154)
|
| 239 |
+
%ileft_142 = arith.muli %y_141, %ileft : tensor<256x2x1xi32, #linear1> loc(#loc156)
|
| 240 |
+
%ileft_143 = "tt.reduce"(%ileft_142) <{axis = 1 : i32}> ({
|
| 241 |
+
^bb0(%ileft_419: i32 loc(callsite(#loc1 at #loc157)), %ileft_420: i32 loc(callsite(#loc1 at #loc157))):
|
| 242 |
+
%ileft_421 = arith.addi %ileft_419, %ileft_420 : i32 loc(#loc203)
|
| 243 |
+
tt.reduce.return %ileft_421 : i32 loc(#loc193)
|
| 244 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc193)
|
| 245 |
+
%ileft_144 = tt.expand_dims %ileft_143 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc158)
|
| 246 |
+
%ileft_145 = tt.broadcast %ileft_144 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc159)
|
| 247 |
+
%iright_146 = arith.muli %y_141, %iright : tensor<256x2x1xi32, #linear1> loc(#loc160)
|
| 248 |
+
%iright_147 = "tt.reduce"(%iright_146) <{axis = 1 : i32}> ({
|
| 249 |
+
^bb0(%iright_419: i32 loc(callsite(#loc1 at #loc161)), %iright_420: i32 loc(callsite(#loc1 at #loc161))):
|
| 250 |
+
%iright_421 = arith.addi %iright_419, %iright_420 : i32 loc(#loc204)
|
| 251 |
+
tt.reduce.return %iright_421 : i32 loc(#loc195)
|
| 252 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc195)
|
| 253 |
+
%iright_148 = tt.expand_dims %iright_147 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc162)
|
| 254 |
+
%iright_149 = tt.broadcast %iright_148 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc163)
|
| 255 |
+
%ileft_150 = tt.reshape %ileft_145 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc164)
|
| 256 |
+
%iright_151 = tt.reshape %iright_149 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc165)
|
| 257 |
+
%y_idx_152 = tt.reshape %new_idxs_140 : tensor<32x16xi32, #linear> -> tensor<256x2x1xi32, #linear1> loc(#loc166)
|
| 258 |
+
%left_idx_153 = arith.muli %y_idx_152, %ileft : tensor<256x2x1xi32, #linear1> loc(#loc168)
|
| 259 |
+
%left_idx_154 = "tt.reduce"(%left_idx_153) <{axis = 1 : i32}> ({
|
| 260 |
+
^bb0(%left_idx_419: i32 loc(callsite(#loc1 at #loc169)), %left_idx_420: i32 loc(callsite(#loc1 at #loc169))):
|
| 261 |
+
%left_idx_421 = arith.addi %left_idx_419, %left_idx_420 : i32 loc(#loc205)
|
| 262 |
+
tt.reduce.return %left_idx_421 : i32 loc(#loc198)
|
| 263 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc198)
|
| 264 |
+
%left_idx_155 = tt.expand_dims %left_idx_154 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc170)
|
| 265 |
+
%left_idx_156 = tt.broadcast %left_idx_155 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc171)
|
| 266 |
+
%right_idx_157 = arith.muli %y_idx_152, %iright : tensor<256x2x1xi32, #linear1> loc(#loc173)
|
| 267 |
+
%right_idx_158 = "tt.reduce"(%right_idx_157) <{axis = 1 : i32}> ({
|
| 268 |
+
^bb0(%right_idx_419: i32 loc(callsite(#loc1 at #loc174)), %right_idx_420: i32 loc(callsite(#loc1 at #loc174))):
|
| 269 |
+
%right_idx_421 = arith.addi %right_idx_419, %right_idx_420 : i32 loc(#loc206)
|
| 270 |
+
tt.reduce.return %right_idx_421 : i32 loc(#loc201)
|
| 271 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc201)
|
| 272 |
+
%right_idx_159 = tt.expand_dims %right_idx_158 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc175)
|
| 273 |
+
%right_idx_160 = tt.broadcast %right_idx_159 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc176)
|
| 274 |
+
%left_idx_161 = tt.reshape %left_idx_156 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc177)
|
| 275 |
+
%right_idx_162 = tt.reshape %right_idx_160 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc178)
|
| 276 |
+
%cond_163 = arith.cmpi slt, %ileft_150, %iright_151 : tensor<32x16xi32, #linear> loc(#loc179)
|
| 277 |
+
%eq_164 = arith.cmpi eq, %ileft_150, %iright_151 : tensor<32x16xi32, #linear> loc(#loc180)
|
| 278 |
+
%cond_165 = arith.cmpi sgt, %left_idx_161, %right_idx_162 : tensor<32x16xi32, #linear> loc(#loc181)
|
| 279 |
+
%cond_166 = arith.andi %eq_164, %cond_165 : tensor<32x16xi1, #linear> loc(#loc182)
|
| 280 |
+
%cond_167 = arith.ori %cond_163, %cond_166 : tensor<32x16xi1, #linear> loc(#loc183)
|
| 281 |
+
%cond_168 = arith.extui %cond_167 : tensor<32x16xi1, #linear> to tensor<32x16xi32, #linear> loc(#loc184)
|
| 282 |
+
%cond_169 = arith.xori %cond_168, %flip_103 : tensor<32x16xi32, #linear> loc(#loc184)
|
| 283 |
+
%cond_170 = arith.cmpi ne, %cond_169, %cst : tensor<32x16xi32, #linear> loc(#loc185)
|
| 284 |
+
%ret_171 = arith.xori %ileft_150, %iright_151 : tensor<32x16xi32, #linear> loc(#loc186)
|
| 285 |
+
%ret_172 = arith.select %cond_170, %ret_171, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc187)
|
| 286 |
+
%ret_173 = arith.xori %ret_137, %ret_172 : tensor<32x16xi32, #linear> loc(#loc188)
|
| 287 |
+
%new_idxs_174 = arith.xori %left_idx_161, %right_idx_162 : tensor<32x16xi32, #linear> loc(#loc189)
|
| 288 |
+
%new_idxs_175 = arith.select %cond_170, %new_idxs_174, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc190)
|
| 289 |
+
%new_idxs_176 = arith.xori %new_idxs_140, %new_idxs_175 : tensor<32x16xi32, #linear> loc(#loc191)
|
| 290 |
+
%flip_177 = tt.broadcast %flip_47 : tensor<1x2x1xi32, #linear4> -> tensor<32x2x8xi32, #linear4> loc(#loc147)
|
| 291 |
+
%flip_178 = tt.reshape %flip_177 : tensor<32x2x8xi32, #linear4> -> tensor<32x16xi32, #linear> loc(#loc148)
|
| 292 |
+
%y_179 = tt.reshape %ret_173 : tensor<32x16xi32, #linear> -> tensor<64x2x4xi32, #linear3> loc(#loc154)
|
| 293 |
+
%ileft_180 = tt.broadcast %left_mask_52 : tensor<1x2x1xi32, #linear3> -> tensor<64x2x4xi32, #linear3> loc(#loc156)
|
| 294 |
+
%ileft_181 = arith.muli %y_179, %ileft_180 : tensor<64x2x4xi32, #linear3> loc(#loc156)
|
| 295 |
+
%ileft_182 = "tt.reduce"(%ileft_181) <{axis = 1 : i32}> ({
|
| 296 |
+
^bb0(%ileft_419: i32 loc(callsite(#loc1 at #loc157)), %ileft_420: i32 loc(callsite(#loc1 at #loc157))):
|
| 297 |
+
%ileft_421 = arith.addi %ileft_419, %ileft_420 : i32 loc(#loc203)
|
| 298 |
+
tt.reduce.return %ileft_421 : i32 loc(#loc193)
|
| 299 |
+
}) : (tensor<64x2x4xi32, #linear3>) -> tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> loc(#loc193)
|
| 300 |
+
%ileft_183 = tt.expand_dims %ileft_182 {axis = 1 : i32} : tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> -> tensor<64x1x4xi32, #linear3> loc(#loc158)
|
| 301 |
+
%ileft_184 = tt.broadcast %ileft_183 : tensor<64x1x4xi32, #linear3> -> tensor<64x2x4xi32, #linear3> loc(#loc159)
|
| 302 |
+
%iright_185 = arith.muli %y_179, %flip_102 : tensor<64x2x4xi32, #linear3> loc(#loc160)
|
| 303 |
+
%iright_186 = "tt.reduce"(%iright_185) <{axis = 1 : i32}> ({
|
| 304 |
+
^bb0(%iright_419: i32 loc(callsite(#loc1 at #loc161)), %iright_420: i32 loc(callsite(#loc1 at #loc161))):
|
| 305 |
+
%iright_421 = arith.addi %iright_419, %iright_420 : i32 loc(#loc204)
|
| 306 |
+
tt.reduce.return %iright_421 : i32 loc(#loc195)
|
| 307 |
+
}) : (tensor<64x2x4xi32, #linear3>) -> tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> loc(#loc195)
|
| 308 |
+
%iright_187 = tt.expand_dims %iright_186 {axis = 1 : i32} : tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> -> tensor<64x1x4xi32, #linear3> loc(#loc162)
|
| 309 |
+
%iright_188 = tt.broadcast %iright_187 : tensor<64x1x4xi32, #linear3> -> tensor<64x2x4xi32, #linear3> loc(#loc163)
|
| 310 |
+
%ileft_189 = tt.reshape %ileft_184 : tensor<64x2x4xi32, #linear3> -> tensor<32x16xi32, #linear> loc(#loc164)
|
| 311 |
+
%iright_190 = tt.reshape %iright_188 : tensor<64x2x4xi32, #linear3> -> tensor<32x16xi32, #linear> loc(#loc165)
|
| 312 |
+
%y_idx_191 = tt.reshape %new_idxs_176 : tensor<32x16xi32, #linear> -> tensor<64x2x4xi32, #linear3> loc(#loc166)
|
| 313 |
+
%left_idx_192 = arith.muli %y_idx_191, %ileft_180 : tensor<64x2x4xi32, #linear3> loc(#loc168)
|
| 314 |
+
%left_idx_193 = "tt.reduce"(%left_idx_192) <{axis = 1 : i32}> ({
|
| 315 |
+
^bb0(%left_idx_419: i32 loc(callsite(#loc1 at #loc169)), %left_idx_420: i32 loc(callsite(#loc1 at #loc169))):
|
| 316 |
+
%left_idx_421 = arith.addi %left_idx_419, %left_idx_420 : i32 loc(#loc205)
|
| 317 |
+
tt.reduce.return %left_idx_421 : i32 loc(#loc198)
|
| 318 |
+
}) : (tensor<64x2x4xi32, #linear3>) -> tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> loc(#loc198)
|
| 319 |
+
%left_idx_194 = tt.expand_dims %left_idx_193 {axis = 1 : i32} : tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> -> tensor<64x1x4xi32, #linear3> loc(#loc170)
|
| 320 |
+
%left_idx_195 = tt.broadcast %left_idx_194 : tensor<64x1x4xi32, #linear3> -> tensor<64x2x4xi32, #linear3> loc(#loc171)
|
| 321 |
+
%right_idx_196 = arith.muli %y_idx_191, %flip_102 : tensor<64x2x4xi32, #linear3> loc(#loc173)
|
| 322 |
+
%right_idx_197 = "tt.reduce"(%right_idx_196) <{axis = 1 : i32}> ({
|
| 323 |
+
^bb0(%right_idx_419: i32 loc(callsite(#loc1 at #loc174)), %right_idx_420: i32 loc(callsite(#loc1 at #loc174))):
|
| 324 |
+
%right_idx_421 = arith.addi %right_idx_419, %right_idx_420 : i32 loc(#loc206)
|
| 325 |
+
tt.reduce.return %right_idx_421 : i32 loc(#loc201)
|
| 326 |
+
}) : (tensor<64x2x4xi32, #linear3>) -> tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> loc(#loc201)
|
| 327 |
+
%right_idx_198 = tt.expand_dims %right_idx_197 {axis = 1 : i32} : tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> -> tensor<64x1x4xi32, #linear3> loc(#loc175)
|
| 328 |
+
%right_idx_199 = tt.broadcast %right_idx_198 : tensor<64x1x4xi32, #linear3> -> tensor<64x2x4xi32, #linear3> loc(#loc176)
|
| 329 |
+
%left_idx_200 = tt.reshape %left_idx_195 : tensor<64x2x4xi32, #linear3> -> tensor<32x16xi32, #linear> loc(#loc177)
|
| 330 |
+
%right_idx_201 = tt.reshape %right_idx_199 : tensor<64x2x4xi32, #linear3> -> tensor<32x16xi32, #linear> loc(#loc178)
|
| 331 |
+
%cond_202 = arith.cmpi slt, %ileft_189, %iright_190 : tensor<32x16xi32, #linear> loc(#loc179)
|
| 332 |
+
%eq_203 = arith.cmpi eq, %ileft_189, %iright_190 : tensor<32x16xi32, #linear> loc(#loc180)
|
| 333 |
+
%cond_204 = arith.cmpi sgt, %left_idx_200, %right_idx_201 : tensor<32x16xi32, #linear> loc(#loc181)
|
| 334 |
+
%cond_205 = arith.andi %eq_203, %cond_204 : tensor<32x16xi1, #linear> loc(#loc182)
|
| 335 |
+
%cond_206 = arith.ori %cond_202, %cond_205 : tensor<32x16xi1, #linear> loc(#loc183)
|
| 336 |
+
%cond_207 = arith.extui %cond_206 : tensor<32x16xi1, #linear> to tensor<32x16xi32, #linear> loc(#loc184)
|
| 337 |
+
%cond_208 = arith.xori %cond_207, %flip_178 : tensor<32x16xi32, #linear> loc(#loc184)
|
| 338 |
+
%cond_209 = arith.cmpi ne, %cond_208, %cst : tensor<32x16xi32, #linear> loc(#loc185)
|
| 339 |
+
%ret_210 = arith.xori %ileft_189, %iright_190 : tensor<32x16xi32, #linear> loc(#loc186)
|
| 340 |
+
%ret_211 = arith.select %cond_209, %ret_210, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc187)
|
| 341 |
+
%ret_212 = arith.xori %ret_173, %ret_211 : tensor<32x16xi32, #linear> loc(#loc188)
|
| 342 |
+
%new_idxs_213 = arith.xori %left_idx_200, %right_idx_201 : tensor<32x16xi32, #linear> loc(#loc189)
|
| 343 |
+
%new_idxs_214 = arith.select %cond_209, %new_idxs_213, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc190)
|
| 344 |
+
%new_idxs_215 = arith.xori %new_idxs_176, %new_idxs_214 : tensor<32x16xi32, #linear> loc(#loc191)
|
| 345 |
+
%y_216 = tt.reshape %ret_212 : tensor<32x16xi32, #linear> -> tensor<128x2x2xi32, #linear2> loc(#loc154)
|
| 346 |
+
%ileft_217 = arith.muli %y_216, %ileft_105 : tensor<128x2x2xi32, #linear2> loc(#loc156)
|
| 347 |
+
%ileft_218 = "tt.reduce"(%ileft_217) <{axis = 1 : i32}> ({
|
| 348 |
+
^bb0(%ileft_419: i32 loc(callsite(#loc1 at #loc157)), %ileft_420: i32 loc(callsite(#loc1 at #loc157))):
|
| 349 |
+
%ileft_421 = arith.addi %ileft_419, %ileft_420 : i32 loc(#loc203)
|
| 350 |
+
tt.reduce.return %ileft_421 : i32 loc(#loc193)
|
| 351 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc193)
|
| 352 |
+
%ileft_219 = tt.expand_dims %ileft_218 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc158)
|
| 353 |
+
%ileft_220 = tt.broadcast %ileft_219 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc159)
|
| 354 |
+
%iright_221 = arith.muli %y_216, %flip_48 : tensor<128x2x2xi32, #linear2> loc(#loc160)
|
| 355 |
+
%iright_222 = "tt.reduce"(%iright_221) <{axis = 1 : i32}> ({
|
| 356 |
+
^bb0(%iright_419: i32 loc(callsite(#loc1 at #loc161)), %iright_420: i32 loc(callsite(#loc1 at #loc161))):
|
| 357 |
+
%iright_421 = arith.addi %iright_419, %iright_420 : i32 loc(#loc204)
|
| 358 |
+
tt.reduce.return %iright_421 : i32 loc(#loc195)
|
| 359 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc195)
|
| 360 |
+
%iright_223 = tt.expand_dims %iright_222 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc162)
|
| 361 |
+
%iright_224 = tt.broadcast %iright_223 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc163)
|
| 362 |
+
%ileft_225 = tt.reshape %ileft_220 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc164)
|
| 363 |
+
%iright_226 = tt.reshape %iright_224 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc165)
|
| 364 |
+
%y_idx_227 = tt.reshape %new_idxs_215 : tensor<32x16xi32, #linear> -> tensor<128x2x2xi32, #linear2> loc(#loc166)
|
| 365 |
+
%left_idx_228 = arith.muli %y_idx_227, %ileft_105 : tensor<128x2x2xi32, #linear2> loc(#loc168)
|
| 366 |
+
%left_idx_229 = "tt.reduce"(%left_idx_228) <{axis = 1 : i32}> ({
|
| 367 |
+
^bb0(%left_idx_419: i32 loc(callsite(#loc1 at #loc169)), %left_idx_420: i32 loc(callsite(#loc1 at #loc169))):
|
| 368 |
+
%left_idx_421 = arith.addi %left_idx_419, %left_idx_420 : i32 loc(#loc205)
|
| 369 |
+
tt.reduce.return %left_idx_421 : i32 loc(#loc198)
|
| 370 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc198)
|
| 371 |
+
%left_idx_230 = tt.expand_dims %left_idx_229 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc170)
|
| 372 |
+
%left_idx_231 = tt.broadcast %left_idx_230 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc171)
|
| 373 |
+
%right_idx_232 = arith.muli %y_idx_227, %flip_48 : tensor<128x2x2xi32, #linear2> loc(#loc173)
|
| 374 |
+
%right_idx_233 = "tt.reduce"(%right_idx_232) <{axis = 1 : i32}> ({
|
| 375 |
+
^bb0(%right_idx_419: i32 loc(callsite(#loc1 at #loc174)), %right_idx_420: i32 loc(callsite(#loc1 at #loc174))):
|
| 376 |
+
%right_idx_421 = arith.addi %right_idx_419, %right_idx_420 : i32 loc(#loc206)
|
| 377 |
+
tt.reduce.return %right_idx_421 : i32 loc(#loc201)
|
| 378 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc201)
|
| 379 |
+
%right_idx_234 = tt.expand_dims %right_idx_233 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc175)
|
| 380 |
+
%right_idx_235 = tt.broadcast %right_idx_234 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc176)
|
| 381 |
+
%left_idx_236 = tt.reshape %left_idx_231 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc177)
|
| 382 |
+
%right_idx_237 = tt.reshape %right_idx_235 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc178)
|
| 383 |
+
%cond_238 = arith.cmpi slt, %ileft_225, %iright_226 : tensor<32x16xi32, #linear> loc(#loc179)
|
| 384 |
+
%eq_239 = arith.cmpi eq, %ileft_225, %iright_226 : tensor<32x16xi32, #linear> loc(#loc180)
|
| 385 |
+
%cond_240 = arith.cmpi sgt, %left_idx_236, %right_idx_237 : tensor<32x16xi32, #linear> loc(#loc181)
|
| 386 |
+
%cond_241 = arith.andi %eq_239, %cond_240 : tensor<32x16xi1, #linear> loc(#loc182)
|
| 387 |
+
%cond_242 = arith.ori %cond_238, %cond_241 : tensor<32x16xi1, #linear> loc(#loc183)
|
| 388 |
+
%cond_243 = arith.extui %cond_242 : tensor<32x16xi1, #linear> to tensor<32x16xi32, #linear> loc(#loc184)
|
| 389 |
+
%cond_244 = arith.xori %cond_243, %flip_178 : tensor<32x16xi32, #linear> loc(#loc184)
|
| 390 |
+
%cond_245 = arith.cmpi ne, %cond_244, %cst : tensor<32x16xi32, #linear> loc(#loc185)
|
| 391 |
+
%ret_246 = arith.xori %ileft_225, %iright_226 : tensor<32x16xi32, #linear> loc(#loc186)
|
| 392 |
+
%ret_247 = arith.select %cond_245, %ret_246, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc187)
|
| 393 |
+
%ret_248 = arith.xori %ret_212, %ret_247 : tensor<32x16xi32, #linear> loc(#loc188)
|
| 394 |
+
%new_idxs_249 = arith.xori %left_idx_236, %right_idx_237 : tensor<32x16xi32, #linear> loc(#loc189)
|
| 395 |
+
%new_idxs_250 = arith.select %cond_245, %new_idxs_249, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc190)
|
| 396 |
+
%new_idxs_251 = arith.xori %new_idxs_215, %new_idxs_250 : tensor<32x16xi32, #linear> loc(#loc191)
|
| 397 |
+
%y_252 = tt.reshape %ret_248 : tensor<32x16xi32, #linear> -> tensor<256x2x1xi32, #linear1> loc(#loc154)
|
| 398 |
+
%ileft_253 = arith.muli %y_252, %ileft : tensor<256x2x1xi32, #linear1> loc(#loc156)
|
| 399 |
+
%ileft_254 = "tt.reduce"(%ileft_253) <{axis = 1 : i32}> ({
|
| 400 |
+
^bb0(%ileft_419: i32 loc(callsite(#loc1 at #loc157)), %ileft_420: i32 loc(callsite(#loc1 at #loc157))):
|
| 401 |
+
%ileft_421 = arith.addi %ileft_419, %ileft_420 : i32 loc(#loc203)
|
| 402 |
+
tt.reduce.return %ileft_421 : i32 loc(#loc193)
|
| 403 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc193)
|
| 404 |
+
%ileft_255 = tt.expand_dims %ileft_254 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc158)
|
| 405 |
+
%ileft_256 = tt.broadcast %ileft_255 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc159)
|
| 406 |
+
%iright_257 = arith.muli %y_252, %iright : tensor<256x2x1xi32, #linear1> loc(#loc160)
|
| 407 |
+
%iright_258 = "tt.reduce"(%iright_257) <{axis = 1 : i32}> ({
|
| 408 |
+
^bb0(%iright_419: i32 loc(callsite(#loc1 at #loc161)), %iright_420: i32 loc(callsite(#loc1 at #loc161))):
|
| 409 |
+
%iright_421 = arith.addi %iright_419, %iright_420 : i32 loc(#loc204)
|
| 410 |
+
tt.reduce.return %iright_421 : i32 loc(#loc195)
|
| 411 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc195)
|
| 412 |
+
%iright_259 = tt.expand_dims %iright_258 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc162)
|
| 413 |
+
%iright_260 = tt.broadcast %iright_259 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc163)
|
| 414 |
+
%ileft_261 = tt.reshape %ileft_256 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc164)
|
| 415 |
+
%iright_262 = tt.reshape %iright_260 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc165)
|
| 416 |
+
%y_idx_263 = tt.reshape %new_idxs_251 : tensor<32x16xi32, #linear> -> tensor<256x2x1xi32, #linear1> loc(#loc166)
|
| 417 |
+
%left_idx_264 = arith.muli %y_idx_263, %ileft : tensor<256x2x1xi32, #linear1> loc(#loc168)
|
| 418 |
+
%left_idx_265 = "tt.reduce"(%left_idx_264) <{axis = 1 : i32}> ({
|
| 419 |
+
^bb0(%left_idx_419: i32 loc(callsite(#loc1 at #loc169)), %left_idx_420: i32 loc(callsite(#loc1 at #loc169))):
|
| 420 |
+
%left_idx_421 = arith.addi %left_idx_419, %left_idx_420 : i32 loc(#loc205)
|
| 421 |
+
tt.reduce.return %left_idx_421 : i32 loc(#loc198)
|
| 422 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc198)
|
| 423 |
+
%left_idx_266 = tt.expand_dims %left_idx_265 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc170)
|
| 424 |
+
%left_idx_267 = tt.broadcast %left_idx_266 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc171)
|
| 425 |
+
%right_idx_268 = arith.muli %y_idx_263, %iright : tensor<256x2x1xi32, #linear1> loc(#loc173)
|
| 426 |
+
%right_idx_269 = "tt.reduce"(%right_idx_268) <{axis = 1 : i32}> ({
|
| 427 |
+
^bb0(%right_idx_419: i32 loc(callsite(#loc1 at #loc174)), %right_idx_420: i32 loc(callsite(#loc1 at #loc174))):
|
| 428 |
+
%right_idx_421 = arith.addi %right_idx_419, %right_idx_420 : i32 loc(#loc206)
|
| 429 |
+
tt.reduce.return %right_idx_421 : i32 loc(#loc201)
|
| 430 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc201)
|
| 431 |
+
%right_idx_270 = tt.expand_dims %right_idx_269 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc175)
|
| 432 |
+
%right_idx_271 = tt.broadcast %right_idx_270 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc176)
|
| 433 |
+
%left_idx_272 = tt.reshape %left_idx_267 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc177)
|
| 434 |
+
%right_idx_273 = tt.reshape %right_idx_271 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc178)
|
| 435 |
+
%cond_274 = arith.cmpi slt, %ileft_261, %iright_262 : tensor<32x16xi32, #linear> loc(#loc179)
|
| 436 |
+
%eq_275 = arith.cmpi eq, %ileft_261, %iright_262 : tensor<32x16xi32, #linear> loc(#loc180)
|
| 437 |
+
%cond_276 = arith.cmpi sgt, %left_idx_272, %right_idx_273 : tensor<32x16xi32, #linear> loc(#loc181)
|
| 438 |
+
%cond_277 = arith.andi %eq_275, %cond_276 : tensor<32x16xi1, #linear> loc(#loc182)
|
| 439 |
+
%cond_278 = arith.ori %cond_274, %cond_277 : tensor<32x16xi1, #linear> loc(#loc183)
|
| 440 |
+
%cond_279 = arith.extui %cond_278 : tensor<32x16xi1, #linear> to tensor<32x16xi32, #linear> loc(#loc184)
|
| 441 |
+
%cond_280 = arith.xori %cond_279, %flip_178 : tensor<32x16xi32, #linear> loc(#loc184)
|
| 442 |
+
%cond_281 = arith.cmpi ne, %cond_280, %cst : tensor<32x16xi32, #linear> loc(#loc185)
|
| 443 |
+
%ret_282 = arith.xori %ileft_261, %iright_262 : tensor<32x16xi32, #linear> loc(#loc186)
|
| 444 |
+
%ret_283 = arith.select %cond_281, %ret_282, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc187)
|
| 445 |
+
%ret_284 = arith.xori %ret_248, %ret_283 : tensor<32x16xi32, #linear> loc(#loc188)
|
| 446 |
+
%new_idxs_285 = arith.xori %left_idx_272, %right_idx_273 : tensor<32x16xi32, #linear> loc(#loc189)
|
| 447 |
+
%new_idxs_286 = arith.select %cond_281, %new_idxs_285, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc190)
|
| 448 |
+
%new_idxs_287 = arith.xori %new_idxs_251, %new_idxs_286 : tensor<32x16xi32, #linear> loc(#loc191)
|
| 449 |
+
%y_288 = tt.reshape %ret_284 : tensor<32x16xi32, #linear> -> tensor<32x2x8xi32, #linear4> loc(#loc154)
|
| 450 |
+
%ileft_289 = tt.broadcast %left_mask_53 : tensor<1x2x1xi32, #linear4> -> tensor<32x2x8xi32, #linear4> loc(#loc156)
|
| 451 |
+
%ileft_290 = arith.muli %y_288, %ileft_289 : tensor<32x2x8xi32, #linear4> loc(#loc156)
|
| 452 |
+
%ileft_291 = "tt.reduce"(%ileft_290) <{axis = 1 : i32}> ({
|
| 453 |
+
^bb0(%ileft_419: i32 loc(callsite(#loc1 at #loc157)), %ileft_420: i32 loc(callsite(#loc1 at #loc157))):
|
| 454 |
+
%ileft_421 = arith.addi %ileft_419, %ileft_420 : i32 loc(#loc203)
|
| 455 |
+
tt.reduce.return %ileft_421 : i32 loc(#loc193)
|
| 456 |
+
}) : (tensor<32x2x8xi32, #linear4>) -> tensor<32x8xi32, #ttg.slice<{dim = 1, parent = #linear4}>> loc(#loc193)
|
| 457 |
+
%ileft_292 = tt.expand_dims %ileft_291 {axis = 1 : i32} : tensor<32x8xi32, #ttg.slice<{dim = 1, parent = #linear4}>> -> tensor<32x1x8xi32, #linear4> loc(#loc158)
|
| 458 |
+
%ileft_293 = tt.broadcast %ileft_292 : tensor<32x1x8xi32, #linear4> -> tensor<32x2x8xi32, #linear4> loc(#loc159)
|
| 459 |
+
%iright_294 = arith.muli %y_288, %flip_177 : tensor<32x2x8xi32, #linear4> loc(#loc160)
|
| 460 |
+
%iright_295 = "tt.reduce"(%iright_294) <{axis = 1 : i32}> ({
|
| 461 |
+
^bb0(%iright_419: i32 loc(callsite(#loc1 at #loc161)), %iright_420: i32 loc(callsite(#loc1 at #loc161))):
|
| 462 |
+
%iright_421 = arith.addi %iright_419, %iright_420 : i32 loc(#loc204)
|
| 463 |
+
tt.reduce.return %iright_421 : i32 loc(#loc195)
|
| 464 |
+
}) : (tensor<32x2x8xi32, #linear4>) -> tensor<32x8xi32, #ttg.slice<{dim = 1, parent = #linear4}>> loc(#loc195)
|
| 465 |
+
%iright_296 = tt.expand_dims %iright_295 {axis = 1 : i32} : tensor<32x8xi32, #ttg.slice<{dim = 1, parent = #linear4}>> -> tensor<32x1x8xi32, #linear4> loc(#loc162)
|
| 466 |
+
%iright_297 = tt.broadcast %iright_296 : tensor<32x1x8xi32, #linear4> -> tensor<32x2x8xi32, #linear4> loc(#loc163)
|
| 467 |
+
%ileft_298 = tt.reshape %ileft_293 : tensor<32x2x8xi32, #linear4> -> tensor<32x16xi32, #linear> loc(#loc164)
|
| 468 |
+
%iright_299 = tt.reshape %iright_297 : tensor<32x2x8xi32, #linear4> -> tensor<32x16xi32, #linear> loc(#loc165)
|
| 469 |
+
%y_idx_300 = tt.reshape %new_idxs_287 : tensor<32x16xi32, #linear> -> tensor<32x2x8xi32, #linear4> loc(#loc166)
|
| 470 |
+
%left_idx_301 = arith.muli %y_idx_300, %ileft_289 : tensor<32x2x8xi32, #linear4> loc(#loc168)
|
| 471 |
+
%left_idx_302 = "tt.reduce"(%left_idx_301) <{axis = 1 : i32}> ({
|
| 472 |
+
^bb0(%left_idx_419: i32 loc(callsite(#loc1 at #loc169)), %left_idx_420: i32 loc(callsite(#loc1 at #loc169))):
|
| 473 |
+
%left_idx_421 = arith.addi %left_idx_419, %left_idx_420 : i32 loc(#loc205)
|
| 474 |
+
tt.reduce.return %left_idx_421 : i32 loc(#loc198)
|
| 475 |
+
}) : (tensor<32x2x8xi32, #linear4>) -> tensor<32x8xi32, #ttg.slice<{dim = 1, parent = #linear4}>> loc(#loc198)
|
| 476 |
+
%left_idx_303 = tt.expand_dims %left_idx_302 {axis = 1 : i32} : tensor<32x8xi32, #ttg.slice<{dim = 1, parent = #linear4}>> -> tensor<32x1x8xi32, #linear4> loc(#loc170)
|
| 477 |
+
%left_idx_304 = tt.broadcast %left_idx_303 : tensor<32x1x8xi32, #linear4> -> tensor<32x2x8xi32, #linear4> loc(#loc171)
|
| 478 |
+
%right_idx_305 = arith.muli %y_idx_300, %flip_177 : tensor<32x2x8xi32, #linear4> loc(#loc173)
|
| 479 |
+
%right_idx_306 = "tt.reduce"(%right_idx_305) <{axis = 1 : i32}> ({
|
| 480 |
+
^bb0(%right_idx_419: i32 loc(callsite(#loc1 at #loc174)), %right_idx_420: i32 loc(callsite(#loc1 at #loc174))):
|
| 481 |
+
%right_idx_421 = arith.addi %right_idx_419, %right_idx_420 : i32 loc(#loc206)
|
| 482 |
+
tt.reduce.return %right_idx_421 : i32 loc(#loc201)
|
| 483 |
+
}) : (tensor<32x2x8xi32, #linear4>) -> tensor<32x8xi32, #ttg.slice<{dim = 1, parent = #linear4}>> loc(#loc201)
|
| 484 |
+
%right_idx_307 = tt.expand_dims %right_idx_306 {axis = 1 : i32} : tensor<32x8xi32, #ttg.slice<{dim = 1, parent = #linear4}>> -> tensor<32x1x8xi32, #linear4> loc(#loc175)
|
| 485 |
+
%right_idx_308 = tt.broadcast %right_idx_307 : tensor<32x1x8xi32, #linear4> -> tensor<32x2x8xi32, #linear4> loc(#loc176)
|
| 486 |
+
%left_idx_309 = tt.reshape %left_idx_304 : tensor<32x2x8xi32, #linear4> -> tensor<32x16xi32, #linear> loc(#loc177)
|
| 487 |
+
%right_idx_310 = tt.reshape %right_idx_308 : tensor<32x2x8xi32, #linear4> -> tensor<32x16xi32, #linear> loc(#loc178)
|
| 488 |
+
%cond_311 = arith.cmpi slt, %ileft_298, %iright_299 : tensor<32x16xi32, #linear> loc(#loc179)
|
| 489 |
+
%eq_312 = arith.cmpi eq, %ileft_298, %iright_299 : tensor<32x16xi32, #linear> loc(#loc180)
|
| 490 |
+
%cond_313 = arith.cmpi sgt, %left_idx_309, %right_idx_310 : tensor<32x16xi32, #linear> loc(#loc181)
|
| 491 |
+
%cond_314 = arith.andi %eq_312, %cond_313 : tensor<32x16xi1, #linear> loc(#loc182)
|
| 492 |
+
%cond_315 = arith.ori %cond_311, %cond_314 : tensor<32x16xi1, #linear> loc(#loc183)
|
| 493 |
+
%ret_316 = arith.xori %ileft_298, %iright_299 : tensor<32x16xi32, #linear> loc(#loc186)
|
| 494 |
+
%ret_317 = arith.select %cond_315, %ret_316, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc187)
|
| 495 |
+
%ret_318 = arith.xori %ret_284, %ret_317 : tensor<32x16xi32, #linear> loc(#loc188)
|
| 496 |
+
%new_idxs_319 = arith.xori %left_idx_309, %right_idx_310 : tensor<32x16xi32, #linear> loc(#loc189)
|
| 497 |
+
%new_idxs_320 = arith.select %cond_315, %new_idxs_319, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc190)
|
| 498 |
+
%new_idxs_321 = arith.xori %new_idxs_287, %new_idxs_320 : tensor<32x16xi32, #linear> loc(#loc191)
|
| 499 |
+
%y_322 = tt.reshape %ret_318 : tensor<32x16xi32, #linear> -> tensor<64x2x4xi32, #linear3> loc(#loc154)
|
| 500 |
+
%ileft_323 = arith.muli %y_322, %ileft_180 : tensor<64x2x4xi32, #linear3> loc(#loc156)
|
| 501 |
+
%ileft_324 = "tt.reduce"(%ileft_323) <{axis = 1 : i32}> ({
|
| 502 |
+
^bb0(%ileft_419: i32 loc(callsite(#loc1 at #loc157)), %ileft_420: i32 loc(callsite(#loc1 at #loc157))):
|
| 503 |
+
%ileft_421 = arith.addi %ileft_419, %ileft_420 : i32 loc(#loc203)
|
| 504 |
+
tt.reduce.return %ileft_421 : i32 loc(#loc193)
|
| 505 |
+
}) : (tensor<64x2x4xi32, #linear3>) -> tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> loc(#loc193)
|
| 506 |
+
%ileft_325 = tt.expand_dims %ileft_324 {axis = 1 : i32} : tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> -> tensor<64x1x4xi32, #linear3> loc(#loc158)
|
| 507 |
+
%ileft_326 = tt.broadcast %ileft_325 : tensor<64x1x4xi32, #linear3> -> tensor<64x2x4xi32, #linear3> loc(#loc159)
|
| 508 |
+
%iright_327 = arith.muli %y_322, %flip_102 : tensor<64x2x4xi32, #linear3> loc(#loc160)
|
| 509 |
+
%iright_328 = "tt.reduce"(%iright_327) <{axis = 1 : i32}> ({
|
| 510 |
+
^bb0(%iright_419: i32 loc(callsite(#loc1 at #loc161)), %iright_420: i32 loc(callsite(#loc1 at #loc161))):
|
| 511 |
+
%iright_421 = arith.addi %iright_419, %iright_420 : i32 loc(#loc204)
|
| 512 |
+
tt.reduce.return %iright_421 : i32 loc(#loc195)
|
| 513 |
+
}) : (tensor<64x2x4xi32, #linear3>) -> tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> loc(#loc195)
|
| 514 |
+
%iright_329 = tt.expand_dims %iright_328 {axis = 1 : i32} : tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> -> tensor<64x1x4xi32, #linear3> loc(#loc162)
|
| 515 |
+
%iright_330 = tt.broadcast %iright_329 : tensor<64x1x4xi32, #linear3> -> tensor<64x2x4xi32, #linear3> loc(#loc163)
|
| 516 |
+
%ileft_331 = tt.reshape %ileft_326 : tensor<64x2x4xi32, #linear3> -> tensor<32x16xi32, #linear> loc(#loc164)
|
| 517 |
+
%iright_332 = tt.reshape %iright_330 : tensor<64x2x4xi32, #linear3> -> tensor<32x16xi32, #linear> loc(#loc165)
|
| 518 |
+
%y_idx_333 = tt.reshape %new_idxs_321 : tensor<32x16xi32, #linear> -> tensor<64x2x4xi32, #linear3> loc(#loc166)
|
| 519 |
+
%left_idx_334 = arith.muli %y_idx_333, %ileft_180 : tensor<64x2x4xi32, #linear3> loc(#loc168)
|
| 520 |
+
%left_idx_335 = "tt.reduce"(%left_idx_334) <{axis = 1 : i32}> ({
|
| 521 |
+
^bb0(%left_idx_419: i32 loc(callsite(#loc1 at #loc169)), %left_idx_420: i32 loc(callsite(#loc1 at #loc169))):
|
| 522 |
+
%left_idx_421 = arith.addi %left_idx_419, %left_idx_420 : i32 loc(#loc205)
|
| 523 |
+
tt.reduce.return %left_idx_421 : i32 loc(#loc198)
|
| 524 |
+
}) : (tensor<64x2x4xi32, #linear3>) -> tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> loc(#loc198)
|
| 525 |
+
%left_idx_336 = tt.expand_dims %left_idx_335 {axis = 1 : i32} : tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> -> tensor<64x1x4xi32, #linear3> loc(#loc170)
|
| 526 |
+
%left_idx_337 = tt.broadcast %left_idx_336 : tensor<64x1x4xi32, #linear3> -> tensor<64x2x4xi32, #linear3> loc(#loc171)
|
| 527 |
+
%right_idx_338 = arith.muli %y_idx_333, %flip_102 : tensor<64x2x4xi32, #linear3> loc(#loc173)
|
| 528 |
+
%right_idx_339 = "tt.reduce"(%right_idx_338) <{axis = 1 : i32}> ({
|
| 529 |
+
^bb0(%right_idx_419: i32 loc(callsite(#loc1 at #loc174)), %right_idx_420: i32 loc(callsite(#loc1 at #loc174))):
|
| 530 |
+
%right_idx_421 = arith.addi %right_idx_419, %right_idx_420 : i32 loc(#loc206)
|
| 531 |
+
tt.reduce.return %right_idx_421 : i32 loc(#loc201)
|
| 532 |
+
}) : (tensor<64x2x4xi32, #linear3>) -> tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> loc(#loc201)
|
| 533 |
+
%right_idx_340 = tt.expand_dims %right_idx_339 {axis = 1 : i32} : tensor<64x4xi32, #ttg.slice<{dim = 1, parent = #linear3}>> -> tensor<64x1x4xi32, #linear3> loc(#loc175)
|
| 534 |
+
%right_idx_341 = tt.broadcast %right_idx_340 : tensor<64x1x4xi32, #linear3> -> tensor<64x2x4xi32, #linear3> loc(#loc176)
|
| 535 |
+
%left_idx_342 = tt.reshape %left_idx_337 : tensor<64x2x4xi32, #linear3> -> tensor<32x16xi32, #linear> loc(#loc177)
|
| 536 |
+
%right_idx_343 = tt.reshape %right_idx_341 : tensor<64x2x4xi32, #linear3> -> tensor<32x16xi32, #linear> loc(#loc178)
|
| 537 |
+
%cond_344 = arith.cmpi slt, %ileft_331, %iright_332 : tensor<32x16xi32, #linear> loc(#loc179)
|
| 538 |
+
%eq_345 = arith.cmpi eq, %ileft_331, %iright_332 : tensor<32x16xi32, #linear> loc(#loc180)
|
| 539 |
+
%cond_346 = arith.cmpi sgt, %left_idx_342, %right_idx_343 : tensor<32x16xi32, #linear> loc(#loc181)
|
| 540 |
+
%cond_347 = arith.andi %eq_345, %cond_346 : tensor<32x16xi1, #linear> loc(#loc182)
|
| 541 |
+
%cond_348 = arith.ori %cond_344, %cond_347 : tensor<32x16xi1, #linear> loc(#loc183)
|
| 542 |
+
%ret_349 = arith.xori %ileft_331, %iright_332 : tensor<32x16xi32, #linear> loc(#loc186)
|
| 543 |
+
%ret_350 = arith.select %cond_348, %ret_349, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc187)
|
| 544 |
+
%ret_351 = arith.xori %ret_318, %ret_350 : tensor<32x16xi32, #linear> loc(#loc188)
|
| 545 |
+
%new_idxs_352 = arith.xori %left_idx_342, %right_idx_343 : tensor<32x16xi32, #linear> loc(#loc189)
|
| 546 |
+
%new_idxs_353 = arith.select %cond_348, %new_idxs_352, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc190)
|
| 547 |
+
%new_idxs_354 = arith.xori %new_idxs_321, %new_idxs_353 : tensor<32x16xi32, #linear> loc(#loc191)
|
| 548 |
+
%y_355 = tt.reshape %ret_351 : tensor<32x16xi32, #linear> -> tensor<128x2x2xi32, #linear2> loc(#loc154)
|
| 549 |
+
%ileft_356 = arith.muli %y_355, %ileft_105 : tensor<128x2x2xi32, #linear2> loc(#loc156)
|
| 550 |
+
%ileft_357 = "tt.reduce"(%ileft_356) <{axis = 1 : i32}> ({
|
| 551 |
+
^bb0(%ileft_419: i32 loc(callsite(#loc1 at #loc157)), %ileft_420: i32 loc(callsite(#loc1 at #loc157))):
|
| 552 |
+
%ileft_421 = arith.addi %ileft_419, %ileft_420 : i32 loc(#loc203)
|
| 553 |
+
tt.reduce.return %ileft_421 : i32 loc(#loc193)
|
| 554 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc193)
|
| 555 |
+
%ileft_358 = tt.expand_dims %ileft_357 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc158)
|
| 556 |
+
%ileft_359 = tt.broadcast %ileft_358 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc159)
|
| 557 |
+
%iright_360 = arith.muli %y_355, %flip_48 : tensor<128x2x2xi32, #linear2> loc(#loc160)
|
| 558 |
+
%iright_361 = "tt.reduce"(%iright_360) <{axis = 1 : i32}> ({
|
| 559 |
+
^bb0(%iright_419: i32 loc(callsite(#loc1 at #loc161)), %iright_420: i32 loc(callsite(#loc1 at #loc161))):
|
| 560 |
+
%iright_421 = arith.addi %iright_419, %iright_420 : i32 loc(#loc204)
|
| 561 |
+
tt.reduce.return %iright_421 : i32 loc(#loc195)
|
| 562 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc195)
|
| 563 |
+
%iright_362 = tt.expand_dims %iright_361 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc162)
|
| 564 |
+
%iright_363 = tt.broadcast %iright_362 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc163)
|
| 565 |
+
%ileft_364 = tt.reshape %ileft_359 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc164)
|
| 566 |
+
%iright_365 = tt.reshape %iright_363 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc165)
|
| 567 |
+
%y_idx_366 = tt.reshape %new_idxs_354 : tensor<32x16xi32, #linear> -> tensor<128x2x2xi32, #linear2> loc(#loc166)
|
| 568 |
+
%left_idx_367 = arith.muli %y_idx_366, %ileft_105 : tensor<128x2x2xi32, #linear2> loc(#loc168)
|
| 569 |
+
%left_idx_368 = "tt.reduce"(%left_idx_367) <{axis = 1 : i32}> ({
|
| 570 |
+
^bb0(%left_idx_419: i32 loc(callsite(#loc1 at #loc169)), %left_idx_420: i32 loc(callsite(#loc1 at #loc169))):
|
| 571 |
+
%left_idx_421 = arith.addi %left_idx_419, %left_idx_420 : i32 loc(#loc205)
|
| 572 |
+
tt.reduce.return %left_idx_421 : i32 loc(#loc198)
|
| 573 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc198)
|
| 574 |
+
%left_idx_369 = tt.expand_dims %left_idx_368 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc170)
|
| 575 |
+
%left_idx_370 = tt.broadcast %left_idx_369 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc171)
|
| 576 |
+
%right_idx_371 = arith.muli %y_idx_366, %flip_48 : tensor<128x2x2xi32, #linear2> loc(#loc173)
|
| 577 |
+
%right_idx_372 = "tt.reduce"(%right_idx_371) <{axis = 1 : i32}> ({
|
| 578 |
+
^bb0(%right_idx_419: i32 loc(callsite(#loc1 at #loc174)), %right_idx_420: i32 loc(callsite(#loc1 at #loc174))):
|
| 579 |
+
%right_idx_421 = arith.addi %right_idx_419, %right_idx_420 : i32 loc(#loc206)
|
| 580 |
+
tt.reduce.return %right_idx_421 : i32 loc(#loc201)
|
| 581 |
+
}) : (tensor<128x2x2xi32, #linear2>) -> tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> loc(#loc201)
|
| 582 |
+
%right_idx_373 = tt.expand_dims %right_idx_372 {axis = 1 : i32} : tensor<128x2xi32, #ttg.slice<{dim = 1, parent = #linear2}>> -> tensor<128x1x2xi32, #linear2> loc(#loc175)
|
| 583 |
+
%right_idx_374 = tt.broadcast %right_idx_373 : tensor<128x1x2xi32, #linear2> -> tensor<128x2x2xi32, #linear2> loc(#loc176)
|
| 584 |
+
%left_idx_375 = tt.reshape %left_idx_370 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc177)
|
| 585 |
+
%right_idx_376 = tt.reshape %right_idx_374 : tensor<128x2x2xi32, #linear2> -> tensor<32x16xi32, #linear> loc(#loc178)
|
| 586 |
+
%cond_377 = arith.cmpi slt, %ileft_364, %iright_365 : tensor<32x16xi32, #linear> loc(#loc179)
|
| 587 |
+
%eq_378 = arith.cmpi eq, %ileft_364, %iright_365 : tensor<32x16xi32, #linear> loc(#loc180)
|
| 588 |
+
%cond_379 = arith.cmpi sgt, %left_idx_375, %right_idx_376 : tensor<32x16xi32, #linear> loc(#loc181)
|
| 589 |
+
%cond_380 = arith.andi %eq_378, %cond_379 : tensor<32x16xi1, #linear> loc(#loc182)
|
| 590 |
+
%cond_381 = arith.ori %cond_377, %cond_380 : tensor<32x16xi1, #linear> loc(#loc183)
|
| 591 |
+
%ret_382 = arith.xori %ileft_364, %iright_365 : tensor<32x16xi32, #linear> loc(#loc186)
|
| 592 |
+
%ret_383 = arith.select %cond_381, %ret_382, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc187)
|
| 593 |
+
%ret_384 = arith.xori %ret_351, %ret_383 : tensor<32x16xi32, #linear> loc(#loc188)
|
| 594 |
+
%new_idxs_385 = arith.xori %left_idx_375, %right_idx_376 : tensor<32x16xi32, #linear> loc(#loc189)
|
| 595 |
+
%new_idxs_386 = arith.select %cond_381, %new_idxs_385, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc190)
|
| 596 |
+
%new_idxs_387 = arith.xori %new_idxs_354, %new_idxs_386 : tensor<32x16xi32, #linear> loc(#loc191)
|
| 597 |
+
%y_388 = tt.reshape %ret_384 : tensor<32x16xi32, #linear> -> tensor<256x2x1xi32, #linear1> loc(#loc154)
|
| 598 |
+
%ileft_389 = arith.muli %y_388, %ileft : tensor<256x2x1xi32, #linear1> loc(#loc156)
|
| 599 |
+
%ileft_390 = "tt.reduce"(%ileft_389) <{axis = 1 : i32}> ({
|
| 600 |
+
^bb0(%ileft_419: i32 loc(callsite(#loc1 at #loc157)), %ileft_420: i32 loc(callsite(#loc1 at #loc157))):
|
| 601 |
+
%ileft_421 = arith.addi %ileft_419, %ileft_420 : i32 loc(#loc203)
|
| 602 |
+
tt.reduce.return %ileft_421 : i32 loc(#loc193)
|
| 603 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc193)
|
| 604 |
+
%ileft_391 = tt.expand_dims %ileft_390 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc158)
|
| 605 |
+
%ileft_392 = tt.broadcast %ileft_391 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc159)
|
| 606 |
+
%iright_393 = arith.muli %y_388, %iright : tensor<256x2x1xi32, #linear1> loc(#loc160)
|
| 607 |
+
%iright_394 = "tt.reduce"(%iright_393) <{axis = 1 : i32}> ({
|
| 608 |
+
^bb0(%iright_419: i32 loc(callsite(#loc1 at #loc161)), %iright_420: i32 loc(callsite(#loc1 at #loc161))):
|
| 609 |
+
%iright_421 = arith.addi %iright_419, %iright_420 : i32 loc(#loc204)
|
| 610 |
+
tt.reduce.return %iright_421 : i32 loc(#loc195)
|
| 611 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc195)
|
| 612 |
+
%iright_395 = tt.expand_dims %iright_394 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc162)
|
| 613 |
+
%iright_396 = tt.broadcast %iright_395 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc163)
|
| 614 |
+
%ileft_397 = tt.reshape %ileft_392 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc164)
|
| 615 |
+
%iright_398 = tt.reshape %iright_396 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc165)
|
| 616 |
+
%y_idx_399 = tt.reshape %new_idxs_387 : tensor<32x16xi32, #linear> -> tensor<256x2x1xi32, #linear1> loc(#loc166)
|
| 617 |
+
%left_idx_400 = arith.muli %y_idx_399, %ileft : tensor<256x2x1xi32, #linear1> loc(#loc168)
|
| 618 |
+
%left_idx_401 = "tt.reduce"(%left_idx_400) <{axis = 1 : i32}> ({
|
| 619 |
+
^bb0(%left_idx_419: i32 loc(callsite(#loc1 at #loc169)), %left_idx_420: i32 loc(callsite(#loc1 at #loc169))):
|
| 620 |
+
%left_idx_421 = arith.addi %left_idx_419, %left_idx_420 : i32 loc(#loc205)
|
| 621 |
+
tt.reduce.return %left_idx_421 : i32 loc(#loc198)
|
| 622 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc198)
|
| 623 |
+
%left_idx_402 = tt.expand_dims %left_idx_401 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc170)
|
| 624 |
+
%left_idx_403 = tt.broadcast %left_idx_402 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc171)
|
| 625 |
+
%right_idx_404 = arith.muli %y_idx_399, %iright : tensor<256x2x1xi32, #linear1> loc(#loc173)
|
| 626 |
+
%right_idx_405 = "tt.reduce"(%right_idx_404) <{axis = 1 : i32}> ({
|
| 627 |
+
^bb0(%right_idx_419: i32 loc(callsite(#loc1 at #loc174)), %right_idx_420: i32 loc(callsite(#loc1 at #loc174))):
|
| 628 |
+
%right_idx_421 = arith.addi %right_idx_419, %right_idx_420 : i32 loc(#loc206)
|
| 629 |
+
tt.reduce.return %right_idx_421 : i32 loc(#loc201)
|
| 630 |
+
}) : (tensor<256x2x1xi32, #linear1>) -> tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> loc(#loc201)
|
| 631 |
+
%right_idx_406 = tt.expand_dims %right_idx_405 {axis = 1 : i32} : tensor<256x1xi32, #ttg.slice<{dim = 1, parent = #linear1}>> -> tensor<256x1x1xi32, #linear1> loc(#loc175)
|
| 632 |
+
%right_idx_407 = tt.broadcast %right_idx_406 : tensor<256x1x1xi32, #linear1> -> tensor<256x2x1xi32, #linear1> loc(#loc176)
|
| 633 |
+
%left_idx_408 = tt.reshape %left_idx_403 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc177)
|
| 634 |
+
%right_idx_409 = tt.reshape %right_idx_407 : tensor<256x2x1xi32, #linear1> -> tensor<32x16xi32, #linear> loc(#loc178)
|
| 635 |
+
%cond_410 = arith.cmpi slt, %ileft_397, %iright_398 : tensor<32x16xi32, #linear> loc(#loc179)
|
| 636 |
+
%eq_411 = arith.cmpi eq, %ileft_397, %iright_398 : tensor<32x16xi32, #linear> loc(#loc180)
|
| 637 |
+
%cond_412 = arith.cmpi sgt, %left_idx_408, %right_idx_409 : tensor<32x16xi32, #linear> loc(#loc181)
|
| 638 |
+
%cond_413 = arith.andi %eq_411, %cond_412 : tensor<32x16xi1, #linear> loc(#loc182)
|
| 639 |
+
%cond_414 = arith.ori %cond_410, %cond_413 : tensor<32x16xi1, #linear> loc(#loc183)
|
| 640 |
+
%new_idxs_415 = arith.xori %left_idx_408, %right_idx_409 : tensor<32x16xi32, #linear> loc(#loc189)
|
| 641 |
+
%new_idxs_416 = arith.select %cond_414, %new_idxs_415, %cst : tensor<32x16xi1, #linear>, tensor<32x16xi32, #linear> loc(#loc190)
|
| 642 |
+
%new_idxs_417 = arith.xori %new_idxs_387, %new_idxs_416 : tensor<32x16xi32, #linear> loc(#loc191)
|
| 643 |
+
%tmp7 = arith.extsi %tmp0_36 : tensor<32x16xi32, #blocked> to tensor<32x16xi64, #blocked> loc(#loc141)
|
| 644 |
+
%tmp10 = arith.select %tmp0_34, %tmp7, %cst_0 : tensor<32x16xi1, #blocked>, tensor<32x16xi64, #blocked> loc(#loc142)
|
| 645 |
+
%tmp11 = "tt.reduce"(%tmp10) <{axis = 1 : i32}> ({
|
| 646 |
+
^bb0(%tmp11_419: i64 loc(callsite(#loc1 at #loc143)), %tmp11_420: i64 loc(callsite(#loc1 at #loc143))):
|
| 647 |
+
%tmp11_421 = arith.addi %tmp11_419, %tmp11_420 : i64 loc(#loc192)
|
| 648 |
+
tt.reduce.return %tmp11_421 : i64 loc(#loc152)
|
| 649 |
+
}) : (tensor<32x16xi64, #blocked>) -> tensor<32xi64, #ttg.slice<{dim = 1, parent = #blocked}>> loc(#loc152)
|
| 650 |
+
%tmp11_418 = tt.expand_dims %tmp11 {axis = 1 : i32} : tensor<32xi64, #ttg.slice<{dim = 1, parent = #blocked}>> -> tensor<32x1xi64, #blocked> loc(#loc144)
|
| 651 |
+
%tmp14 = arith.trunci %tmp11_418 : tensor<32x1xi64, #blocked> to tensor<32x1xi32, #blocked> loc(#loc145)
|
| 652 |
+
%0 = arith.muli %xindex_19, %cst_4 : tensor<32x1xi32, #blocked1> loc(#loc70)
|
| 653 |
+
%1 = tt.broadcast %r0_index_25 : tensor<1x16xi32, #blocked1> -> tensor<32x16xi32, #blocked1> loc(#loc71)
|
| 654 |
+
%2 = tt.broadcast %0 : tensor<32x1xi32, #blocked1> -> tensor<32x16xi32, #blocked1> loc(#loc71)
|
| 655 |
+
%3 = arith.addi %1, %2 : tensor<32x16xi32, #blocked1> loc(#loc71)
|
| 656 |
+
%4 = tt.splat %out_ptr2 : !tt.ptr<i32> -> tensor<32x16x!tt.ptr<i32>, #blocked1> loc(#loc72)
|
| 657 |
+
%5 = tt.addptr %4, %3 : tensor<32x16x!tt.ptr<i32>, #blocked1>, tensor<32x16xi32, #blocked1> loc(#loc72)
|
| 658 |
+
%6 = ttg.convert_layout %new_idxs_417 : tensor<32x16xi32, #linear> -> tensor<32x16xi32, #blocked1> loc(#loc73)
|
| 659 |
+
tt.store %5, %6, %tmp0_35 : tensor<32x16x!tt.ptr<i32>, #blocked1> loc(#loc73)
|
| 660 |
+
%7 = tt.splat %out_ptr3 : !tt.ptr<i32> -> tensor<32x1x!tt.ptr<i32>, #blocked> loc(#loc74)
|
| 661 |
+
%8 = tt.addptr %7, %xindex_18 : tensor<32x1x!tt.ptr<i32>, #blocked>, tensor<32x1xi32, #blocked> loc(#loc74)
|
| 662 |
+
tt.store %8, %tmp14, %xmask : tensor<32x1x!tt.ptr<i32>, #blocked> loc(#loc75)
|
| 663 |
+
tt.return loc(#loc76)
|
| 664 |
+
} loc(#loc)
|
| 665 |
+
} loc(#loc)
|
| 666 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":24:28)
|
| 667 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":24:33)
|
| 668 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":25:44)
|
| 669 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":25:23)
|
| 670 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":26:21)
|
| 671 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":27:38)
|
| 672 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":33:19)
|
| 673 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":34:19)
|
| 674 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:38)
|
| 675 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:35)
|
| 676 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:49)
|
| 677 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:45)
|
| 678 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:30)
|
| 679 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:54)
|
| 680 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":38:19)
|
| 681 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":40:33)
|
| 682 |
+
#loc18 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":627:44)
|
| 683 |
+
#loc21 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":627:60)
|
| 684 |
+
#loc22 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":627:68)
|
| 685 |
+
#loc23 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":533:22)
|
| 686 |
+
#loc25 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":537:21)
|
| 687 |
+
#loc26 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":538:40)
|
| 688 |
+
#loc27 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:36)
|
| 689 |
+
#loc29 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:15)
|
| 690 |
+
#loc30 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":538:65)
|
| 691 |
+
#loc31 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":538:78)
|
| 692 |
+
#loc32 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":539:41)
|
| 693 |
+
#loc34 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":539:67)
|
| 694 |
+
#loc35 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":539:80)
|
| 695 |
+
#loc36 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":540:30)
|
| 696 |
+
#loc37 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":541:32)
|
| 697 |
+
#loc38 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":546:29)
|
| 698 |
+
#loc39 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":548:36)
|
| 699 |
+
#loc40 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":548:23)
|
| 700 |
+
#loc41 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":290:25)
|
| 701 |
+
#loc43 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":548:53)
|
| 702 |
+
#loc44 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":548:66)
|
| 703 |
+
#loc45 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":551:37)
|
| 704 |
+
#loc46 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":551:23)
|
| 705 |
+
#loc48 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":551:54)
|
| 706 |
+
#loc49 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":551:67)
|
| 707 |
+
#loc50 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":553:36)
|
| 708 |
+
#loc51 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":554:38)
|
| 709 |
+
#loc52 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":574:22)
|
| 710 |
+
#loc53 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":591:21)
|
| 711 |
+
#loc54 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":594:40)
|
| 712 |
+
#loc55 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":594:29)
|
| 713 |
+
#loc56 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":594:23)
|
| 714 |
+
#loc57 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":599:19)
|
| 715 |
+
#loc58 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":599:28)
|
| 716 |
+
#loc59 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":600:38)
|
| 717 |
+
#loc60 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":600:46)
|
| 718 |
+
#loc61 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":600:15)
|
| 719 |
+
#loc62 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":601:48)
|
| 720 |
+
#loc63 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":601:59)
|
| 721 |
+
#loc64 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":601:22)
|
| 722 |
+
#loc65 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":42:19)
|
| 723 |
+
#loc66 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":44:34)
|
| 724 |
+
#loc68 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":45:29)
|
| 725 |
+
#loc69 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":48:21)
|
| 726 |
+
#loc70 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":49:35)
|
| 727 |
+
#loc71 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":49:32)
|
| 728 |
+
#loc72 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":49:25)
|
| 729 |
+
#loc73 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":49:47)
|
| 730 |
+
#loc74 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":50:25)
|
| 731 |
+
#loc75 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":50:37)
|
| 732 |
+
#loc76 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":50:4)
|
| 733 |
+
#loc82 = loc("xoffset"(#loc2))
|
| 734 |
+
#loc83 = loc("xoffset"(#loc3))
|
| 735 |
+
#loc84 = loc("xindex"(#loc4))
|
| 736 |
+
#loc85 = loc("xindex"(#loc5))
|
| 737 |
+
#loc86 = loc("xmask"(#loc6))
|
| 738 |
+
#loc87 = loc("r0_index"(#loc7))
|
| 739 |
+
#loc88 = loc("x0"(#loc8))
|
| 740 |
+
#loc89 = loc("x1"(#loc9))
|
| 741 |
+
#loc90 = loc("tmp0"(#loc10))
|
| 742 |
+
#loc91 = loc("tmp0"(#loc11))
|
| 743 |
+
#loc92 = loc("tmp0"(#loc12))
|
| 744 |
+
#loc93 = loc("tmp0"(#loc13))
|
| 745 |
+
#loc94 = loc("tmp0"(#loc14))
|
| 746 |
+
#loc95 = loc("tmp0"(#loc15))
|
| 747 |
+
#loc96 = loc("tmp2"(#loc16))
|
| 748 |
+
#loc97 = loc("tmp4"(#loc17))
|
| 749 |
+
#loc98 = loc("flip"(#loc18))
|
| 750 |
+
#loc100 = loc("flip"(#loc21))
|
| 751 |
+
#loc101 = loc("flip"(#loc22))
|
| 752 |
+
#loc102 = loc("y"(#loc23))
|
| 753 |
+
#loc103 = loc("left_mask"(#loc25))
|
| 754 |
+
#loc104 = loc("ileft"(#loc26))
|
| 755 |
+
#loc106 = loc("ileft"(#loc30))
|
| 756 |
+
#loc107 = loc("ileft"(#loc31))
|
| 757 |
+
#loc108 = loc("iright"(#loc32))
|
| 758 |
+
#loc110 = loc("iright"(#loc34))
|
| 759 |
+
#loc111 = loc("iright"(#loc35))
|
| 760 |
+
#loc112 = loc("ileft"(#loc36))
|
| 761 |
+
#loc113 = loc("iright"(#loc37))
|
| 762 |
+
#loc114 = loc("y_idx"(#loc38))
|
| 763 |
+
#loc115 = loc("left_idx"(#loc39))
|
| 764 |
+
#loc116 = loc("left_idx"(#loc40))
|
| 765 |
+
#loc117 = loc("input"(#loc41))
|
| 766 |
+
#loc119 = loc("left_idx"(#loc43))
|
| 767 |
+
#loc120 = loc("left_idx"(#loc44))
|
| 768 |
+
#loc121 = loc("right_idx"(#loc45))
|
| 769 |
+
#loc122 = loc("right_idx"(#loc46))
|
| 770 |
+
#loc124 = loc("right_idx"(#loc48))
|
| 771 |
+
#loc125 = loc("right_idx"(#loc49))
|
| 772 |
+
#loc126 = loc("left_idx"(#loc50))
|
| 773 |
+
#loc127 = loc("right_idx"(#loc51))
|
| 774 |
+
#loc128 = loc("cond"(#loc52))
|
| 775 |
+
#loc129 = loc("eq"(#loc53))
|
| 776 |
+
#loc130 = loc("cond"(#loc54))
|
| 777 |
+
#loc131 = loc("cond"(#loc55))
|
| 778 |
+
#loc132 = loc("cond"(#loc56))
|
| 779 |
+
#loc133 = loc("cond"(#loc57))
|
| 780 |
+
#loc134 = loc("cond"(#loc58))
|
| 781 |
+
#loc135 = loc("ret"(#loc59))
|
| 782 |
+
#loc136 = loc("ret"(#loc60))
|
| 783 |
+
#loc137 = loc("ret"(#loc61))
|
| 784 |
+
#loc138 = loc("new_idxs"(#loc62))
|
| 785 |
+
#loc139 = loc("new_idxs"(#loc63))
|
| 786 |
+
#loc140 = loc("new_idxs"(#loc64))
|
| 787 |
+
#loc141 = loc("tmp7"(#loc65))
|
| 788 |
+
#loc142 = loc("tmp10"(#loc66))
|
| 789 |
+
#loc144 = loc("tmp11"(#loc68))
|
| 790 |
+
#loc145 = loc("tmp14"(#loc69))
|
| 791 |
+
#loc146 = loc(callsite(#loc98 at #loc99))
|
| 792 |
+
#loc147 = loc(callsite(#loc100 at #loc99))
|
| 793 |
+
#loc148 = loc(callsite(#loc101 at #loc99))
|
| 794 |
+
#loc150 = loc("cond"(#loc128))
|
| 795 |
+
#loc151 = loc("eq"(#loc129))
|
| 796 |
+
#loc152 = loc(callsite(#loc27 at #loc143))
|
| 797 |
+
#loc154 = loc(callsite(#loc102 at #loc149))
|
| 798 |
+
#loc155 = loc(callsite(#loc103 at #loc149))
|
| 799 |
+
#loc156 = loc(callsite(#loc104 at #loc149))
|
| 800 |
+
#loc158 = loc(callsite(#loc106 at #loc149))
|
| 801 |
+
#loc159 = loc(callsite(#loc107 at #loc149))
|
| 802 |
+
#loc160 = loc(callsite(#loc108 at #loc149))
|
| 803 |
+
#loc162 = loc(callsite(#loc110 at #loc149))
|
| 804 |
+
#loc163 = loc(callsite(#loc111 at #loc149))
|
| 805 |
+
#loc164 = loc(callsite(#loc112 at #loc149))
|
| 806 |
+
#loc165 = loc(callsite(#loc113 at #loc149))
|
| 807 |
+
#loc166 = loc(callsite(#loc114 at #loc149))
|
| 808 |
+
#loc167 = loc(callsite(#loc115 at #loc149))
|
| 809 |
+
#loc168 = loc(callsite(#loc116 at #loc149))
|
| 810 |
+
#loc170 = loc(callsite(#loc119 at #loc149))
|
| 811 |
+
#loc171 = loc(callsite(#loc120 at #loc149))
|
| 812 |
+
#loc172 = loc(callsite(#loc121 at #loc149))
|
| 813 |
+
#loc173 = loc(callsite(#loc122 at #loc149))
|
| 814 |
+
#loc175 = loc(callsite(#loc124 at #loc149))
|
| 815 |
+
#loc176 = loc(callsite(#loc125 at #loc149))
|
| 816 |
+
#loc177 = loc(callsite(#loc126 at #loc149))
|
| 817 |
+
#loc178 = loc(callsite(#loc127 at #loc149))
|
| 818 |
+
#loc179 = loc(callsite(#loc150 at #loc149))
|
| 819 |
+
#loc180 = loc(callsite(#loc151 at #loc149))
|
| 820 |
+
#loc181 = loc(callsite(#loc130 at #loc149))
|
| 821 |
+
#loc182 = loc(callsite(#loc131 at #loc149))
|
| 822 |
+
#loc183 = loc(callsite(#loc132 at #loc149))
|
| 823 |
+
#loc184 = loc(callsite(#loc133 at #loc149))
|
| 824 |
+
#loc185 = loc(callsite(#loc134 at #loc149))
|
| 825 |
+
#loc186 = loc(callsite(#loc135 at #loc149))
|
| 826 |
+
#loc187 = loc(callsite(#loc136 at #loc149))
|
| 827 |
+
#loc188 = loc(callsite(#loc137 at #loc149))
|
| 828 |
+
#loc189 = loc(callsite(#loc138 at #loc149))
|
| 829 |
+
#loc190 = loc(callsite(#loc139 at #loc149))
|
| 830 |
+
#loc191 = loc(callsite(#loc140 at #loc149))
|
| 831 |
+
#loc192 = loc(callsite(#loc29 at #loc152))
|
| 832 |
+
#loc193 = loc(callsite(#loc27 at #loc157))
|
| 833 |
+
#loc195 = loc(callsite(#loc27 at #loc161))
|
| 834 |
+
#loc197 = loc(callsite(#loc117 at #loc169))
|
| 835 |
+
#loc198 = loc(callsite(#loc27 at #loc169))
|
| 836 |
+
#loc200 = loc(callsite(#loc117 at #loc174))
|
| 837 |
+
#loc201 = loc(callsite(#loc27 at #loc174))
|
| 838 |
+
#loc203 = loc(callsite(#loc29 at #loc193))
|
| 839 |
+
#loc204 = loc(callsite(#loc29 at #loc195))
|
| 840 |
+
#loc205 = loc(callsite(#loc29 at #loc198))
|
| 841 |
+
#loc206 = loc(callsite(#loc29 at #loc201))
|
SpecForge-ext/cache/compiled_kernels/triton/0/4UGGVGDRILN3TLIINH7RHS2I4LFNQRBLGHV63VTW6KCPABX4L4FA/triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3.ttir
ADDED
|
@@ -0,0 +1,799 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#loc = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":18:0)
|
| 2 |
+
#loc1 = loc(unknown)
|
| 3 |
+
#loc2 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":41:67)
|
| 4 |
+
#loc23 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":662:12)
|
| 5 |
+
#loc28 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":634:73)
|
| 6 |
+
#loc32 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":538:51)
|
| 7 |
+
#loc37 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":539:53)
|
| 8 |
+
#loc46 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":548:50)
|
| 9 |
+
#loc51 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":551:51)
|
| 10 |
+
#loc70 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":45:26)
|
| 11 |
+
#loc80 = loc("in_ptr0"(#loc))
|
| 12 |
+
#loc81 = loc("out_ptr2"(#loc))
|
| 13 |
+
#loc82 = loc("out_ptr3"(#loc))
|
| 14 |
+
#loc83 = loc("xnumel"(#loc))
|
| 15 |
+
#loc84 = loc("r0_numel"(#loc))
|
| 16 |
+
#loc106 = loc(callsite(#loc23 at #loc2))
|
| 17 |
+
#loc113 = loc("ileft"(#loc32))
|
| 18 |
+
#loc117 = loc("iright"(#loc37))
|
| 19 |
+
#loc126 = loc("left_idx"(#loc46))
|
| 20 |
+
#loc131 = loc("right_idx"(#loc51))
|
| 21 |
+
#loc150 = loc("tmp11"(#loc70))
|
| 22 |
+
#loc157 = loc(callsite(#loc28 at #loc106))
|
| 23 |
+
#loc161 = loc(callsite(#loc1 at #loc150))
|
| 24 |
+
#loc165 = loc(callsite(#loc113 at #loc157))
|
| 25 |
+
#loc169 = loc(callsite(#loc117 at #loc157))
|
| 26 |
+
#loc177 = loc(callsite(#loc126 at #loc157))
|
| 27 |
+
#loc182 = loc(callsite(#loc131 at #loc157))
|
| 28 |
+
#loc202 = loc(callsite(#loc1 at #loc165))
|
| 29 |
+
#loc204 = loc(callsite(#loc1 at #loc169))
|
| 30 |
+
#loc207 = loc(callsite(#loc1 at #loc177))
|
| 31 |
+
#loc210 = loc(callsite(#loc1 at #loc182))
|
| 32 |
+
module {
|
| 33 |
+
tt.func public @triton_per_fused__to_copy_clone_slice_sort_sum_transpose_3(%in_ptr0: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("in_ptr0"(#loc)), %out_ptr2: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("out_ptr2"(#loc)), %out_ptr3: !tt.ptr<i32> {tt.divisibility = 16 : i32} loc("out_ptr3"(#loc)), %xnumel: i32 {tt.divisibility = 16 : i32} loc("xnumel"(#loc)), %r0_numel: i32 {tt.divisibility = 16 : i32} loc("r0_numel"(#loc))) attributes {noinline = false} {
|
| 34 |
+
%cst = arith.constant dense<1> : tensor<1x2x1xi32> loc(#loc85)
|
| 35 |
+
%cst_0 = arith.constant dense<0> : tensor<32x16xi32> loc(#loc1)
|
| 36 |
+
%tmp10 = arith.constant dense<0> : tensor<32x16xi64> loc(#loc86)
|
| 37 |
+
%tmp0 = arith.constant dense<272> : tensor<32x1xi32> loc(#loc87)
|
| 38 |
+
%tmp0_1 = arith.constant dense<17> : tensor<1x16xi32> loc(#loc88)
|
| 39 |
+
%cst_2 = arith.constant dense<16> : tensor<32x1xi32> loc(#loc1)
|
| 40 |
+
%xmask = arith.constant dense<128> : tensor<32x1xi32> loc(#loc89)
|
| 41 |
+
%c32_i32 = arith.constant 32 : i32 loc(#loc1)
|
| 42 |
+
%xoffset = tt.get_program_id x : i32 loc(#loc90)
|
| 43 |
+
%xoffset_3 = arith.muli %xoffset, %c32_i32 : i32 loc(#loc91)
|
| 44 |
+
%xindex = tt.make_range {end = 32 : i32, start = 0 : i32} : tensor<32xi32> loc(#loc92)
|
| 45 |
+
%xindex_4 = tt.expand_dims %xindex {axis = 1 : i32} : tensor<32xi32> -> tensor<32x1xi32> loc(#loc93)
|
| 46 |
+
%xindex_5 = tt.splat %xoffset_3 : i32 -> tensor<32x1xi32> loc(#loc94)
|
| 47 |
+
%xindex_6 = arith.addi %xindex_5, %xindex_4 : tensor<32x1xi32> loc(#loc94)
|
| 48 |
+
%xmask_7 = arith.cmpi slt, %xindex_6, %xmask : tensor<32x1xi32> loc(#loc89)
|
| 49 |
+
%r0_index = tt.make_range {end = 16 : i32, start = 0 : i32} : tensor<16xi32> loc(#loc95)
|
| 50 |
+
%r0_index_8 = tt.expand_dims %r0_index {axis = 0 : i32} : tensor<16xi32> -> tensor<1x16xi32> loc(#loc96)
|
| 51 |
+
%x0 = arith.remsi %xindex_6, %cst_2 : tensor<32x1xi32> loc(#loc97)
|
| 52 |
+
%x1 = arith.divsi %xindex_6, %cst_2 : tensor<32x1xi32> loc(#loc98)
|
| 53 |
+
%tmp0_9 = arith.muli %r0_index_8, %tmp0_1 : tensor<1x16xi32> loc(#loc88)
|
| 54 |
+
%tmp0_10 = tt.broadcast %x0 : tensor<32x1xi32> -> tensor<32x16xi32> loc(#loc99)
|
| 55 |
+
%tmp0_11 = tt.broadcast %tmp0_9 : tensor<1x16xi32> -> tensor<32x16xi32> loc(#loc99)
|
| 56 |
+
%tmp0_12 = arith.addi %tmp0_10, %tmp0_11 : tensor<32x16xi32> loc(#loc99)
|
| 57 |
+
%tmp0_13 = arith.muli %x1, %tmp0 : tensor<32x1xi32> loc(#loc87)
|
| 58 |
+
%tmp0_14 = tt.broadcast %tmp0_13 : tensor<32x1xi32> -> tensor<32x16xi32> loc(#loc100)
|
| 59 |
+
%tmp0_15 = arith.addi %tmp0_12, %tmp0_14 : tensor<32x16xi32> loc(#loc100)
|
| 60 |
+
%tmp0_16 = tt.splat %in_ptr0 : !tt.ptr<i32> -> tensor<32x16x!tt.ptr<i32>> loc(#loc101)
|
| 61 |
+
%tmp0_17 = tt.addptr %tmp0_16, %tmp0_15 : tensor<32x16x!tt.ptr<i32>>, tensor<32x16xi32> loc(#loc101)
|
| 62 |
+
%tmp0_18 = tt.broadcast %xmask_7 : tensor<32x1xi1> -> tensor<32x16xi1> loc(#loc102)
|
| 63 |
+
%tmp0_19 = tt.load %tmp0_17, %tmp0_18, %cst_0 : tensor<32x16x!tt.ptr<i32>> loc(#loc102)
|
| 64 |
+
%tmp2 = arith.trunci %r0_index_8 : tensor<1x16xi32> to tensor<1x16xi16> loc(#loc103)
|
| 65 |
+
%tmp4 = tt.broadcast %tmp2 : tensor<1x16xi16> -> tensor<32x16xi16> loc(#loc104)
|
| 66 |
+
%flip = tt.make_range {end = 2 : i32, start = 0 : i32} : tensor<2xi32> loc(#loc153)
|
| 67 |
+
%flip_20 = tt.expand_dims %flip {axis = 0 : i32} : tensor<2xi32> -> tensor<1x2xi32> loc(#loc154)
|
| 68 |
+
%flip_21 = tt.expand_dims %flip_20 {axis = 2 : i32} : tensor<1x2xi32> -> tensor<1x2x1xi32> loc(#loc154)
|
| 69 |
+
%flip_22 = tt.broadcast %flip_21 : tensor<1x2x1xi32> -> tensor<128x2x2xi32> loc(#loc155)
|
| 70 |
+
%flip_23 = tt.reshape %flip_22 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc156)
|
| 71 |
+
%y = tt.reshape %tmp0_19 : tensor<32x16xi32> -> tensor<256x2x1xi32> loc(#loc162)
|
| 72 |
+
%left_mask = arith.subi %cst, %flip_21 : tensor<1x2x1xi32> loc(#loc163)
|
| 73 |
+
%ileft = tt.broadcast %left_mask : tensor<1x2x1xi32> -> tensor<256x2x1xi32> loc(#loc164)
|
| 74 |
+
%ileft_24 = arith.muli %y, %ileft : tensor<256x2x1xi32> loc(#loc164)
|
| 75 |
+
%ileft_25 = "tt.reduce"(%ileft_24) <{axis = 1 : i32}> ({
|
| 76 |
+
^bb0(%ileft_377: i32 loc(callsite(#loc1 at #loc165)), %ileft_378: i32 loc(callsite(#loc1 at #loc165))):
|
| 77 |
+
%ileft_379 = arith.addi %ileft_377, %ileft_378 : i32 loc(#loc211)
|
| 78 |
+
tt.reduce.return %ileft_379 : i32 loc(#loc201)
|
| 79 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc201)
|
| 80 |
+
%ileft_26 = tt.expand_dims %ileft_25 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc166)
|
| 81 |
+
%ileft_27 = tt.broadcast %ileft_26 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc167)
|
| 82 |
+
%iright = tt.broadcast %flip_21 : tensor<1x2x1xi32> -> tensor<256x2x1xi32> loc(#loc168)
|
| 83 |
+
%iright_28 = arith.muli %y, %iright : tensor<256x2x1xi32> loc(#loc168)
|
| 84 |
+
%iright_29 = "tt.reduce"(%iright_28) <{axis = 1 : i32}> ({
|
| 85 |
+
^bb0(%iright_377: i32 loc(callsite(#loc1 at #loc169)), %iright_378: i32 loc(callsite(#loc1 at #loc169))):
|
| 86 |
+
%iright_379 = arith.addi %iright_377, %iright_378 : i32 loc(#loc212)
|
| 87 |
+
tt.reduce.return %iright_379 : i32 loc(#loc203)
|
| 88 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc203)
|
| 89 |
+
%iright_30 = tt.expand_dims %iright_29 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc170)
|
| 90 |
+
%iright_31 = tt.broadcast %iright_30 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc171)
|
| 91 |
+
%ileft_32 = tt.reshape %ileft_27 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc172)
|
| 92 |
+
%iright_33 = tt.reshape %iright_31 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc173)
|
| 93 |
+
%y_idx = tt.reshape %tmp4 : tensor<32x16xi16> -> tensor<256x2x1xi16> loc(#loc174)
|
| 94 |
+
%left_idx = arith.trunci %left_mask : tensor<1x2x1xi32> to tensor<1x2x1xi16> loc(#loc175)
|
| 95 |
+
%left_idx_34 = tt.broadcast %left_idx : tensor<1x2x1xi16> -> tensor<256x2x1xi16> loc(#loc176)
|
| 96 |
+
%left_idx_35 = arith.muli %y_idx, %left_idx_34 : tensor<256x2x1xi16> loc(#loc176)
|
| 97 |
+
%input = arith.extsi %left_idx_35 : tensor<256x2x1xi16> to tensor<256x2x1xi32> loc(#loc205)
|
| 98 |
+
%left_idx_36 = "tt.reduce"(%input) <{axis = 1 : i32}> ({
|
| 99 |
+
^bb0(%left_idx_377: i32 loc(callsite(#loc1 at #loc177)), %left_idx_378: i32 loc(callsite(#loc1 at #loc177))):
|
| 100 |
+
%left_idx_379 = arith.addi %left_idx_377, %left_idx_378 : i32 loc(#loc213)
|
| 101 |
+
tt.reduce.return %left_idx_379 : i32 loc(#loc206)
|
| 102 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc206)
|
| 103 |
+
%left_idx_37 = tt.expand_dims %left_idx_36 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc178)
|
| 104 |
+
%left_idx_38 = tt.broadcast %left_idx_37 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc179)
|
| 105 |
+
%right_idx = arith.trunci %flip_21 : tensor<1x2x1xi32> to tensor<1x2x1xi16> loc(#loc180)
|
| 106 |
+
%right_idx_39 = tt.broadcast %right_idx : tensor<1x2x1xi16> -> tensor<256x2x1xi16> loc(#loc181)
|
| 107 |
+
%right_idx_40 = arith.muli %y_idx, %right_idx_39 : tensor<256x2x1xi16> loc(#loc181)
|
| 108 |
+
%input_41 = arith.extsi %right_idx_40 : tensor<256x2x1xi16> to tensor<256x2x1xi32> loc(#loc208)
|
| 109 |
+
%right_idx_42 = "tt.reduce"(%input_41) <{axis = 1 : i32}> ({
|
| 110 |
+
^bb0(%right_idx_377: i32 loc(callsite(#loc1 at #loc182)), %right_idx_378: i32 loc(callsite(#loc1 at #loc182))):
|
| 111 |
+
%right_idx_379 = arith.addi %right_idx_377, %right_idx_378 : i32 loc(#loc214)
|
| 112 |
+
tt.reduce.return %right_idx_379 : i32 loc(#loc209)
|
| 113 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc209)
|
| 114 |
+
%right_idx_43 = tt.expand_dims %right_idx_42 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc183)
|
| 115 |
+
%right_idx_44 = tt.broadcast %right_idx_43 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc184)
|
| 116 |
+
%left_idx_45 = tt.reshape %left_idx_38 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc185)
|
| 117 |
+
%right_idx_46 = tt.reshape %right_idx_44 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc186)
|
| 118 |
+
%cond = arith.cmpi slt, %ileft_32, %iright_33 : tensor<32x16xi32> loc(#loc187)
|
| 119 |
+
%eq = arith.cmpi eq, %ileft_32, %iright_33 : tensor<32x16xi32> loc(#loc188)
|
| 120 |
+
%cond_47 = arith.cmpi sgt, %left_idx_45, %right_idx_46 : tensor<32x16xi32> loc(#loc189)
|
| 121 |
+
%cond_48 = arith.andi %eq, %cond_47 : tensor<32x16xi1> loc(#loc190)
|
| 122 |
+
%cond_49 = arith.ori %cond, %cond_48 : tensor<32x16xi1> loc(#loc191)
|
| 123 |
+
%cond_50 = arith.extui %cond_49 : tensor<32x16xi1> to tensor<32x16xi32> loc(#loc192)
|
| 124 |
+
%cond_51 = arith.xori %cond_50, %flip_23 : tensor<32x16xi32> loc(#loc192)
|
| 125 |
+
%cond_52 = arith.cmpi ne, %cond_51, %cst_0 : tensor<32x16xi32> loc(#loc193)
|
| 126 |
+
%ret = arith.xori %ileft_32, %iright_33 : tensor<32x16xi32> loc(#loc194)
|
| 127 |
+
%ret_53 = arith.select %cond_52, %ret, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc195)
|
| 128 |
+
%ret_54 = arith.xori %tmp0_19, %ret_53 : tensor<32x16xi32> loc(#loc196)
|
| 129 |
+
%new_idxs = arith.xori %left_idx_45, %right_idx_46 : tensor<32x16xi32> loc(#loc197)
|
| 130 |
+
%new_idxs_55 = arith.select %cond_52, %new_idxs, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc198)
|
| 131 |
+
%new_idxs_56 = arith.extsi %tmp2 : tensor<1x16xi16> to tensor<1x16xi32> loc(#loc199)
|
| 132 |
+
%new_idxs_57 = tt.broadcast %new_idxs_56 : tensor<1x16xi32> -> tensor<32x16xi32> loc(#loc199)
|
| 133 |
+
%new_idxs_58 = arith.xori %new_idxs_57, %new_idxs_55 : tensor<32x16xi32> loc(#loc199)
|
| 134 |
+
%flip_59 = tt.broadcast %flip_21 : tensor<1x2x1xi32> -> tensor<64x2x4xi32> loc(#loc155)
|
| 135 |
+
%flip_60 = tt.reshape %flip_59 : tensor<64x2x4xi32> -> tensor<32x16xi32> loc(#loc156)
|
| 136 |
+
%y_61 = tt.reshape %ret_54 : tensor<32x16xi32> -> tensor<128x2x2xi32> loc(#loc162)
|
| 137 |
+
%ileft_62 = tt.broadcast %left_mask : tensor<1x2x1xi32> -> tensor<128x2x2xi32> loc(#loc164)
|
| 138 |
+
%ileft_63 = arith.muli %y_61, %ileft_62 : tensor<128x2x2xi32> loc(#loc164)
|
| 139 |
+
%ileft_64 = "tt.reduce"(%ileft_63) <{axis = 1 : i32}> ({
|
| 140 |
+
^bb0(%ileft_377: i32 loc(callsite(#loc1 at #loc165)), %ileft_378: i32 loc(callsite(#loc1 at #loc165))):
|
| 141 |
+
%ileft_379 = arith.addi %ileft_377, %ileft_378 : i32 loc(#loc211)
|
| 142 |
+
tt.reduce.return %ileft_379 : i32 loc(#loc201)
|
| 143 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc201)
|
| 144 |
+
%ileft_65 = tt.expand_dims %ileft_64 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc166)
|
| 145 |
+
%ileft_66 = tt.broadcast %ileft_65 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc167)
|
| 146 |
+
%iright_67 = arith.muli %y_61, %flip_22 : tensor<128x2x2xi32> loc(#loc168)
|
| 147 |
+
%iright_68 = "tt.reduce"(%iright_67) <{axis = 1 : i32}> ({
|
| 148 |
+
^bb0(%iright_377: i32 loc(callsite(#loc1 at #loc169)), %iright_378: i32 loc(callsite(#loc1 at #loc169))):
|
| 149 |
+
%iright_379 = arith.addi %iright_377, %iright_378 : i32 loc(#loc212)
|
| 150 |
+
tt.reduce.return %iright_379 : i32 loc(#loc203)
|
| 151 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc203)
|
| 152 |
+
%iright_69 = tt.expand_dims %iright_68 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc170)
|
| 153 |
+
%iright_70 = tt.broadcast %iright_69 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc171)
|
| 154 |
+
%ileft_71 = tt.reshape %ileft_66 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc172)
|
| 155 |
+
%iright_72 = tt.reshape %iright_70 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc173)
|
| 156 |
+
%y_idx_73 = tt.reshape %new_idxs_58 : tensor<32x16xi32> -> tensor<128x2x2xi32> loc(#loc174)
|
| 157 |
+
%left_idx_74 = arith.muli %y_idx_73, %ileft_62 : tensor<128x2x2xi32> loc(#loc176)
|
| 158 |
+
%left_idx_75 = "tt.reduce"(%left_idx_74) <{axis = 1 : i32}> ({
|
| 159 |
+
^bb0(%left_idx_377: i32 loc(callsite(#loc1 at #loc177)), %left_idx_378: i32 loc(callsite(#loc1 at #loc177))):
|
| 160 |
+
%left_idx_379 = arith.addi %left_idx_377, %left_idx_378 : i32 loc(#loc213)
|
| 161 |
+
tt.reduce.return %left_idx_379 : i32 loc(#loc206)
|
| 162 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc206)
|
| 163 |
+
%left_idx_76 = tt.expand_dims %left_idx_75 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc178)
|
| 164 |
+
%left_idx_77 = tt.broadcast %left_idx_76 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc179)
|
| 165 |
+
%right_idx_78 = arith.muli %y_idx_73, %flip_22 : tensor<128x2x2xi32> loc(#loc181)
|
| 166 |
+
%right_idx_79 = "tt.reduce"(%right_idx_78) <{axis = 1 : i32}> ({
|
| 167 |
+
^bb0(%right_idx_377: i32 loc(callsite(#loc1 at #loc182)), %right_idx_378: i32 loc(callsite(#loc1 at #loc182))):
|
| 168 |
+
%right_idx_379 = arith.addi %right_idx_377, %right_idx_378 : i32 loc(#loc214)
|
| 169 |
+
tt.reduce.return %right_idx_379 : i32 loc(#loc209)
|
| 170 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc209)
|
| 171 |
+
%right_idx_80 = tt.expand_dims %right_idx_79 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc183)
|
| 172 |
+
%right_idx_81 = tt.broadcast %right_idx_80 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc184)
|
| 173 |
+
%left_idx_82 = tt.reshape %left_idx_77 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc185)
|
| 174 |
+
%right_idx_83 = tt.reshape %right_idx_81 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc186)
|
| 175 |
+
%cond_84 = arith.cmpi slt, %ileft_71, %iright_72 : tensor<32x16xi32> loc(#loc187)
|
| 176 |
+
%eq_85 = arith.cmpi eq, %ileft_71, %iright_72 : tensor<32x16xi32> loc(#loc188)
|
| 177 |
+
%cond_86 = arith.cmpi sgt, %left_idx_82, %right_idx_83 : tensor<32x16xi32> loc(#loc189)
|
| 178 |
+
%cond_87 = arith.andi %eq_85, %cond_86 : tensor<32x16xi1> loc(#loc190)
|
| 179 |
+
%cond_88 = arith.ori %cond_84, %cond_87 : tensor<32x16xi1> loc(#loc191)
|
| 180 |
+
%cond_89 = arith.extui %cond_88 : tensor<32x16xi1> to tensor<32x16xi32> loc(#loc192)
|
| 181 |
+
%cond_90 = arith.xori %cond_89, %flip_60 : tensor<32x16xi32> loc(#loc192)
|
| 182 |
+
%cond_91 = arith.cmpi ne, %cond_90, %cst_0 : tensor<32x16xi32> loc(#loc193)
|
| 183 |
+
%ret_92 = arith.xori %ileft_71, %iright_72 : tensor<32x16xi32> loc(#loc194)
|
| 184 |
+
%ret_93 = arith.select %cond_91, %ret_92, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc195)
|
| 185 |
+
%ret_94 = arith.xori %ret_54, %ret_93 : tensor<32x16xi32> loc(#loc196)
|
| 186 |
+
%new_idxs_95 = arith.xori %left_idx_82, %right_idx_83 : tensor<32x16xi32> loc(#loc197)
|
| 187 |
+
%new_idxs_96 = arith.select %cond_91, %new_idxs_95, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc198)
|
| 188 |
+
%new_idxs_97 = arith.xori %new_idxs_58, %new_idxs_96 : tensor<32x16xi32> loc(#loc199)
|
| 189 |
+
%y_98 = tt.reshape %ret_94 : tensor<32x16xi32> -> tensor<256x2x1xi32> loc(#loc162)
|
| 190 |
+
%ileft_99 = arith.muli %y_98, %ileft : tensor<256x2x1xi32> loc(#loc164)
|
| 191 |
+
%ileft_100 = "tt.reduce"(%ileft_99) <{axis = 1 : i32}> ({
|
| 192 |
+
^bb0(%ileft_377: i32 loc(callsite(#loc1 at #loc165)), %ileft_378: i32 loc(callsite(#loc1 at #loc165))):
|
| 193 |
+
%ileft_379 = arith.addi %ileft_377, %ileft_378 : i32 loc(#loc211)
|
| 194 |
+
tt.reduce.return %ileft_379 : i32 loc(#loc201)
|
| 195 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc201)
|
| 196 |
+
%ileft_101 = tt.expand_dims %ileft_100 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc166)
|
| 197 |
+
%ileft_102 = tt.broadcast %ileft_101 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc167)
|
| 198 |
+
%iright_103 = arith.muli %y_98, %iright : tensor<256x2x1xi32> loc(#loc168)
|
| 199 |
+
%iright_104 = "tt.reduce"(%iright_103) <{axis = 1 : i32}> ({
|
| 200 |
+
^bb0(%iright_377: i32 loc(callsite(#loc1 at #loc169)), %iright_378: i32 loc(callsite(#loc1 at #loc169))):
|
| 201 |
+
%iright_379 = arith.addi %iright_377, %iright_378 : i32 loc(#loc212)
|
| 202 |
+
tt.reduce.return %iright_379 : i32 loc(#loc203)
|
| 203 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc203)
|
| 204 |
+
%iright_105 = tt.expand_dims %iright_104 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc170)
|
| 205 |
+
%iright_106 = tt.broadcast %iright_105 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc171)
|
| 206 |
+
%ileft_107 = tt.reshape %ileft_102 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc172)
|
| 207 |
+
%iright_108 = tt.reshape %iright_106 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc173)
|
| 208 |
+
%y_idx_109 = tt.reshape %new_idxs_97 : tensor<32x16xi32> -> tensor<256x2x1xi32> loc(#loc174)
|
| 209 |
+
%left_idx_110 = arith.muli %y_idx_109, %ileft : tensor<256x2x1xi32> loc(#loc176)
|
| 210 |
+
%left_idx_111 = "tt.reduce"(%left_idx_110) <{axis = 1 : i32}> ({
|
| 211 |
+
^bb0(%left_idx_377: i32 loc(callsite(#loc1 at #loc177)), %left_idx_378: i32 loc(callsite(#loc1 at #loc177))):
|
| 212 |
+
%left_idx_379 = arith.addi %left_idx_377, %left_idx_378 : i32 loc(#loc213)
|
| 213 |
+
tt.reduce.return %left_idx_379 : i32 loc(#loc206)
|
| 214 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc206)
|
| 215 |
+
%left_idx_112 = tt.expand_dims %left_idx_111 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc178)
|
| 216 |
+
%left_idx_113 = tt.broadcast %left_idx_112 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc179)
|
| 217 |
+
%right_idx_114 = arith.muli %y_idx_109, %iright : tensor<256x2x1xi32> loc(#loc181)
|
| 218 |
+
%right_idx_115 = "tt.reduce"(%right_idx_114) <{axis = 1 : i32}> ({
|
| 219 |
+
^bb0(%right_idx_377: i32 loc(callsite(#loc1 at #loc182)), %right_idx_378: i32 loc(callsite(#loc1 at #loc182))):
|
| 220 |
+
%right_idx_379 = arith.addi %right_idx_377, %right_idx_378 : i32 loc(#loc214)
|
| 221 |
+
tt.reduce.return %right_idx_379 : i32 loc(#loc209)
|
| 222 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc209)
|
| 223 |
+
%right_idx_116 = tt.expand_dims %right_idx_115 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc183)
|
| 224 |
+
%right_idx_117 = tt.broadcast %right_idx_116 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc184)
|
| 225 |
+
%left_idx_118 = tt.reshape %left_idx_113 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc185)
|
| 226 |
+
%right_idx_119 = tt.reshape %right_idx_117 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc186)
|
| 227 |
+
%cond_120 = arith.cmpi slt, %ileft_107, %iright_108 : tensor<32x16xi32> loc(#loc187)
|
| 228 |
+
%eq_121 = arith.cmpi eq, %ileft_107, %iright_108 : tensor<32x16xi32> loc(#loc188)
|
| 229 |
+
%cond_122 = arith.cmpi sgt, %left_idx_118, %right_idx_119 : tensor<32x16xi32> loc(#loc189)
|
| 230 |
+
%cond_123 = arith.andi %eq_121, %cond_122 : tensor<32x16xi1> loc(#loc190)
|
| 231 |
+
%cond_124 = arith.ori %cond_120, %cond_123 : tensor<32x16xi1> loc(#loc191)
|
| 232 |
+
%cond_125 = arith.extui %cond_124 : tensor<32x16xi1> to tensor<32x16xi32> loc(#loc192)
|
| 233 |
+
%cond_126 = arith.xori %cond_125, %flip_60 : tensor<32x16xi32> loc(#loc192)
|
| 234 |
+
%cond_127 = arith.cmpi ne, %cond_126, %cst_0 : tensor<32x16xi32> loc(#loc193)
|
| 235 |
+
%ret_128 = arith.xori %ileft_107, %iright_108 : tensor<32x16xi32> loc(#loc194)
|
| 236 |
+
%ret_129 = arith.select %cond_127, %ret_128, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc195)
|
| 237 |
+
%ret_130 = arith.xori %ret_94, %ret_129 : tensor<32x16xi32> loc(#loc196)
|
| 238 |
+
%new_idxs_131 = arith.xori %left_idx_118, %right_idx_119 : tensor<32x16xi32> loc(#loc197)
|
| 239 |
+
%new_idxs_132 = arith.select %cond_127, %new_idxs_131, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc198)
|
| 240 |
+
%new_idxs_133 = arith.xori %new_idxs_97, %new_idxs_132 : tensor<32x16xi32> loc(#loc199)
|
| 241 |
+
%flip_134 = tt.broadcast %flip_21 : tensor<1x2x1xi32> -> tensor<32x2x8xi32> loc(#loc155)
|
| 242 |
+
%flip_135 = tt.reshape %flip_134 : tensor<32x2x8xi32> -> tensor<32x16xi32> loc(#loc156)
|
| 243 |
+
%y_136 = tt.reshape %ret_130 : tensor<32x16xi32> -> tensor<64x2x4xi32> loc(#loc162)
|
| 244 |
+
%ileft_137 = tt.broadcast %left_mask : tensor<1x2x1xi32> -> tensor<64x2x4xi32> loc(#loc164)
|
| 245 |
+
%ileft_138 = arith.muli %y_136, %ileft_137 : tensor<64x2x4xi32> loc(#loc164)
|
| 246 |
+
%ileft_139 = "tt.reduce"(%ileft_138) <{axis = 1 : i32}> ({
|
| 247 |
+
^bb0(%ileft_377: i32 loc(callsite(#loc1 at #loc165)), %ileft_378: i32 loc(callsite(#loc1 at #loc165))):
|
| 248 |
+
%ileft_379 = arith.addi %ileft_377, %ileft_378 : i32 loc(#loc211)
|
| 249 |
+
tt.reduce.return %ileft_379 : i32 loc(#loc201)
|
| 250 |
+
}) : (tensor<64x2x4xi32>) -> tensor<64x4xi32> loc(#loc201)
|
| 251 |
+
%ileft_140 = tt.expand_dims %ileft_139 {axis = 1 : i32} : tensor<64x4xi32> -> tensor<64x1x4xi32> loc(#loc166)
|
| 252 |
+
%ileft_141 = tt.broadcast %ileft_140 : tensor<64x1x4xi32> -> tensor<64x2x4xi32> loc(#loc167)
|
| 253 |
+
%iright_142 = arith.muli %y_136, %flip_59 : tensor<64x2x4xi32> loc(#loc168)
|
| 254 |
+
%iright_143 = "tt.reduce"(%iright_142) <{axis = 1 : i32}> ({
|
| 255 |
+
^bb0(%iright_377: i32 loc(callsite(#loc1 at #loc169)), %iright_378: i32 loc(callsite(#loc1 at #loc169))):
|
| 256 |
+
%iright_379 = arith.addi %iright_377, %iright_378 : i32 loc(#loc212)
|
| 257 |
+
tt.reduce.return %iright_379 : i32 loc(#loc203)
|
| 258 |
+
}) : (tensor<64x2x4xi32>) -> tensor<64x4xi32> loc(#loc203)
|
| 259 |
+
%iright_144 = tt.expand_dims %iright_143 {axis = 1 : i32} : tensor<64x4xi32> -> tensor<64x1x4xi32> loc(#loc170)
|
| 260 |
+
%iright_145 = tt.broadcast %iright_144 : tensor<64x1x4xi32> -> tensor<64x2x4xi32> loc(#loc171)
|
| 261 |
+
%ileft_146 = tt.reshape %ileft_141 : tensor<64x2x4xi32> -> tensor<32x16xi32> loc(#loc172)
|
| 262 |
+
%iright_147 = tt.reshape %iright_145 : tensor<64x2x4xi32> -> tensor<32x16xi32> loc(#loc173)
|
| 263 |
+
%y_idx_148 = tt.reshape %new_idxs_133 : tensor<32x16xi32> -> tensor<64x2x4xi32> loc(#loc174)
|
| 264 |
+
%left_idx_149 = arith.muli %y_idx_148, %ileft_137 : tensor<64x2x4xi32> loc(#loc176)
|
| 265 |
+
%left_idx_150 = "tt.reduce"(%left_idx_149) <{axis = 1 : i32}> ({
|
| 266 |
+
^bb0(%left_idx_377: i32 loc(callsite(#loc1 at #loc177)), %left_idx_378: i32 loc(callsite(#loc1 at #loc177))):
|
| 267 |
+
%left_idx_379 = arith.addi %left_idx_377, %left_idx_378 : i32 loc(#loc213)
|
| 268 |
+
tt.reduce.return %left_idx_379 : i32 loc(#loc206)
|
| 269 |
+
}) : (tensor<64x2x4xi32>) -> tensor<64x4xi32> loc(#loc206)
|
| 270 |
+
%left_idx_151 = tt.expand_dims %left_idx_150 {axis = 1 : i32} : tensor<64x4xi32> -> tensor<64x1x4xi32> loc(#loc178)
|
| 271 |
+
%left_idx_152 = tt.broadcast %left_idx_151 : tensor<64x1x4xi32> -> tensor<64x2x4xi32> loc(#loc179)
|
| 272 |
+
%right_idx_153 = arith.muli %y_idx_148, %flip_59 : tensor<64x2x4xi32> loc(#loc181)
|
| 273 |
+
%right_idx_154 = "tt.reduce"(%right_idx_153) <{axis = 1 : i32}> ({
|
| 274 |
+
^bb0(%right_idx_377: i32 loc(callsite(#loc1 at #loc182)), %right_idx_378: i32 loc(callsite(#loc1 at #loc182))):
|
| 275 |
+
%right_idx_379 = arith.addi %right_idx_377, %right_idx_378 : i32 loc(#loc214)
|
| 276 |
+
tt.reduce.return %right_idx_379 : i32 loc(#loc209)
|
| 277 |
+
}) : (tensor<64x2x4xi32>) -> tensor<64x4xi32> loc(#loc209)
|
| 278 |
+
%right_idx_155 = tt.expand_dims %right_idx_154 {axis = 1 : i32} : tensor<64x4xi32> -> tensor<64x1x4xi32> loc(#loc183)
|
| 279 |
+
%right_idx_156 = tt.broadcast %right_idx_155 : tensor<64x1x4xi32> -> tensor<64x2x4xi32> loc(#loc184)
|
| 280 |
+
%left_idx_157 = tt.reshape %left_idx_152 : tensor<64x2x4xi32> -> tensor<32x16xi32> loc(#loc185)
|
| 281 |
+
%right_idx_158 = tt.reshape %right_idx_156 : tensor<64x2x4xi32> -> tensor<32x16xi32> loc(#loc186)
|
| 282 |
+
%cond_159 = arith.cmpi slt, %ileft_146, %iright_147 : tensor<32x16xi32> loc(#loc187)
|
| 283 |
+
%eq_160 = arith.cmpi eq, %ileft_146, %iright_147 : tensor<32x16xi32> loc(#loc188)
|
| 284 |
+
%cond_161 = arith.cmpi sgt, %left_idx_157, %right_idx_158 : tensor<32x16xi32> loc(#loc189)
|
| 285 |
+
%cond_162 = arith.andi %eq_160, %cond_161 : tensor<32x16xi1> loc(#loc190)
|
| 286 |
+
%cond_163 = arith.ori %cond_159, %cond_162 : tensor<32x16xi1> loc(#loc191)
|
| 287 |
+
%cond_164 = arith.extui %cond_163 : tensor<32x16xi1> to tensor<32x16xi32> loc(#loc192)
|
| 288 |
+
%cond_165 = arith.xori %cond_164, %flip_135 : tensor<32x16xi32> loc(#loc192)
|
| 289 |
+
%cond_166 = arith.cmpi ne, %cond_165, %cst_0 : tensor<32x16xi32> loc(#loc193)
|
| 290 |
+
%ret_167 = arith.xori %ileft_146, %iright_147 : tensor<32x16xi32> loc(#loc194)
|
| 291 |
+
%ret_168 = arith.select %cond_166, %ret_167, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc195)
|
| 292 |
+
%ret_169 = arith.xori %ret_130, %ret_168 : tensor<32x16xi32> loc(#loc196)
|
| 293 |
+
%new_idxs_170 = arith.xori %left_idx_157, %right_idx_158 : tensor<32x16xi32> loc(#loc197)
|
| 294 |
+
%new_idxs_171 = arith.select %cond_166, %new_idxs_170, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc198)
|
| 295 |
+
%new_idxs_172 = arith.xori %new_idxs_133, %new_idxs_171 : tensor<32x16xi32> loc(#loc199)
|
| 296 |
+
%y_173 = tt.reshape %ret_169 : tensor<32x16xi32> -> tensor<128x2x2xi32> loc(#loc162)
|
| 297 |
+
%ileft_174 = arith.muli %y_173, %ileft_62 : tensor<128x2x2xi32> loc(#loc164)
|
| 298 |
+
%ileft_175 = "tt.reduce"(%ileft_174) <{axis = 1 : i32}> ({
|
| 299 |
+
^bb0(%ileft_377: i32 loc(callsite(#loc1 at #loc165)), %ileft_378: i32 loc(callsite(#loc1 at #loc165))):
|
| 300 |
+
%ileft_379 = arith.addi %ileft_377, %ileft_378 : i32 loc(#loc211)
|
| 301 |
+
tt.reduce.return %ileft_379 : i32 loc(#loc201)
|
| 302 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc201)
|
| 303 |
+
%ileft_176 = tt.expand_dims %ileft_175 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc166)
|
| 304 |
+
%ileft_177 = tt.broadcast %ileft_176 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc167)
|
| 305 |
+
%iright_178 = arith.muli %y_173, %flip_22 : tensor<128x2x2xi32> loc(#loc168)
|
| 306 |
+
%iright_179 = "tt.reduce"(%iright_178) <{axis = 1 : i32}> ({
|
| 307 |
+
^bb0(%iright_377: i32 loc(callsite(#loc1 at #loc169)), %iright_378: i32 loc(callsite(#loc1 at #loc169))):
|
| 308 |
+
%iright_379 = arith.addi %iright_377, %iright_378 : i32 loc(#loc212)
|
| 309 |
+
tt.reduce.return %iright_379 : i32 loc(#loc203)
|
| 310 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc203)
|
| 311 |
+
%iright_180 = tt.expand_dims %iright_179 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc170)
|
| 312 |
+
%iright_181 = tt.broadcast %iright_180 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc171)
|
| 313 |
+
%ileft_182 = tt.reshape %ileft_177 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc172)
|
| 314 |
+
%iright_183 = tt.reshape %iright_181 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc173)
|
| 315 |
+
%y_idx_184 = tt.reshape %new_idxs_172 : tensor<32x16xi32> -> tensor<128x2x2xi32> loc(#loc174)
|
| 316 |
+
%left_idx_185 = arith.muli %y_idx_184, %ileft_62 : tensor<128x2x2xi32> loc(#loc176)
|
| 317 |
+
%left_idx_186 = "tt.reduce"(%left_idx_185) <{axis = 1 : i32}> ({
|
| 318 |
+
^bb0(%left_idx_377: i32 loc(callsite(#loc1 at #loc177)), %left_idx_378: i32 loc(callsite(#loc1 at #loc177))):
|
| 319 |
+
%left_idx_379 = arith.addi %left_idx_377, %left_idx_378 : i32 loc(#loc213)
|
| 320 |
+
tt.reduce.return %left_idx_379 : i32 loc(#loc206)
|
| 321 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc206)
|
| 322 |
+
%left_idx_187 = tt.expand_dims %left_idx_186 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc178)
|
| 323 |
+
%left_idx_188 = tt.broadcast %left_idx_187 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc179)
|
| 324 |
+
%right_idx_189 = arith.muli %y_idx_184, %flip_22 : tensor<128x2x2xi32> loc(#loc181)
|
| 325 |
+
%right_idx_190 = "tt.reduce"(%right_idx_189) <{axis = 1 : i32}> ({
|
| 326 |
+
^bb0(%right_idx_377: i32 loc(callsite(#loc1 at #loc182)), %right_idx_378: i32 loc(callsite(#loc1 at #loc182))):
|
| 327 |
+
%right_idx_379 = arith.addi %right_idx_377, %right_idx_378 : i32 loc(#loc214)
|
| 328 |
+
tt.reduce.return %right_idx_379 : i32 loc(#loc209)
|
| 329 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc209)
|
| 330 |
+
%right_idx_191 = tt.expand_dims %right_idx_190 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc183)
|
| 331 |
+
%right_idx_192 = tt.broadcast %right_idx_191 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc184)
|
| 332 |
+
%left_idx_193 = tt.reshape %left_idx_188 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc185)
|
| 333 |
+
%right_idx_194 = tt.reshape %right_idx_192 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc186)
|
| 334 |
+
%cond_195 = arith.cmpi slt, %ileft_182, %iright_183 : tensor<32x16xi32> loc(#loc187)
|
| 335 |
+
%eq_196 = arith.cmpi eq, %ileft_182, %iright_183 : tensor<32x16xi32> loc(#loc188)
|
| 336 |
+
%cond_197 = arith.cmpi sgt, %left_idx_193, %right_idx_194 : tensor<32x16xi32> loc(#loc189)
|
| 337 |
+
%cond_198 = arith.andi %eq_196, %cond_197 : tensor<32x16xi1> loc(#loc190)
|
| 338 |
+
%cond_199 = arith.ori %cond_195, %cond_198 : tensor<32x16xi1> loc(#loc191)
|
| 339 |
+
%cond_200 = arith.extui %cond_199 : tensor<32x16xi1> to tensor<32x16xi32> loc(#loc192)
|
| 340 |
+
%cond_201 = arith.xori %cond_200, %flip_135 : tensor<32x16xi32> loc(#loc192)
|
| 341 |
+
%cond_202 = arith.cmpi ne, %cond_201, %cst_0 : tensor<32x16xi32> loc(#loc193)
|
| 342 |
+
%ret_203 = arith.xori %ileft_182, %iright_183 : tensor<32x16xi32> loc(#loc194)
|
| 343 |
+
%ret_204 = arith.select %cond_202, %ret_203, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc195)
|
| 344 |
+
%ret_205 = arith.xori %ret_169, %ret_204 : tensor<32x16xi32> loc(#loc196)
|
| 345 |
+
%new_idxs_206 = arith.xori %left_idx_193, %right_idx_194 : tensor<32x16xi32> loc(#loc197)
|
| 346 |
+
%new_idxs_207 = arith.select %cond_202, %new_idxs_206, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc198)
|
| 347 |
+
%new_idxs_208 = arith.xori %new_idxs_172, %new_idxs_207 : tensor<32x16xi32> loc(#loc199)
|
| 348 |
+
%y_209 = tt.reshape %ret_205 : tensor<32x16xi32> -> tensor<256x2x1xi32> loc(#loc162)
|
| 349 |
+
%ileft_210 = arith.muli %y_209, %ileft : tensor<256x2x1xi32> loc(#loc164)
|
| 350 |
+
%ileft_211 = "tt.reduce"(%ileft_210) <{axis = 1 : i32}> ({
|
| 351 |
+
^bb0(%ileft_377: i32 loc(callsite(#loc1 at #loc165)), %ileft_378: i32 loc(callsite(#loc1 at #loc165))):
|
| 352 |
+
%ileft_379 = arith.addi %ileft_377, %ileft_378 : i32 loc(#loc211)
|
| 353 |
+
tt.reduce.return %ileft_379 : i32 loc(#loc201)
|
| 354 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc201)
|
| 355 |
+
%ileft_212 = tt.expand_dims %ileft_211 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc166)
|
| 356 |
+
%ileft_213 = tt.broadcast %ileft_212 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc167)
|
| 357 |
+
%iright_214 = arith.muli %y_209, %iright : tensor<256x2x1xi32> loc(#loc168)
|
| 358 |
+
%iright_215 = "tt.reduce"(%iright_214) <{axis = 1 : i32}> ({
|
| 359 |
+
^bb0(%iright_377: i32 loc(callsite(#loc1 at #loc169)), %iright_378: i32 loc(callsite(#loc1 at #loc169))):
|
| 360 |
+
%iright_379 = arith.addi %iright_377, %iright_378 : i32 loc(#loc212)
|
| 361 |
+
tt.reduce.return %iright_379 : i32 loc(#loc203)
|
| 362 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc203)
|
| 363 |
+
%iright_216 = tt.expand_dims %iright_215 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc170)
|
| 364 |
+
%iright_217 = tt.broadcast %iright_216 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc171)
|
| 365 |
+
%ileft_218 = tt.reshape %ileft_213 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc172)
|
| 366 |
+
%iright_219 = tt.reshape %iright_217 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc173)
|
| 367 |
+
%y_idx_220 = tt.reshape %new_idxs_208 : tensor<32x16xi32> -> tensor<256x2x1xi32> loc(#loc174)
|
| 368 |
+
%left_idx_221 = arith.muli %y_idx_220, %ileft : tensor<256x2x1xi32> loc(#loc176)
|
| 369 |
+
%left_idx_222 = "tt.reduce"(%left_idx_221) <{axis = 1 : i32}> ({
|
| 370 |
+
^bb0(%left_idx_377: i32 loc(callsite(#loc1 at #loc177)), %left_idx_378: i32 loc(callsite(#loc1 at #loc177))):
|
| 371 |
+
%left_idx_379 = arith.addi %left_idx_377, %left_idx_378 : i32 loc(#loc213)
|
| 372 |
+
tt.reduce.return %left_idx_379 : i32 loc(#loc206)
|
| 373 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc206)
|
| 374 |
+
%left_idx_223 = tt.expand_dims %left_idx_222 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc178)
|
| 375 |
+
%left_idx_224 = tt.broadcast %left_idx_223 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc179)
|
| 376 |
+
%right_idx_225 = arith.muli %y_idx_220, %iright : tensor<256x2x1xi32> loc(#loc181)
|
| 377 |
+
%right_idx_226 = "tt.reduce"(%right_idx_225) <{axis = 1 : i32}> ({
|
| 378 |
+
^bb0(%right_idx_377: i32 loc(callsite(#loc1 at #loc182)), %right_idx_378: i32 loc(callsite(#loc1 at #loc182))):
|
| 379 |
+
%right_idx_379 = arith.addi %right_idx_377, %right_idx_378 : i32 loc(#loc214)
|
| 380 |
+
tt.reduce.return %right_idx_379 : i32 loc(#loc209)
|
| 381 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc209)
|
| 382 |
+
%right_idx_227 = tt.expand_dims %right_idx_226 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc183)
|
| 383 |
+
%right_idx_228 = tt.broadcast %right_idx_227 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc184)
|
| 384 |
+
%left_idx_229 = tt.reshape %left_idx_224 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc185)
|
| 385 |
+
%right_idx_230 = tt.reshape %right_idx_228 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc186)
|
| 386 |
+
%cond_231 = arith.cmpi slt, %ileft_218, %iright_219 : tensor<32x16xi32> loc(#loc187)
|
| 387 |
+
%eq_232 = arith.cmpi eq, %ileft_218, %iright_219 : tensor<32x16xi32> loc(#loc188)
|
| 388 |
+
%cond_233 = arith.cmpi sgt, %left_idx_229, %right_idx_230 : tensor<32x16xi32> loc(#loc189)
|
| 389 |
+
%cond_234 = arith.andi %eq_232, %cond_233 : tensor<32x16xi1> loc(#loc190)
|
| 390 |
+
%cond_235 = arith.ori %cond_231, %cond_234 : tensor<32x16xi1> loc(#loc191)
|
| 391 |
+
%cond_236 = arith.extui %cond_235 : tensor<32x16xi1> to tensor<32x16xi32> loc(#loc192)
|
| 392 |
+
%cond_237 = arith.xori %cond_236, %flip_135 : tensor<32x16xi32> loc(#loc192)
|
| 393 |
+
%cond_238 = arith.cmpi ne, %cond_237, %cst_0 : tensor<32x16xi32> loc(#loc193)
|
| 394 |
+
%ret_239 = arith.xori %ileft_218, %iright_219 : tensor<32x16xi32> loc(#loc194)
|
| 395 |
+
%ret_240 = arith.select %cond_238, %ret_239, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc195)
|
| 396 |
+
%ret_241 = arith.xori %ret_205, %ret_240 : tensor<32x16xi32> loc(#loc196)
|
| 397 |
+
%new_idxs_242 = arith.xori %left_idx_229, %right_idx_230 : tensor<32x16xi32> loc(#loc197)
|
| 398 |
+
%new_idxs_243 = arith.select %cond_238, %new_idxs_242, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc198)
|
| 399 |
+
%new_idxs_244 = arith.xori %new_idxs_208, %new_idxs_243 : tensor<32x16xi32> loc(#loc199)
|
| 400 |
+
%y_245 = tt.reshape %ret_241 : tensor<32x16xi32> -> tensor<32x2x8xi32> loc(#loc162)
|
| 401 |
+
%ileft_246 = tt.broadcast %left_mask : tensor<1x2x1xi32> -> tensor<32x2x8xi32> loc(#loc164)
|
| 402 |
+
%ileft_247 = arith.muli %y_245, %ileft_246 : tensor<32x2x8xi32> loc(#loc164)
|
| 403 |
+
%ileft_248 = "tt.reduce"(%ileft_247) <{axis = 1 : i32}> ({
|
| 404 |
+
^bb0(%ileft_377: i32 loc(callsite(#loc1 at #loc165)), %ileft_378: i32 loc(callsite(#loc1 at #loc165))):
|
| 405 |
+
%ileft_379 = arith.addi %ileft_377, %ileft_378 : i32 loc(#loc211)
|
| 406 |
+
tt.reduce.return %ileft_379 : i32 loc(#loc201)
|
| 407 |
+
}) : (tensor<32x2x8xi32>) -> tensor<32x8xi32> loc(#loc201)
|
| 408 |
+
%ileft_249 = tt.expand_dims %ileft_248 {axis = 1 : i32} : tensor<32x8xi32> -> tensor<32x1x8xi32> loc(#loc166)
|
| 409 |
+
%ileft_250 = tt.broadcast %ileft_249 : tensor<32x1x8xi32> -> tensor<32x2x8xi32> loc(#loc167)
|
| 410 |
+
%iright_251 = arith.muli %y_245, %flip_134 : tensor<32x2x8xi32> loc(#loc168)
|
| 411 |
+
%iright_252 = "tt.reduce"(%iright_251) <{axis = 1 : i32}> ({
|
| 412 |
+
^bb0(%iright_377: i32 loc(callsite(#loc1 at #loc169)), %iright_378: i32 loc(callsite(#loc1 at #loc169))):
|
| 413 |
+
%iright_379 = arith.addi %iright_377, %iright_378 : i32 loc(#loc212)
|
| 414 |
+
tt.reduce.return %iright_379 : i32 loc(#loc203)
|
| 415 |
+
}) : (tensor<32x2x8xi32>) -> tensor<32x8xi32> loc(#loc203)
|
| 416 |
+
%iright_253 = tt.expand_dims %iright_252 {axis = 1 : i32} : tensor<32x8xi32> -> tensor<32x1x8xi32> loc(#loc170)
|
| 417 |
+
%iright_254 = tt.broadcast %iright_253 : tensor<32x1x8xi32> -> tensor<32x2x8xi32> loc(#loc171)
|
| 418 |
+
%ileft_255 = tt.reshape %ileft_250 : tensor<32x2x8xi32> -> tensor<32x16xi32> loc(#loc172)
|
| 419 |
+
%iright_256 = tt.reshape %iright_254 : tensor<32x2x8xi32> -> tensor<32x16xi32> loc(#loc173)
|
| 420 |
+
%y_idx_257 = tt.reshape %new_idxs_244 : tensor<32x16xi32> -> tensor<32x2x8xi32> loc(#loc174)
|
| 421 |
+
%left_idx_258 = arith.muli %y_idx_257, %ileft_246 : tensor<32x2x8xi32> loc(#loc176)
|
| 422 |
+
%left_idx_259 = "tt.reduce"(%left_idx_258) <{axis = 1 : i32}> ({
|
| 423 |
+
^bb0(%left_idx_377: i32 loc(callsite(#loc1 at #loc177)), %left_idx_378: i32 loc(callsite(#loc1 at #loc177))):
|
| 424 |
+
%left_idx_379 = arith.addi %left_idx_377, %left_idx_378 : i32 loc(#loc213)
|
| 425 |
+
tt.reduce.return %left_idx_379 : i32 loc(#loc206)
|
| 426 |
+
}) : (tensor<32x2x8xi32>) -> tensor<32x8xi32> loc(#loc206)
|
| 427 |
+
%left_idx_260 = tt.expand_dims %left_idx_259 {axis = 1 : i32} : tensor<32x8xi32> -> tensor<32x1x8xi32> loc(#loc178)
|
| 428 |
+
%left_idx_261 = tt.broadcast %left_idx_260 : tensor<32x1x8xi32> -> tensor<32x2x8xi32> loc(#loc179)
|
| 429 |
+
%right_idx_262 = arith.muli %y_idx_257, %flip_134 : tensor<32x2x8xi32> loc(#loc181)
|
| 430 |
+
%right_idx_263 = "tt.reduce"(%right_idx_262) <{axis = 1 : i32}> ({
|
| 431 |
+
^bb0(%right_idx_377: i32 loc(callsite(#loc1 at #loc182)), %right_idx_378: i32 loc(callsite(#loc1 at #loc182))):
|
| 432 |
+
%right_idx_379 = arith.addi %right_idx_377, %right_idx_378 : i32 loc(#loc214)
|
| 433 |
+
tt.reduce.return %right_idx_379 : i32 loc(#loc209)
|
| 434 |
+
}) : (tensor<32x2x8xi32>) -> tensor<32x8xi32> loc(#loc209)
|
| 435 |
+
%right_idx_264 = tt.expand_dims %right_idx_263 {axis = 1 : i32} : tensor<32x8xi32> -> tensor<32x1x8xi32> loc(#loc183)
|
| 436 |
+
%right_idx_265 = tt.broadcast %right_idx_264 : tensor<32x1x8xi32> -> tensor<32x2x8xi32> loc(#loc184)
|
| 437 |
+
%left_idx_266 = tt.reshape %left_idx_261 : tensor<32x2x8xi32> -> tensor<32x16xi32> loc(#loc185)
|
| 438 |
+
%right_idx_267 = tt.reshape %right_idx_265 : tensor<32x2x8xi32> -> tensor<32x16xi32> loc(#loc186)
|
| 439 |
+
%cond_268 = arith.cmpi slt, %ileft_255, %iright_256 : tensor<32x16xi32> loc(#loc187)
|
| 440 |
+
%eq_269 = arith.cmpi eq, %ileft_255, %iright_256 : tensor<32x16xi32> loc(#loc188)
|
| 441 |
+
%cond_270 = arith.cmpi sgt, %left_idx_266, %right_idx_267 : tensor<32x16xi32> loc(#loc189)
|
| 442 |
+
%cond_271 = arith.andi %eq_269, %cond_270 : tensor<32x16xi1> loc(#loc190)
|
| 443 |
+
%cond_272 = arith.ori %cond_268, %cond_271 : tensor<32x16xi1> loc(#loc191)
|
| 444 |
+
%ret_273 = arith.xori %ileft_255, %iright_256 : tensor<32x16xi32> loc(#loc194)
|
| 445 |
+
%ret_274 = arith.select %cond_272, %ret_273, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc195)
|
| 446 |
+
%ret_275 = arith.xori %ret_241, %ret_274 : tensor<32x16xi32> loc(#loc196)
|
| 447 |
+
%new_idxs_276 = arith.xori %left_idx_266, %right_idx_267 : tensor<32x16xi32> loc(#loc197)
|
| 448 |
+
%new_idxs_277 = arith.select %cond_272, %new_idxs_276, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc198)
|
| 449 |
+
%new_idxs_278 = arith.xori %new_idxs_244, %new_idxs_277 : tensor<32x16xi32> loc(#loc199)
|
| 450 |
+
%y_279 = tt.reshape %ret_275 : tensor<32x16xi32> -> tensor<64x2x4xi32> loc(#loc162)
|
| 451 |
+
%ileft_280 = arith.muli %y_279, %ileft_137 : tensor<64x2x4xi32> loc(#loc164)
|
| 452 |
+
%ileft_281 = "tt.reduce"(%ileft_280) <{axis = 1 : i32}> ({
|
| 453 |
+
^bb0(%ileft_377: i32 loc(callsite(#loc1 at #loc165)), %ileft_378: i32 loc(callsite(#loc1 at #loc165))):
|
| 454 |
+
%ileft_379 = arith.addi %ileft_377, %ileft_378 : i32 loc(#loc211)
|
| 455 |
+
tt.reduce.return %ileft_379 : i32 loc(#loc201)
|
| 456 |
+
}) : (tensor<64x2x4xi32>) -> tensor<64x4xi32> loc(#loc201)
|
| 457 |
+
%ileft_282 = tt.expand_dims %ileft_281 {axis = 1 : i32} : tensor<64x4xi32> -> tensor<64x1x4xi32> loc(#loc166)
|
| 458 |
+
%ileft_283 = tt.broadcast %ileft_282 : tensor<64x1x4xi32> -> tensor<64x2x4xi32> loc(#loc167)
|
| 459 |
+
%iright_284 = arith.muli %y_279, %flip_59 : tensor<64x2x4xi32> loc(#loc168)
|
| 460 |
+
%iright_285 = "tt.reduce"(%iright_284) <{axis = 1 : i32}> ({
|
| 461 |
+
^bb0(%iright_377: i32 loc(callsite(#loc1 at #loc169)), %iright_378: i32 loc(callsite(#loc1 at #loc169))):
|
| 462 |
+
%iright_379 = arith.addi %iright_377, %iright_378 : i32 loc(#loc212)
|
| 463 |
+
tt.reduce.return %iright_379 : i32 loc(#loc203)
|
| 464 |
+
}) : (tensor<64x2x4xi32>) -> tensor<64x4xi32> loc(#loc203)
|
| 465 |
+
%iright_286 = tt.expand_dims %iright_285 {axis = 1 : i32} : tensor<64x4xi32> -> tensor<64x1x4xi32> loc(#loc170)
|
| 466 |
+
%iright_287 = tt.broadcast %iright_286 : tensor<64x1x4xi32> -> tensor<64x2x4xi32> loc(#loc171)
|
| 467 |
+
%ileft_288 = tt.reshape %ileft_283 : tensor<64x2x4xi32> -> tensor<32x16xi32> loc(#loc172)
|
| 468 |
+
%iright_289 = tt.reshape %iright_287 : tensor<64x2x4xi32> -> tensor<32x16xi32> loc(#loc173)
|
| 469 |
+
%y_idx_290 = tt.reshape %new_idxs_278 : tensor<32x16xi32> -> tensor<64x2x4xi32> loc(#loc174)
|
| 470 |
+
%left_idx_291 = arith.muli %y_idx_290, %ileft_137 : tensor<64x2x4xi32> loc(#loc176)
|
| 471 |
+
%left_idx_292 = "tt.reduce"(%left_idx_291) <{axis = 1 : i32}> ({
|
| 472 |
+
^bb0(%left_idx_377: i32 loc(callsite(#loc1 at #loc177)), %left_idx_378: i32 loc(callsite(#loc1 at #loc177))):
|
| 473 |
+
%left_idx_379 = arith.addi %left_idx_377, %left_idx_378 : i32 loc(#loc213)
|
| 474 |
+
tt.reduce.return %left_idx_379 : i32 loc(#loc206)
|
| 475 |
+
}) : (tensor<64x2x4xi32>) -> tensor<64x4xi32> loc(#loc206)
|
| 476 |
+
%left_idx_293 = tt.expand_dims %left_idx_292 {axis = 1 : i32} : tensor<64x4xi32> -> tensor<64x1x4xi32> loc(#loc178)
|
| 477 |
+
%left_idx_294 = tt.broadcast %left_idx_293 : tensor<64x1x4xi32> -> tensor<64x2x4xi32> loc(#loc179)
|
| 478 |
+
%right_idx_295 = arith.muli %y_idx_290, %flip_59 : tensor<64x2x4xi32> loc(#loc181)
|
| 479 |
+
%right_idx_296 = "tt.reduce"(%right_idx_295) <{axis = 1 : i32}> ({
|
| 480 |
+
^bb0(%right_idx_377: i32 loc(callsite(#loc1 at #loc182)), %right_idx_378: i32 loc(callsite(#loc1 at #loc182))):
|
| 481 |
+
%right_idx_379 = arith.addi %right_idx_377, %right_idx_378 : i32 loc(#loc214)
|
| 482 |
+
tt.reduce.return %right_idx_379 : i32 loc(#loc209)
|
| 483 |
+
}) : (tensor<64x2x4xi32>) -> tensor<64x4xi32> loc(#loc209)
|
| 484 |
+
%right_idx_297 = tt.expand_dims %right_idx_296 {axis = 1 : i32} : tensor<64x4xi32> -> tensor<64x1x4xi32> loc(#loc183)
|
| 485 |
+
%right_idx_298 = tt.broadcast %right_idx_297 : tensor<64x1x4xi32> -> tensor<64x2x4xi32> loc(#loc184)
|
| 486 |
+
%left_idx_299 = tt.reshape %left_idx_294 : tensor<64x2x4xi32> -> tensor<32x16xi32> loc(#loc185)
|
| 487 |
+
%right_idx_300 = tt.reshape %right_idx_298 : tensor<64x2x4xi32> -> tensor<32x16xi32> loc(#loc186)
|
| 488 |
+
%cond_301 = arith.cmpi slt, %ileft_288, %iright_289 : tensor<32x16xi32> loc(#loc187)
|
| 489 |
+
%eq_302 = arith.cmpi eq, %ileft_288, %iright_289 : tensor<32x16xi32> loc(#loc188)
|
| 490 |
+
%cond_303 = arith.cmpi sgt, %left_idx_299, %right_idx_300 : tensor<32x16xi32> loc(#loc189)
|
| 491 |
+
%cond_304 = arith.andi %eq_302, %cond_303 : tensor<32x16xi1> loc(#loc190)
|
| 492 |
+
%cond_305 = arith.ori %cond_301, %cond_304 : tensor<32x16xi1> loc(#loc191)
|
| 493 |
+
%ret_306 = arith.xori %ileft_288, %iright_289 : tensor<32x16xi32> loc(#loc194)
|
| 494 |
+
%ret_307 = arith.select %cond_305, %ret_306, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc195)
|
| 495 |
+
%ret_308 = arith.xori %ret_275, %ret_307 : tensor<32x16xi32> loc(#loc196)
|
| 496 |
+
%new_idxs_309 = arith.xori %left_idx_299, %right_idx_300 : tensor<32x16xi32> loc(#loc197)
|
| 497 |
+
%new_idxs_310 = arith.select %cond_305, %new_idxs_309, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc198)
|
| 498 |
+
%new_idxs_311 = arith.xori %new_idxs_278, %new_idxs_310 : tensor<32x16xi32> loc(#loc199)
|
| 499 |
+
%y_312 = tt.reshape %ret_308 : tensor<32x16xi32> -> tensor<128x2x2xi32> loc(#loc162)
|
| 500 |
+
%ileft_313 = arith.muli %y_312, %ileft_62 : tensor<128x2x2xi32> loc(#loc164)
|
| 501 |
+
%ileft_314 = "tt.reduce"(%ileft_313) <{axis = 1 : i32}> ({
|
| 502 |
+
^bb0(%ileft_377: i32 loc(callsite(#loc1 at #loc165)), %ileft_378: i32 loc(callsite(#loc1 at #loc165))):
|
| 503 |
+
%ileft_379 = arith.addi %ileft_377, %ileft_378 : i32 loc(#loc211)
|
| 504 |
+
tt.reduce.return %ileft_379 : i32 loc(#loc201)
|
| 505 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc201)
|
| 506 |
+
%ileft_315 = tt.expand_dims %ileft_314 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc166)
|
| 507 |
+
%ileft_316 = tt.broadcast %ileft_315 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc167)
|
| 508 |
+
%iright_317 = arith.muli %y_312, %flip_22 : tensor<128x2x2xi32> loc(#loc168)
|
| 509 |
+
%iright_318 = "tt.reduce"(%iright_317) <{axis = 1 : i32}> ({
|
| 510 |
+
^bb0(%iright_377: i32 loc(callsite(#loc1 at #loc169)), %iright_378: i32 loc(callsite(#loc1 at #loc169))):
|
| 511 |
+
%iright_379 = arith.addi %iright_377, %iright_378 : i32 loc(#loc212)
|
| 512 |
+
tt.reduce.return %iright_379 : i32 loc(#loc203)
|
| 513 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc203)
|
| 514 |
+
%iright_319 = tt.expand_dims %iright_318 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc170)
|
| 515 |
+
%iright_320 = tt.broadcast %iright_319 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc171)
|
| 516 |
+
%ileft_321 = tt.reshape %ileft_316 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc172)
|
| 517 |
+
%iright_322 = tt.reshape %iright_320 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc173)
|
| 518 |
+
%y_idx_323 = tt.reshape %new_idxs_311 : tensor<32x16xi32> -> tensor<128x2x2xi32> loc(#loc174)
|
| 519 |
+
%left_idx_324 = arith.muli %y_idx_323, %ileft_62 : tensor<128x2x2xi32> loc(#loc176)
|
| 520 |
+
%left_idx_325 = "tt.reduce"(%left_idx_324) <{axis = 1 : i32}> ({
|
| 521 |
+
^bb0(%left_idx_377: i32 loc(callsite(#loc1 at #loc177)), %left_idx_378: i32 loc(callsite(#loc1 at #loc177))):
|
| 522 |
+
%left_idx_379 = arith.addi %left_idx_377, %left_idx_378 : i32 loc(#loc213)
|
| 523 |
+
tt.reduce.return %left_idx_379 : i32 loc(#loc206)
|
| 524 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc206)
|
| 525 |
+
%left_idx_326 = tt.expand_dims %left_idx_325 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc178)
|
| 526 |
+
%left_idx_327 = tt.broadcast %left_idx_326 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc179)
|
| 527 |
+
%right_idx_328 = arith.muli %y_idx_323, %flip_22 : tensor<128x2x2xi32> loc(#loc181)
|
| 528 |
+
%right_idx_329 = "tt.reduce"(%right_idx_328) <{axis = 1 : i32}> ({
|
| 529 |
+
^bb0(%right_idx_377: i32 loc(callsite(#loc1 at #loc182)), %right_idx_378: i32 loc(callsite(#loc1 at #loc182))):
|
| 530 |
+
%right_idx_379 = arith.addi %right_idx_377, %right_idx_378 : i32 loc(#loc214)
|
| 531 |
+
tt.reduce.return %right_idx_379 : i32 loc(#loc209)
|
| 532 |
+
}) : (tensor<128x2x2xi32>) -> tensor<128x2xi32> loc(#loc209)
|
| 533 |
+
%right_idx_330 = tt.expand_dims %right_idx_329 {axis = 1 : i32} : tensor<128x2xi32> -> tensor<128x1x2xi32> loc(#loc183)
|
| 534 |
+
%right_idx_331 = tt.broadcast %right_idx_330 : tensor<128x1x2xi32> -> tensor<128x2x2xi32> loc(#loc184)
|
| 535 |
+
%left_idx_332 = tt.reshape %left_idx_327 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc185)
|
| 536 |
+
%right_idx_333 = tt.reshape %right_idx_331 : tensor<128x2x2xi32> -> tensor<32x16xi32> loc(#loc186)
|
| 537 |
+
%cond_334 = arith.cmpi slt, %ileft_321, %iright_322 : tensor<32x16xi32> loc(#loc187)
|
| 538 |
+
%eq_335 = arith.cmpi eq, %ileft_321, %iright_322 : tensor<32x16xi32> loc(#loc188)
|
| 539 |
+
%cond_336 = arith.cmpi sgt, %left_idx_332, %right_idx_333 : tensor<32x16xi32> loc(#loc189)
|
| 540 |
+
%cond_337 = arith.andi %eq_335, %cond_336 : tensor<32x16xi1> loc(#loc190)
|
| 541 |
+
%cond_338 = arith.ori %cond_334, %cond_337 : tensor<32x16xi1> loc(#loc191)
|
| 542 |
+
%ret_339 = arith.xori %ileft_321, %iright_322 : tensor<32x16xi32> loc(#loc194)
|
| 543 |
+
%ret_340 = arith.select %cond_338, %ret_339, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc195)
|
| 544 |
+
%ret_341 = arith.xori %ret_308, %ret_340 : tensor<32x16xi32> loc(#loc196)
|
| 545 |
+
%new_idxs_342 = arith.xori %left_idx_332, %right_idx_333 : tensor<32x16xi32> loc(#loc197)
|
| 546 |
+
%new_idxs_343 = arith.select %cond_338, %new_idxs_342, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc198)
|
| 547 |
+
%new_idxs_344 = arith.xori %new_idxs_311, %new_idxs_343 : tensor<32x16xi32> loc(#loc199)
|
| 548 |
+
%y_345 = tt.reshape %ret_341 : tensor<32x16xi32> -> tensor<256x2x1xi32> loc(#loc162)
|
| 549 |
+
%ileft_346 = arith.muli %y_345, %ileft : tensor<256x2x1xi32> loc(#loc164)
|
| 550 |
+
%ileft_347 = "tt.reduce"(%ileft_346) <{axis = 1 : i32}> ({
|
| 551 |
+
^bb0(%ileft_377: i32 loc(callsite(#loc1 at #loc165)), %ileft_378: i32 loc(callsite(#loc1 at #loc165))):
|
| 552 |
+
%ileft_379 = arith.addi %ileft_377, %ileft_378 : i32 loc(#loc211)
|
| 553 |
+
tt.reduce.return %ileft_379 : i32 loc(#loc201)
|
| 554 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc201)
|
| 555 |
+
%ileft_348 = tt.expand_dims %ileft_347 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc166)
|
| 556 |
+
%ileft_349 = tt.broadcast %ileft_348 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc167)
|
| 557 |
+
%iright_350 = arith.muli %y_345, %iright : tensor<256x2x1xi32> loc(#loc168)
|
| 558 |
+
%iright_351 = "tt.reduce"(%iright_350) <{axis = 1 : i32}> ({
|
| 559 |
+
^bb0(%iright_377: i32 loc(callsite(#loc1 at #loc169)), %iright_378: i32 loc(callsite(#loc1 at #loc169))):
|
| 560 |
+
%iright_379 = arith.addi %iright_377, %iright_378 : i32 loc(#loc212)
|
| 561 |
+
tt.reduce.return %iright_379 : i32 loc(#loc203)
|
| 562 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc203)
|
| 563 |
+
%iright_352 = tt.expand_dims %iright_351 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc170)
|
| 564 |
+
%iright_353 = tt.broadcast %iright_352 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc171)
|
| 565 |
+
%ileft_354 = tt.reshape %ileft_349 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc172)
|
| 566 |
+
%iright_355 = tt.reshape %iright_353 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc173)
|
| 567 |
+
%y_idx_356 = tt.reshape %new_idxs_344 : tensor<32x16xi32> -> tensor<256x2x1xi32> loc(#loc174)
|
| 568 |
+
%left_idx_357 = arith.muli %y_idx_356, %ileft : tensor<256x2x1xi32> loc(#loc176)
|
| 569 |
+
%left_idx_358 = "tt.reduce"(%left_idx_357) <{axis = 1 : i32}> ({
|
| 570 |
+
^bb0(%left_idx_377: i32 loc(callsite(#loc1 at #loc177)), %left_idx_378: i32 loc(callsite(#loc1 at #loc177))):
|
| 571 |
+
%left_idx_379 = arith.addi %left_idx_377, %left_idx_378 : i32 loc(#loc213)
|
| 572 |
+
tt.reduce.return %left_idx_379 : i32 loc(#loc206)
|
| 573 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc206)
|
| 574 |
+
%left_idx_359 = tt.expand_dims %left_idx_358 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc178)
|
| 575 |
+
%left_idx_360 = tt.broadcast %left_idx_359 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc179)
|
| 576 |
+
%right_idx_361 = arith.muli %y_idx_356, %iright : tensor<256x2x1xi32> loc(#loc181)
|
| 577 |
+
%right_idx_362 = "tt.reduce"(%right_idx_361) <{axis = 1 : i32}> ({
|
| 578 |
+
^bb0(%right_idx_377: i32 loc(callsite(#loc1 at #loc182)), %right_idx_378: i32 loc(callsite(#loc1 at #loc182))):
|
| 579 |
+
%right_idx_379 = arith.addi %right_idx_377, %right_idx_378 : i32 loc(#loc214)
|
| 580 |
+
tt.reduce.return %right_idx_379 : i32 loc(#loc209)
|
| 581 |
+
}) : (tensor<256x2x1xi32>) -> tensor<256x1xi32> loc(#loc209)
|
| 582 |
+
%right_idx_363 = tt.expand_dims %right_idx_362 {axis = 1 : i32} : tensor<256x1xi32> -> tensor<256x1x1xi32> loc(#loc183)
|
| 583 |
+
%right_idx_364 = tt.broadcast %right_idx_363 : tensor<256x1x1xi32> -> tensor<256x2x1xi32> loc(#loc184)
|
| 584 |
+
%left_idx_365 = tt.reshape %left_idx_360 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc185)
|
| 585 |
+
%right_idx_366 = tt.reshape %right_idx_364 : tensor<256x2x1xi32> -> tensor<32x16xi32> loc(#loc186)
|
| 586 |
+
%cond_367 = arith.cmpi slt, %ileft_354, %iright_355 : tensor<32x16xi32> loc(#loc187)
|
| 587 |
+
%eq_368 = arith.cmpi eq, %ileft_354, %iright_355 : tensor<32x16xi32> loc(#loc188)
|
| 588 |
+
%cond_369 = arith.cmpi sgt, %left_idx_365, %right_idx_366 : tensor<32x16xi32> loc(#loc189)
|
| 589 |
+
%cond_370 = arith.andi %eq_368, %cond_369 : tensor<32x16xi1> loc(#loc190)
|
| 590 |
+
%cond_371 = arith.ori %cond_367, %cond_370 : tensor<32x16xi1> loc(#loc191)
|
| 591 |
+
%new_idxs_372 = arith.xori %left_idx_365, %right_idx_366 : tensor<32x16xi32> loc(#loc197)
|
| 592 |
+
%new_idxs_373 = arith.select %cond_371, %new_idxs_372, %cst_0 : tensor<32x16xi1>, tensor<32x16xi32> loc(#loc198)
|
| 593 |
+
%new_idxs_374 = arith.xori %new_idxs_344, %new_idxs_373 : tensor<32x16xi32> loc(#loc199)
|
| 594 |
+
%tmp7 = arith.extsi %tmp0_19 : tensor<32x16xi32> to tensor<32x16xi64> loc(#loc149)
|
| 595 |
+
%tmp10_375 = arith.select %tmp0_18, %tmp7, %tmp10 : tensor<32x16xi1>, tensor<32x16xi64> loc(#loc86)
|
| 596 |
+
%tmp11 = "tt.reduce"(%tmp10_375) <{axis = 1 : i32}> ({
|
| 597 |
+
^bb0(%tmp11_377: i64 loc(callsite(#loc1 at #loc150)), %tmp11_378: i64 loc(callsite(#loc1 at #loc150))):
|
| 598 |
+
%tmp11_379 = arith.addi %tmp11_377, %tmp11_378 : i64 loc(#loc200)
|
| 599 |
+
tt.reduce.return %tmp11_379 : i64 loc(#loc160)
|
| 600 |
+
}) : (tensor<32x16xi64>) -> tensor<32xi64> loc(#loc160)
|
| 601 |
+
%tmp11_376 = tt.expand_dims %tmp11 {axis = 1 : i32} : tensor<32xi64> -> tensor<32x1xi64> loc(#loc151)
|
| 602 |
+
%tmp14 = arith.trunci %tmp11_376 : tensor<32x1xi64> to tensor<32x1xi32> loc(#loc152)
|
| 603 |
+
%0 = arith.muli %xindex_6, %cst_2 : tensor<32x1xi32> loc(#loc73)
|
| 604 |
+
%1 = tt.broadcast %r0_index_8 : tensor<1x16xi32> -> tensor<32x16xi32> loc(#loc74)
|
| 605 |
+
%2 = tt.broadcast %0 : tensor<32x1xi32> -> tensor<32x16xi32> loc(#loc74)
|
| 606 |
+
%3 = arith.addi %1, %2 : tensor<32x16xi32> loc(#loc74)
|
| 607 |
+
%4 = tt.splat %out_ptr2 : !tt.ptr<i32> -> tensor<32x16x!tt.ptr<i32>> loc(#loc75)
|
| 608 |
+
%5 = tt.addptr %4, %3 : tensor<32x16x!tt.ptr<i32>>, tensor<32x16xi32> loc(#loc75)
|
| 609 |
+
tt.store %5, %new_idxs_374, %tmp0_18 : tensor<32x16x!tt.ptr<i32>> loc(#loc76)
|
| 610 |
+
%6 = tt.splat %out_ptr3 : !tt.ptr<i32> -> tensor<32x1x!tt.ptr<i32>> loc(#loc77)
|
| 611 |
+
%7 = tt.addptr %6, %xindex_6 : tensor<32x1x!tt.ptr<i32>>, tensor<32x1xi32> loc(#loc77)
|
| 612 |
+
tt.store %7, %tmp14, %xmask_7 : tensor<32x1x!tt.ptr<i32>> loc(#loc78)
|
| 613 |
+
tt.return loc(#loc79)
|
| 614 |
+
} loc(#loc)
|
| 615 |
+
} loc(#loc)
|
| 616 |
+
#loc3 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":44:34)
|
| 617 |
+
#loc4 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:49)
|
| 618 |
+
#loc5 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:38)
|
| 619 |
+
#loc6 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":26:21)
|
| 620 |
+
#loc7 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":24:28)
|
| 621 |
+
#loc8 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":24:33)
|
| 622 |
+
#loc9 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":25:36)
|
| 623 |
+
#loc10 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":25:44)
|
| 624 |
+
#loc11 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":25:23)
|
| 625 |
+
#loc12 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":27:28)
|
| 626 |
+
#loc13 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":27:38)
|
| 627 |
+
#loc14 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":33:19)
|
| 628 |
+
#loc15 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":34:19)
|
| 629 |
+
#loc16 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:35)
|
| 630 |
+
#loc17 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:45)
|
| 631 |
+
#loc18 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:30)
|
| 632 |
+
#loc19 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":36:54)
|
| 633 |
+
#loc20 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":38:19)
|
| 634 |
+
#loc21 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":40:33)
|
| 635 |
+
#loc22 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":627:41)
|
| 636 |
+
#loc24 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":627:44)
|
| 637 |
+
#loc25 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":627:60)
|
| 638 |
+
#loc26 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":627:68)
|
| 639 |
+
#loc27 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":533:22)
|
| 640 |
+
#loc29 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":537:21)
|
| 641 |
+
#loc30 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":538:40)
|
| 642 |
+
#loc31 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":291:36)
|
| 643 |
+
#loc33 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":261:15)
|
| 644 |
+
#loc34 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":538:65)
|
| 645 |
+
#loc35 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":538:78)
|
| 646 |
+
#loc36 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":539:41)
|
| 647 |
+
#loc38 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":539:67)
|
| 648 |
+
#loc39 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":539:80)
|
| 649 |
+
#loc40 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":540:30)
|
| 650 |
+
#loc41 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":541:32)
|
| 651 |
+
#loc42 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":546:29)
|
| 652 |
+
#loc43 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":548:36)
|
| 653 |
+
#loc44 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":548:23)
|
| 654 |
+
#loc45 = loc("/workspace/specforge/lib/python3.11/site-packages/triton/language/standard.py":290:25)
|
| 655 |
+
#loc47 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":548:53)
|
| 656 |
+
#loc48 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":548:66)
|
| 657 |
+
#loc49 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":551:37)
|
| 658 |
+
#loc50 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":551:23)
|
| 659 |
+
#loc52 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":551:54)
|
| 660 |
+
#loc53 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":551:67)
|
| 661 |
+
#loc54 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":553:36)
|
| 662 |
+
#loc55 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":554:38)
|
| 663 |
+
#loc56 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":574:22)
|
| 664 |
+
#loc57 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":591:21)
|
| 665 |
+
#loc58 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":594:40)
|
| 666 |
+
#loc59 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":594:29)
|
| 667 |
+
#loc60 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":594:23)
|
| 668 |
+
#loc61 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":599:19)
|
| 669 |
+
#loc62 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":599:28)
|
| 670 |
+
#loc63 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":600:38)
|
| 671 |
+
#loc64 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":600:46)
|
| 672 |
+
#loc65 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":600:15)
|
| 673 |
+
#loc66 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":601:48)
|
| 674 |
+
#loc67 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":601:59)
|
| 675 |
+
#loc68 = loc("/workspace/specforge/lib/python3.11/site-packages/torch/_inductor/runtime/triton_helpers.py":601:22)
|
| 676 |
+
#loc69 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":42:19)
|
| 677 |
+
#loc71 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":45:29)
|
| 678 |
+
#loc72 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":48:21)
|
| 679 |
+
#loc73 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":49:35)
|
| 680 |
+
#loc74 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":49:32)
|
| 681 |
+
#loc75 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":49:25)
|
| 682 |
+
#loc76 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":49:47)
|
| 683 |
+
#loc77 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":50:25)
|
| 684 |
+
#loc78 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":50:37)
|
| 685 |
+
#loc79 = loc("/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/ce/ccefvxdobtenaocabewevaf45h6p7aehp54xd7ccf3732l3vxvwg.py":50:4)
|
| 686 |
+
#loc85 = loc(callsite(#loc1 at #loc2))
|
| 687 |
+
#loc86 = loc("tmp10"(#loc3))
|
| 688 |
+
#loc87 = loc("tmp0"(#loc4))
|
| 689 |
+
#loc88 = loc("tmp0"(#loc5))
|
| 690 |
+
#loc89 = loc("xmask"(#loc6))
|
| 691 |
+
#loc90 = loc("xoffset"(#loc7))
|
| 692 |
+
#loc91 = loc("xoffset"(#loc8))
|
| 693 |
+
#loc92 = loc("xindex"(#loc9))
|
| 694 |
+
#loc93 = loc("xindex"(#loc10))
|
| 695 |
+
#loc94 = loc("xindex"(#loc11))
|
| 696 |
+
#loc95 = loc("r0_index"(#loc12))
|
| 697 |
+
#loc96 = loc("r0_index"(#loc13))
|
| 698 |
+
#loc97 = loc("x0"(#loc14))
|
| 699 |
+
#loc98 = loc("x1"(#loc15))
|
| 700 |
+
#loc99 = loc("tmp0"(#loc16))
|
| 701 |
+
#loc100 = loc("tmp0"(#loc17))
|
| 702 |
+
#loc101 = loc("tmp0"(#loc18))
|
| 703 |
+
#loc102 = loc("tmp0"(#loc19))
|
| 704 |
+
#loc103 = loc("tmp2"(#loc20))
|
| 705 |
+
#loc104 = loc("tmp4"(#loc21))
|
| 706 |
+
#loc105 = loc("flip"(#loc22))
|
| 707 |
+
#loc107 = loc("flip"(#loc24))
|
| 708 |
+
#loc108 = loc("flip"(#loc25))
|
| 709 |
+
#loc109 = loc("flip"(#loc26))
|
| 710 |
+
#loc110 = loc("y"(#loc27))
|
| 711 |
+
#loc111 = loc("left_mask"(#loc29))
|
| 712 |
+
#loc112 = loc("ileft"(#loc30))
|
| 713 |
+
#loc114 = loc("ileft"(#loc34))
|
| 714 |
+
#loc115 = loc("ileft"(#loc35))
|
| 715 |
+
#loc116 = loc("iright"(#loc36))
|
| 716 |
+
#loc118 = loc("iright"(#loc38))
|
| 717 |
+
#loc119 = loc("iright"(#loc39))
|
| 718 |
+
#loc120 = loc("ileft"(#loc40))
|
| 719 |
+
#loc121 = loc("iright"(#loc41))
|
| 720 |
+
#loc122 = loc("y_idx"(#loc42))
|
| 721 |
+
#loc123 = loc("left_idx"(#loc43))
|
| 722 |
+
#loc124 = loc("left_idx"(#loc44))
|
| 723 |
+
#loc125 = loc("input"(#loc45))
|
| 724 |
+
#loc127 = loc("left_idx"(#loc47))
|
| 725 |
+
#loc128 = loc("left_idx"(#loc48))
|
| 726 |
+
#loc129 = loc("right_idx"(#loc49))
|
| 727 |
+
#loc130 = loc("right_idx"(#loc50))
|
| 728 |
+
#loc132 = loc("right_idx"(#loc52))
|
| 729 |
+
#loc133 = loc("right_idx"(#loc53))
|
| 730 |
+
#loc134 = loc("left_idx"(#loc54))
|
| 731 |
+
#loc135 = loc("right_idx"(#loc55))
|
| 732 |
+
#loc136 = loc("cond"(#loc56))
|
| 733 |
+
#loc137 = loc("eq"(#loc57))
|
| 734 |
+
#loc138 = loc("cond"(#loc58))
|
| 735 |
+
#loc139 = loc("cond"(#loc59))
|
| 736 |
+
#loc140 = loc("cond"(#loc60))
|
| 737 |
+
#loc141 = loc("cond"(#loc61))
|
| 738 |
+
#loc142 = loc("cond"(#loc62))
|
| 739 |
+
#loc143 = loc("ret"(#loc63))
|
| 740 |
+
#loc144 = loc("ret"(#loc64))
|
| 741 |
+
#loc145 = loc("ret"(#loc65))
|
| 742 |
+
#loc146 = loc("new_idxs"(#loc66))
|
| 743 |
+
#loc147 = loc("new_idxs"(#loc67))
|
| 744 |
+
#loc148 = loc("new_idxs"(#loc68))
|
| 745 |
+
#loc149 = loc("tmp7"(#loc69))
|
| 746 |
+
#loc151 = loc("tmp11"(#loc71))
|
| 747 |
+
#loc152 = loc("tmp14"(#loc72))
|
| 748 |
+
#loc153 = loc(callsite(#loc105 at #loc106))
|
| 749 |
+
#loc154 = loc(callsite(#loc107 at #loc106))
|
| 750 |
+
#loc155 = loc(callsite(#loc108 at #loc106))
|
| 751 |
+
#loc156 = loc(callsite(#loc109 at #loc106))
|
| 752 |
+
#loc158 = loc("cond"(#loc136))
|
| 753 |
+
#loc159 = loc("eq"(#loc137))
|
| 754 |
+
#loc160 = loc(callsite(#loc31 at #loc150))
|
| 755 |
+
#loc162 = loc(callsite(#loc110 at #loc157))
|
| 756 |
+
#loc163 = loc(callsite(#loc111 at #loc157))
|
| 757 |
+
#loc164 = loc(callsite(#loc112 at #loc157))
|
| 758 |
+
#loc166 = loc(callsite(#loc114 at #loc157))
|
| 759 |
+
#loc167 = loc(callsite(#loc115 at #loc157))
|
| 760 |
+
#loc168 = loc(callsite(#loc116 at #loc157))
|
| 761 |
+
#loc170 = loc(callsite(#loc118 at #loc157))
|
| 762 |
+
#loc171 = loc(callsite(#loc119 at #loc157))
|
| 763 |
+
#loc172 = loc(callsite(#loc120 at #loc157))
|
| 764 |
+
#loc173 = loc(callsite(#loc121 at #loc157))
|
| 765 |
+
#loc174 = loc(callsite(#loc122 at #loc157))
|
| 766 |
+
#loc175 = loc(callsite(#loc123 at #loc157))
|
| 767 |
+
#loc176 = loc(callsite(#loc124 at #loc157))
|
| 768 |
+
#loc178 = loc(callsite(#loc127 at #loc157))
|
| 769 |
+
#loc179 = loc(callsite(#loc128 at #loc157))
|
| 770 |
+
#loc180 = loc(callsite(#loc129 at #loc157))
|
| 771 |
+
#loc181 = loc(callsite(#loc130 at #loc157))
|
| 772 |
+
#loc183 = loc(callsite(#loc132 at #loc157))
|
| 773 |
+
#loc184 = loc(callsite(#loc133 at #loc157))
|
| 774 |
+
#loc185 = loc(callsite(#loc134 at #loc157))
|
| 775 |
+
#loc186 = loc(callsite(#loc135 at #loc157))
|
| 776 |
+
#loc187 = loc(callsite(#loc158 at #loc157))
|
| 777 |
+
#loc188 = loc(callsite(#loc159 at #loc157))
|
| 778 |
+
#loc189 = loc(callsite(#loc138 at #loc157))
|
| 779 |
+
#loc190 = loc(callsite(#loc139 at #loc157))
|
| 780 |
+
#loc191 = loc(callsite(#loc140 at #loc157))
|
| 781 |
+
#loc192 = loc(callsite(#loc141 at #loc157))
|
| 782 |
+
#loc193 = loc(callsite(#loc142 at #loc157))
|
| 783 |
+
#loc194 = loc(callsite(#loc143 at #loc157))
|
| 784 |
+
#loc195 = loc(callsite(#loc144 at #loc157))
|
| 785 |
+
#loc196 = loc(callsite(#loc145 at #loc157))
|
| 786 |
+
#loc197 = loc(callsite(#loc146 at #loc157))
|
| 787 |
+
#loc198 = loc(callsite(#loc147 at #loc157))
|
| 788 |
+
#loc199 = loc(callsite(#loc148 at #loc157))
|
| 789 |
+
#loc200 = loc(callsite(#loc33 at #loc160))
|
| 790 |
+
#loc201 = loc(callsite(#loc31 at #loc165))
|
| 791 |
+
#loc203 = loc(callsite(#loc31 at #loc169))
|
| 792 |
+
#loc205 = loc(callsite(#loc125 at #loc177))
|
| 793 |
+
#loc206 = loc(callsite(#loc31 at #loc177))
|
| 794 |
+
#loc208 = loc(callsite(#loc125 at #loc182))
|
| 795 |
+
#loc209 = loc(callsite(#loc31 at #loc182))
|
| 796 |
+
#loc211 = loc(callsite(#loc33 at #loc201))
|
| 797 |
+
#loc212 = loc(callsite(#loc33 at #loc203))
|
| 798 |
+
#loc213 = loc(callsite(#loc33 at #loc206))
|
| 799 |
+
#loc214 = loc(callsite(#loc33 at #loc209))
|
SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/__grp__triton_poi_fused_new_zeros_1.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"child_paths": {"triton_poi_fused_new_zeros_1.source": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/triton_poi_fused_new_zeros_1.source", "triton_poi_fused_new_zeros_1.ttir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/triton_poi_fused_new_zeros_1.ttir", "triton_poi_fused_new_zeros_1.ttgir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/triton_poi_fused_new_zeros_1.ttgir", "triton_poi_fused_new_zeros_1.llir": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/triton_poi_fused_new_zeros_1.llir", "triton_poi_fused_new_zeros_1.ptx": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/triton_poi_fused_new_zeros_1.ptx", "triton_poi_fused_new_zeros_1.cubin": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/triton_poi_fused_new_zeros_1.cubin", "triton_poi_fused_new_zeros_1.json": "/workspace/hanrui/SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/triton_poi_fused_new_zeros_1.json"}}
|
SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/triton_poi_fused_new_zeros_1.cubin
ADDED
|
Binary file (5.38 kB). View file
|
|
|
SpecForge-ext/cache/compiled_kernels/triton/0/4UWYNBR3KPWQGNAZ5LIIRE7YAZWTQP4CP3JS6GOSLWYDF5K7WTAA/triton_poi_fused_new_zeros_1.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"hash": "e52d86863b53ed033419ead08893f8066d383f827ed32f19d25db032f55fb4c0", "target": {"backend": "cuda", "arch": 90, "warp_size": 32}, "num_warps": 4, "num_ctas": 1, "num_stages": 1, "warp_size": 32, "maxnreg": null, "cluster_dims": [1, 1, 1], "ptx_version": null, "ptx_options": null, "ir_override": null, "enable_fp_fusion": true, "launch_cooperative_grid": false, "launch_pdl": false, "supported_fp8_dtypes": ["fp8e4b15", "fp8e4nv", "fp8e5"], "deprecated_fp8_dot_operand_dtypes": ["fp8e4b15"], "default_dot_input_precision": "tf32", "allowed_dot_input_precisions": ["tf32", "tf32x3", "ieee"], "max_num_imprecise_acc_default": 1073741824, "extern_libs": [["libdevice", "/workspace/specforge/lib/python3.11/site-packages/triton/backends/nvidia/lib/libdevice.10.bc"]], "debug": true, "backend_name": "cuda", "sanitize_overflow": false, "arch": "sm90", "instrumentation_mode": "", "triton_version": "3.5.1", "tensordesc_meta": [], "shared": 0, "tmem_size": 0, "global_scratch_size": 0, "global_scratch_align": 1, "profile_scratch_size": 0, "profile_scratch_align": 1, "name": "triton_poi_fused_new_zeros_1"}
|