[llvm] [AMDGPU] Constant folding for wave-reduce intrinsics (PR #212755)
Matt Arsenault via llvm-commits
llvm-commits at lists.llvm.org
Wed Jul 29 06:08:41 PDT 2026
================
@@ -149,6 +149,242 @@ define amdgpu_cs void @atomic_sub(<4 x i32> inreg %arg) {
ret void
}
+define amdgpu_cs void @atomic_add_constant_0(<4 x i32> inreg %arg) {
+; IR-LABEL: define amdgpu_cs void @atomic_add_constant_0(
+; IR-SAME: <4 x i32> inreg [[ARG:%.*]]) {
+; IR-NEXT: [[_ENTRY:.*:]]
+; IR-NEXT: [[TMP0:%.*]] = call i64 @llvm.amdgcn.ballot.i64(i1 true)
+; IR-NEXT: [[TMP1:%.*]] = trunc i64 [[TMP0]] to i32
+; IR-NEXT: [[TMP2:%.*]] = lshr i64 [[TMP0]], 32
+; IR-NEXT: [[TMP3:%.*]] = trunc i64 [[TMP2]] to i32
+; IR-NEXT: [[TMP4:%.*]] = call i32 @llvm.amdgcn.mbcnt.lo(i32 [[TMP1]], i32 0)
+; IR-NEXT: [[TMP5:%.*]] = call i32 @llvm.amdgcn.mbcnt.hi(i32 [[TMP3]], i32 [[TMP4]])
+; IR-NEXT: [[TMP6:%.*]] = call i64 @llvm.ctpop.i64(i64 [[TMP0]])
+; IR-NEXT: [[TMP7:%.*]] = trunc i64 [[TMP6]] to i32
+; IR-NEXT: [[TMP8:%.*]] = mul i32 0, [[TMP7]]
+; IR-NEXT: [[TMP9:%.*]] = icmp eq i32 [[TMP5]], 0
+; IR-NEXT: br i1 [[TMP9]], label %[[BB10:.*]], label %[[BB12:.*]]
+; IR: [[BB10]]:
+; IR-NEXT: [[TMP11:%.*]] = call i32 @llvm.amdgcn.struct.buffer.atomic.add.i32(i32 [[TMP8]], <4 x i32> [[ARG]], i32 0, i32 0, i32 0, i32 0)
+; IR-NEXT: br label %[[BB12]]
+; IR: [[BB12]]:
+; IR-NEXT: ret void
+;
+; GCN-LABEL: atomic_add_constant_0:
+; GCN: ; %bb.0: ; %.entry
+; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
+; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
+; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
+; GCN-NEXT: s_and_saveexec_b64 s[4:5], vcc
+; GCN-NEXT: s_cbranch_execz .LBB3_2
+; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: v_mov_b32_e32 v0, 0
+; GCN-NEXT: buffer_atomic_add v0, v0, s[0:3], 0 idxen
+; GCN-NEXT: .LBB3_2:
+; GCN-NEXT: s_endpgm
+.entry:
+ call i32 @llvm.amdgcn.struct.buffer.atomic.add.i32(i32 0, <4 x i32> %arg, i32 0, i32 0, i32 0, i32 0)
+ ret void
+}
+
+define amdgpu_cs void @atomic_sub_constant_0(<4 x i32> inreg %arg) {
----------------
arsenm wrote:
Include the type in the function name
https://github.com/llvm/llvm-project/pull/212755
More information about the llvm-commits
mailing list