[llvm] [AMDGPU] Constant folding for wave-reduce intrinsics (PR #212755)

Matt Arsenault via llvm-commits llvm-commits at lists.llvm.org
Wed Jul 29 06:08:41 PDT 2026


================
@@ -149,6 +149,242 @@ define amdgpu_cs void @atomic_sub(<4 x i32> inreg %arg)  {
   ret void
 }
 
+define amdgpu_cs void @atomic_add_constant_0(<4 x i32> inreg %arg)  {
+; IR-LABEL: define amdgpu_cs void @atomic_add_constant_0(
+; IR-SAME: <4 x i32> inreg [[ARG:%.*]]) {
+; IR-NEXT:  [[_ENTRY:.*:]]
+; IR-NEXT:    [[TMP0:%.*]] = call i64 @llvm.amdgcn.ballot.i64(i1 true)
+; IR-NEXT:    [[TMP1:%.*]] = trunc i64 [[TMP0]] to i32
+; IR-NEXT:    [[TMP2:%.*]] = lshr i64 [[TMP0]], 32
+; IR-NEXT:    [[TMP3:%.*]] = trunc i64 [[TMP2]] to i32
+; IR-NEXT:    [[TMP4:%.*]] = call i32 @llvm.amdgcn.mbcnt.lo(i32 [[TMP1]], i32 0)
+; IR-NEXT:    [[TMP5:%.*]] = call i32 @llvm.amdgcn.mbcnt.hi(i32 [[TMP3]], i32 [[TMP4]])
+; IR-NEXT:    [[TMP6:%.*]] = call i64 @llvm.ctpop.i64(i64 [[TMP0]])
+; IR-NEXT:    [[TMP7:%.*]] = trunc i64 [[TMP6]] to i32
+; IR-NEXT:    [[TMP8:%.*]] = mul i32 0, [[TMP7]]
+; IR-NEXT:    [[TMP9:%.*]] = icmp eq i32 [[TMP5]], 0
+; IR-NEXT:    br i1 [[TMP9]], label %[[BB10:.*]], label %[[BB12:.*]]
+; IR:       [[BB10]]:
+; IR-NEXT:    [[TMP11:%.*]] = call i32 @llvm.amdgcn.struct.buffer.atomic.add.i32(i32 [[TMP8]], <4 x i32> [[ARG]], i32 0, i32 0, i32 0, i32 0)
+; IR-NEXT:    br label %[[BB12]]
+; IR:       [[BB12]]:
+; IR-NEXT:    ret void
+;
+; GCN-LABEL: atomic_add_constant_0:
+; GCN:       ; %bb.0: ; %.entry
+; GCN-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
+; GCN-NEXT:    v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
+; GCN-NEXT:    v_cmp_eq_u32_e32 vcc, 0, v0
+; GCN-NEXT:    s_and_saveexec_b64 s[4:5], vcc
+; GCN-NEXT:    s_cbranch_execz .LBB3_2
+; GCN-NEXT:  ; %bb.1:
+; GCN-NEXT:    v_mov_b32_e32 v0, 0
+; GCN-NEXT:    buffer_atomic_add v0, v0, s[0:3], 0 idxen
+; GCN-NEXT:  .LBB3_2:
+; GCN-NEXT:    s_endpgm
+.entry:
+  call i32 @llvm.amdgcn.struct.buffer.atomic.add.i32(i32 0, <4 x i32> %arg, i32 0, i32 0, i32 0, i32 0)
+  ret void
+}
+
+define amdgpu_cs void @atomic_sub_constant_0(<4 x i32> inreg %arg)  {
----------------
arsenm wrote:

Include the type in the function name 

https://github.com/llvm/llvm-project/pull/212755


More information about the llvm-commits mailing list