[llvm] AMDGPU: Mark SCC def dead in wave reduction expansions (PR #226370)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 23:18:40 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Matt Arsenault (arsenm)
<details>
<summary>Changes</summary>
Co-Authored-By: Claude Opus 5 <noreply@<!-- -->anthropic.com>
---
Patch is 422.37 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/226370.diff
9 Files Affected:
- (modified) llvm/lib/Target/AMDGPU/SIISelLowering.cpp (+12-5)
- (modified) llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll (+185-197)
- (modified) llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll (+580-582)
- (modified) llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll (+287-290)
- (modified) llvm/test/CodeGen/AMDGPU/atomic_optimizations_mul_one.ll (+27-27)
- (modified) llvm/test/CodeGen/AMDGPU/atomic_optimizations_pixelshader.ll (+24-26)
- (modified) llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll (+185-197)
- (modified) llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll (+217-213)
- (modified) llvm/test/CodeGen/AMDGPU/insert_waitcnt_for_precise_memory.ll (+52-52)
``````````diff
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index e700adfaa14a3a..f9e4deadc42d7f 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -6055,7 +6055,8 @@ static MachineBasicBlock *lowerWaveReduce(MachineInstr &MI,
auto NewAccumulator =
BuildMI(BB, MI, DL, TII->get(BitCountOpc), NumActiveLanes)
- .addReg(ExecMask);
+ .addReg(ExecMask)
+ .setOperandDead(2); // Dead scc
switch (Opc) {
case AMDGPU::S_XOR_B32:
@@ -6113,7 +6114,8 @@ static MachineBasicBlock *lowerWaveReduce(MachineInstr &MI,
// Take the negation of the source operand.
BuildMI(BB, MI, DL, TII->get(AMDGPU::S_SUB_I32), NegatedVal)
.addImm(0)
- .addReg(SrcReg);
+ .addReg(SrcReg)
+ .setOperandDead(3); // Dead scc
BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MUL_I32), DstReg)
.addReg(NegatedVal)
.addReg(NewAccumulator->getOperand(0).getReg());
@@ -6388,6 +6390,8 @@ static MachineBasicBlock *lowerWaveReduce(MachineInstr &MI,
OpInstr.addImm(0); // opsel
if (hasOMod)
OpInstr.addImm(0); // omod
+ if (TII->isSALU(Opc))
+ OpInstr.setOperandDead(3); // Dead scc
if (ST.getInstrInfo()->isVALU(Opc, /*AllowLDSDMA=*/true)) {
BuildMI(*ComputeLoop, I, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32),
DstReg)
@@ -6503,7 +6507,8 @@ static MachineBasicBlock *lowerWaveReduce(MachineInstr &MI,
case AMDGPU::S_SUB_U64_PSEUDO: {
NewAccumulator = BuildMI(*ComputeLoop, I, DL, TII->get(Opc), DstReg)
.addReg(Accumulator->getOperand(0).getReg())
- .addReg(LaneValue->getOperand(0).getReg());
+ .addReg(LaneValue->getOperand(0).getReg())
+ .setOperandDead(3); // Dead scc
ComputeLoop =
expand64BitScalarArithmetic(*NewAccumulator, ComputeLoop);
break;
@@ -6904,12 +6909,14 @@ static MachineBasicBlock *lowerWaveReduce(MachineInstr &MI,
if (Opc == AMDGPU::S_SUB_I32) {
BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::S_SUB_I32), NegatedReducedVal)
.addImm(0)
- .addReg(ReducedValSGPR);
+ .addReg(ReducedValSGPR)
+ .setOperandDead(3); // Dead scc
} else if (Opc == AMDGPU::S_SUB_U64_PSEUDO) {
auto NegatedValInstr =
BuildMI(*CurrBB, MI, DL, TII->get(Opc), NegatedReducedVal)
.addImm(0)
- .addReg(ReducedValSGPR);
+ .addReg(ReducedValSGPR)
+ .setOperandDead(3); // Dead scc
CurrBB = expand64BitScalarArithmetic(*NegatedValInstr, CurrBB);
}
// Mark the final result as a whole-wave-mode calculation.
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
index 67c115ac2ed767..a912ceb4b16363 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
@@ -23,14 +23,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX6: ; %bb.0: ; %entry
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB0_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_mul_i32 s2, s2, 5
; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
@@ -51,14 +51,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB0_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_mul_i32 s2, s2, 5
; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
@@ -79,14 +79,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB0_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_mul_i32 s2, s2, 5
; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
@@ -105,15 +105,15 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W64-LABEL: add_i32_constant:
; GFX10W64: ; %bb.0: ; %entry
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB0_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: s_mul_i32 s2, s2, 5
; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
@@ -133,14 +133,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W32-LABEL: add_i32_constant:
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB0_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s1
; GFX10W32-NEXT: s_mul_i32 s1, s1, 5
; GFX10W32-NEXT: v_mov_b32_e32 v1, s1
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
@@ -160,19 +160,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-LABEL: add_i32_constant:
; GFX11W64: ; %bb.0: ; %entry
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB0_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], 0 glc
@@ -191,17 +190,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-LABEL: add_i32_constant:
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB0_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W32-NEXT: v_mov_b32_e32 v1, s1
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], 0 glc
@@ -220,19 +219,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-LABEL: add_i32_constant:
; GFX12W64: ; %bb.0: ; %entry
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB0_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: s_wait_kmcnt 0x0
; GFX12W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -252,17 +250,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-LABEL: add_i32_constant:
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB0_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W32-NEXT: v_mov_b32_e32 v1, s1
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -281,19 +279,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-LABEL: add_i32_constant:
; GFX13W64: ; %bb.0: ; %entry
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB0_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: s_wait_kmcnt 0x0
; GFX13W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -312,17 +309,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-LABEL: add_i32_constant:
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB0_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W32-NEXT: v_mov_b32_e32 v1, s1
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -346,26 +343,26 @@ entry:
define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(8) %inout, i32 %additive) {
; GFX6-LABEL: add_i32_uniform:
; GFX6: ; %bb.0: ; %entry
-; GFX6-NEXT: s_load_dword s2, s[4:5], 0x11
+; GFX6-NEXT: s_load_dword s6, s[4:5], 0x11
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB1_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_mul_i32 s3, s2, s3
-; GFX6-NEXT: v_mov_b32_e32 v1, s3
+; GFX6-NEXT: s_mul_i32 s2, s6, s2
+; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX6-NEXT: .LBB1_2:
; GFX6-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX6-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX6-NEXT: s_waitcnt vmcnt(0)
; GFX6-NEXT: v_readfirstlane_b32 s4, v1
; GFX6-NEXT: s_mov_b32 s3, 0xf000
@@ -376,26 +373,26 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX8-LABEL: add_i32_uniform:
; GFX8: ; %bb.0: ; %entry
-; GFX8-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX8-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB1_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_mul_i32 s3, s2, s3
-; GFX8-NEXT: v_mov_b32_e32 v1, s3
+; GFX8-NEXT: s_mul_i32 s2, s6, s2
+; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX8-NEXT: .LBB1_2:
; GFX8-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX8-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: v_readfirstlane_b32 s2, v1
; GFX8-NEXT: v_add_u32_e32 v2, vcc, s2, v0
@@ -406,26 +403,26 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX9-LABEL: add_i32_uniform:
; GFX9: ; %bb.0: ; %entry
-; GFX9-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX9-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB1_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s3, s2, s3
-; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_mul_i32 s2, s6, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX9-NEXT: .LBB1_2:
; GFX9-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX9-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX9-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: v_readfirstlane_b32 s2, v1
; GFX9-NEXT: v_mov_b32_e32 v2, 0
@@ -435,30 +432,29 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX10W64-LABEL: add_i32_uniform:
; GFX10W64: ; %bb.0: ; %entry
-; GFX10W64-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX10W64-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB1_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: s_mul_i32 s3, s2, s3
-; GFX10W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX10W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX10W64-NEXT: .LBB1_2:
; GFX10W64-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)
; GFX10W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX10W64-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX10W64-NEXT: s_waitcnt vmcnt(0)
-; GFX10W64-NEXT: s_mov_b32 null, 0
-; GFX10W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX10W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX10W64-NEXT: ...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/226370
More information about the llvm-commits
mailing list