[llvm] AMDGPU: Mark SCC def dead in wave reduction expansions (PR #226370)
Matt Arsenault via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 23:17:54 PDT 2026
https://github.com/arsenm created https://github.com/llvm/llvm-project/pull/226370
Co-Authored-By: Claude Opus 5 <noreply at anthropic.com>
>From 9d908ca6cec2a31d09495e38efe58f7ff30749b7 Mon Sep 17 00:00:00 2001
From: Matt Arsenault <Matthew.Arsenault at amd.com>
Date: Thu, 24 Sep 2026 23:36:39 +0200
Subject: [PATCH] AMDGPU: Mark SCC def dead in wave reduction expansions
Co-Authored-By: Claude Opus 5 <noreply at anthropic.com>
---
llvm/lib/Target/AMDGPU/SIISelLowering.cpp | 17 +-
.../AMDGPU/atomic_optimizations_buffer.ll | 382 +++---
.../atomic_optimizations_global_pointer.ll | 1162 ++++++++---------
.../atomic_optimizations_local_pointer.ll | 577 ++++----
.../AMDGPU/atomic_optimizations_mul_one.ll | 54 +-
.../atomic_optimizations_pixelshader.ll | 50 +-
.../AMDGPU/atomic_optimizations_raw_buffer.ll | 382 +++---
.../atomic_optimizations_struct_buffer.ll | 430 +++---
.../insert_waitcnt_for_precise_memory.ll | 104 +-
9 files changed, 1569 insertions(+), 1589 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index e700adfaa14a3a..f9e4deadc42d7f 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -6055,7 +6055,8 @@ static MachineBasicBlock *lowerWaveReduce(MachineInstr &MI,
auto NewAccumulator =
BuildMI(BB, MI, DL, TII->get(BitCountOpc), NumActiveLanes)
- .addReg(ExecMask);
+ .addReg(ExecMask)
+ .setOperandDead(2); // Dead scc
switch (Opc) {
case AMDGPU::S_XOR_B32:
@@ -6113,7 +6114,8 @@ static MachineBasicBlock *lowerWaveReduce(MachineInstr &MI,
// Take the negation of the source operand.
BuildMI(BB, MI, DL, TII->get(AMDGPU::S_SUB_I32), NegatedVal)
.addImm(0)
- .addReg(SrcReg);
+ .addReg(SrcReg)
+ .setOperandDead(3); // Dead scc
BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MUL_I32), DstReg)
.addReg(NegatedVal)
.addReg(NewAccumulator->getOperand(0).getReg());
@@ -6388,6 +6390,8 @@ static MachineBasicBlock *lowerWaveReduce(MachineInstr &MI,
OpInstr.addImm(0); // opsel
if (hasOMod)
OpInstr.addImm(0); // omod
+ if (TII->isSALU(Opc))
+ OpInstr.setOperandDead(3); // Dead scc
if (ST.getInstrInfo()->isVALU(Opc, /*AllowLDSDMA=*/true)) {
BuildMI(*ComputeLoop, I, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32),
DstReg)
@@ -6503,7 +6507,8 @@ static MachineBasicBlock *lowerWaveReduce(MachineInstr &MI,
case AMDGPU::S_SUB_U64_PSEUDO: {
NewAccumulator = BuildMI(*ComputeLoop, I, DL, TII->get(Opc), DstReg)
.addReg(Accumulator->getOperand(0).getReg())
- .addReg(LaneValue->getOperand(0).getReg());
+ .addReg(LaneValue->getOperand(0).getReg())
+ .setOperandDead(3); // Dead scc
ComputeLoop =
expand64BitScalarArithmetic(*NewAccumulator, ComputeLoop);
break;
@@ -6904,12 +6909,14 @@ static MachineBasicBlock *lowerWaveReduce(MachineInstr &MI,
if (Opc == AMDGPU::S_SUB_I32) {
BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::S_SUB_I32), NegatedReducedVal)
.addImm(0)
- .addReg(ReducedValSGPR);
+ .addReg(ReducedValSGPR)
+ .setOperandDead(3); // Dead scc
} else if (Opc == AMDGPU::S_SUB_U64_PSEUDO) {
auto NegatedValInstr =
BuildMI(*CurrBB, MI, DL, TII->get(Opc), NegatedReducedVal)
.addImm(0)
- .addReg(ReducedValSGPR);
+ .addReg(ReducedValSGPR)
+ .setOperandDead(3); // Dead scc
CurrBB = expand64BitScalarArithmetic(*NegatedValInstr, CurrBB);
}
// Mark the final result as a whole-wave-mode calculation.
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
index 67c115ac2ed767..a912ceb4b16363 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
@@ -23,14 +23,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX6: ; %bb.0: ; %entry
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB0_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_mul_i32 s2, s2, 5
; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
@@ -51,14 +51,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB0_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_mul_i32 s2, s2, 5
; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
@@ -79,14 +79,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB0_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_mul_i32 s2, s2, 5
; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
@@ -105,15 +105,15 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W64-LABEL: add_i32_constant:
; GFX10W64: ; %bb.0: ; %entry
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB0_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: s_mul_i32 s2, s2, 5
; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
@@ -133,14 +133,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W32-LABEL: add_i32_constant:
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB0_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s1
; GFX10W32-NEXT: s_mul_i32 s1, s1, 5
; GFX10W32-NEXT: v_mov_b32_e32 v1, s1
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
@@ -160,19 +160,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-LABEL: add_i32_constant:
; GFX11W64: ; %bb.0: ; %entry
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB0_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], 0 glc
@@ -191,17 +190,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-LABEL: add_i32_constant:
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB0_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W32-NEXT: v_mov_b32_e32 v1, s1
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], 0 glc
@@ -220,19 +219,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-LABEL: add_i32_constant:
; GFX12W64: ; %bb.0: ; %entry
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB0_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: s_wait_kmcnt 0x0
; GFX12W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -252,17 +250,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-LABEL: add_i32_constant:
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB0_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W32-NEXT: v_mov_b32_e32 v1, s1
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -281,19 +279,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-LABEL: add_i32_constant:
; GFX13W64: ; %bb.0: ; %entry
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB0_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: s_wait_kmcnt 0x0
; GFX13W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -312,17 +309,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-LABEL: add_i32_constant:
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB0_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W32-NEXT: v_mov_b32_e32 v1, s1
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -346,26 +343,26 @@ entry:
define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(8) %inout, i32 %additive) {
; GFX6-LABEL: add_i32_uniform:
; GFX6: ; %bb.0: ; %entry
-; GFX6-NEXT: s_load_dword s2, s[4:5], 0x11
+; GFX6-NEXT: s_load_dword s6, s[4:5], 0x11
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB1_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_mul_i32 s3, s2, s3
-; GFX6-NEXT: v_mov_b32_e32 v1, s3
+; GFX6-NEXT: s_mul_i32 s2, s6, s2
+; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX6-NEXT: .LBB1_2:
; GFX6-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX6-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX6-NEXT: s_waitcnt vmcnt(0)
; GFX6-NEXT: v_readfirstlane_b32 s4, v1
; GFX6-NEXT: s_mov_b32 s3, 0xf000
@@ -376,26 +373,26 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX8-LABEL: add_i32_uniform:
; GFX8: ; %bb.0: ; %entry
-; GFX8-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX8-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB1_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_mul_i32 s3, s2, s3
-; GFX8-NEXT: v_mov_b32_e32 v1, s3
+; GFX8-NEXT: s_mul_i32 s2, s6, s2
+; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX8-NEXT: .LBB1_2:
; GFX8-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX8-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: v_readfirstlane_b32 s2, v1
; GFX8-NEXT: v_add_u32_e32 v2, vcc, s2, v0
@@ -406,26 +403,26 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX9-LABEL: add_i32_uniform:
; GFX9: ; %bb.0: ; %entry
-; GFX9-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX9-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB1_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s3, s2, s3
-; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_mul_i32 s2, s6, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX9-NEXT: .LBB1_2:
; GFX9-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX9-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX9-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: v_readfirstlane_b32 s2, v1
; GFX9-NEXT: v_mov_b32_e32 v2, 0
@@ -435,30 +432,29 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX10W64-LABEL: add_i32_uniform:
; GFX10W64: ; %bb.0: ; %entry
-; GFX10W64-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX10W64-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB1_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: s_mul_i32 s3, s2, s3
-; GFX10W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX10W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX10W64-NEXT: .LBB1_2:
; GFX10W64-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)
; GFX10W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX10W64-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX10W64-NEXT: s_waitcnt vmcnt(0)
-; GFX10W64-NEXT: s_mov_b32 null, 0
-; GFX10W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX10W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: v_mad_u64_u32 v[0:1], s[2:3], s2, v0, s[4:5]
+; GFX10W64-NEXT: v_mad_u64_u32 v[0:1], s[2:3], s6, v0, s[2:3]
; GFX10W64-NEXT: v_mov_b32_e32 v1, 0
; GFX10W64-NEXT: global_store_dword v1, v0, s[0:1]
; GFX10W64-NEXT: s_endpgm
@@ -467,14 +463,14 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: s_load_dword s0, s[4:5], 0x44
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s2, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s1
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s1, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB1_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX10W32-NEXT: s_mul_i32 s2, s0, s2
; GFX10W32-NEXT: v_mov_b32_e32 v1, s2
@@ -494,32 +490,31 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX11W64-LABEL: add_i32_uniform:
; GFX11W64: ; %bb.0: ; %entry
-; GFX11W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX11W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB1_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX11W64-NEXT: s_mul_i32 s2, s6, s2
; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], 0 glc
; GFX11W64-NEXT: .LBB1_2:
; GFX11W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX11W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11W64-NEXT: s_waitcnt vmcnt(0)
-; GFX11W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX11W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11W64-NEXT: v_mad_u64_u32 v[1:2], null, s2, v0, s[4:5]
+; GFX11W64-NEXT: v_mad_u64_u32 v[1:2], null, s6, v0, s[2:3]
; GFX11W64-NEXT: v_mov_b32_e32 v0, 0
; GFX11W64-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11W64-NEXT: s_endpgm
@@ -528,15 +523,15 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s2, exec_lo
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB1_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: s_mul_i32 s2, s0, s2
; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -556,32 +551,32 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX12W64-LABEL: add_i32_uniform:
; GFX12W64: ; %bb.0: ; %entry
-; GFX12W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX12W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB1_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX12W64-NEXT: s_wait_kmcnt 0x0
-; GFX12W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX12W64-NEXT: s_mul_i32 s2, s6, s2
; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
; GFX12W64-NEXT: .LBB1_2:
; GFX12W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX12W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX12W64-NEXT: s_wait_loadcnt 0x0
-; GFX12W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX12W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX12W64-NEXT: s_wait_kmcnt 0x0
+; GFX12W64-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s2, v0, s[4:5]
+; GFX12W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s6, v0, s[2:3]
; GFX12W64-NEXT: v_mov_b32_e32 v1, 0
; GFX12W64-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX12W64-NEXT: s_endpgm
@@ -590,15 +585,15 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s2, exec_lo
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB1_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: s_mul_i32 s2, s0, s2
; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -618,32 +613,31 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX13W64-LABEL: add_i32_uniform:
; GFX13W64: ; %bb.0: ; %entry
-; GFX13W64-NEXT: s_load_b32 s2, s[4:5], 0x44 nv
+; GFX13W64-NEXT: s_load_b32 s6, s[4:5], 0x44 nv
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB1_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX13W64-NEXT: s_wait_kmcnt 0x0
-; GFX13W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX13W64-NEXT: s_mul_i32 s2, s6, s2
; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX13W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
; GFX13W64-NEXT: .LBB1_2:
; GFX13W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX13W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13W64-NEXT: s_wait_loadcnt 0x0
-; GFX13W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX13W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX13W64-NEXT: s_wait_kmcnt 0x0
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX13W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s2, v0, s[4:5]
+; GFX13W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s6, v0, s[2:3]
; GFX13W64-NEXT: v_mov_b32_e32 v1, 0
; GFX13W64-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13W64-NEXT: s_endpgm
@@ -652,15 +646,15 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: s_load_b32 s0, s[4:5], 0x44 nv
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s2, exec_lo
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB1_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: s_mul_i32 s2, s0, s2
; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -1777,14 +1771,14 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX6: ; %bb.0: ; %entry
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB5_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_mul_i32 s2, s2, 5
; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
@@ -1806,14 +1800,14 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB5_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_mul_i32 s2, s2, 5
; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
@@ -1835,14 +1829,14 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB5_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_mul_i32 s2, s2, 5
; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
@@ -1862,15 +1856,15 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W64-LABEL: sub_i32_constant:
; GFX10W64: ; %bb.0: ; %entry
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB5_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: s_mul_i32 s2, s2, 5
; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
@@ -1891,14 +1885,14 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W32-LABEL: sub_i32_constant:
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB5_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s1
; GFX10W32-NEXT: s_mul_i32 s1, s1, 5
; GFX10W32-NEXT: v_mov_b32_e32 v1, s1
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
@@ -1919,19 +1913,18 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-LABEL: sub_i32_constant:
; GFX11W64: ; %bb.0: ; %entry
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB5_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], 0 glc
@@ -1951,17 +1944,17 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-LABEL: sub_i32_constant:
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB5_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W32-NEXT: v_mov_b32_e32 v1, s1
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], 0 glc
@@ -1981,19 +1974,18 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-LABEL: sub_i32_constant:
; GFX12W64: ; %bb.0: ; %entry
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB5_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: s_wait_kmcnt 0x0
; GFX12W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -2014,17 +2006,17 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-LABEL: sub_i32_constant:
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB5_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W32-NEXT: v_mov_b32_e32 v1, s1
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -2044,19 +2036,18 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-LABEL: sub_i32_constant:
; GFX13W64: ; %bb.0: ; %entry
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB5_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: s_wait_kmcnt 0x0
; GFX13W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -2076,17 +2067,17 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-LABEL: sub_i32_constant:
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB5_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W32-NEXT: v_mov_b32_e32 v1, s1
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -2110,26 +2101,26 @@ entry:
define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(8) %inout, i32 %subitive) {
; GFX6-LABEL: sub_i32_uniform:
; GFX6: ; %bb.0: ; %entry
-; GFX6-NEXT: s_load_dword s2, s[4:5], 0x11
+; GFX6-NEXT: s_load_dword s6, s[4:5], 0x11
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB6_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_mul_i32 s3, s2, s3
-; GFX6-NEXT: v_mov_b32_e32 v1, s3
+; GFX6-NEXT: s_mul_i32 s2, s6, s2
+; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: buffer_atomic_sub v1, off, s[8:11], 0 glc
; GFX6-NEXT: .LBB6_2:
; GFX6-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX6-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX6-NEXT: s_waitcnt vmcnt(0)
; GFX6-NEXT: v_readfirstlane_b32 s4, v1
; GFX6-NEXT: s_mov_b32 s3, 0xf000
@@ -2140,26 +2131,26 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX8-LABEL: sub_i32_uniform:
; GFX8: ; %bb.0: ; %entry
-; GFX8-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX8-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB6_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_mul_i32 s3, s2, s3
-; GFX8-NEXT: v_mov_b32_e32 v1, s3
+; GFX8-NEXT: s_mul_i32 s2, s6, s2
+; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: buffer_atomic_sub v1, off, s[8:11], 0 glc
; GFX8-NEXT: .LBB6_2:
; GFX8-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX8-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: v_readfirstlane_b32 s2, v1
; GFX8-NEXT: v_sub_u32_e32 v2, vcc, s2, v0
@@ -2170,26 +2161,26 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX9-LABEL: sub_i32_uniform:
; GFX9: ; %bb.0: ; %entry
-; GFX9-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX9-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB6_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s3, s2, s3
-; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_mul_i32 s2, s6, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: buffer_atomic_sub v1, off, s[8:11], 0 glc
; GFX9-NEXT: .LBB6_2:
; GFX9-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX9-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX9-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: v_readfirstlane_b32 s2, v1
; GFX9-NEXT: v_mov_b32_e32 v2, 0
@@ -2199,27 +2190,27 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX10W64-LABEL: sub_i32_uniform:
; GFX10W64: ; %bb.0: ; %entry
-; GFX10W64-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX10W64-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB6_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: s_mul_i32 s3, s2, s3
-; GFX10W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX10W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: buffer_atomic_sub v1, off, s[8:11], 0 glc
; GFX10W64-NEXT: .LBB6_2:
; GFX10W64-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)
; GFX10W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX10W64-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX10W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX10W64-NEXT: s_waitcnt vmcnt(0)
; GFX10W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX10W64-NEXT: v_mov_b32_e32 v1, 0
@@ -2231,14 +2222,14 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: s_load_dword s0, s[4:5], 0x44
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s2, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s1
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s1, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB6_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX10W32-NEXT: s_mul_i32 s2, s0, s2
; GFX10W32-NEXT: v_mov_b32_e32 v1, s2
@@ -2258,29 +2249,28 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX11W64-LABEL: sub_i32_uniform:
; GFX11W64: ; %bb.0: ; %entry
-; GFX11W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX11W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB6_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX11W64-NEXT: s_mul_i32 s2, s6, s2
; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], 0 glc
; GFX11W64-NEXT: .LBB6_2:
; GFX11W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX11W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX11W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX11W64-NEXT: s_waitcnt vmcnt(0)
; GFX11W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX11W64-NEXT: v_mov_b32_e32 v1, 0
@@ -2293,15 +2283,15 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s2, exec_lo
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB6_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: s_mul_i32 s2, s0, s2
; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -2322,29 +2312,28 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX12W64-LABEL: sub_i32_uniform:
; GFX12W64: ; %bb.0: ; %entry
-; GFX12W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX12W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB6_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX12W64-NEXT: s_wait_kmcnt 0x0
-; GFX12W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX12W64-NEXT: s_mul_i32 s2, s6, s2
; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
; GFX12W64-NEXT: .LBB6_2:
; GFX12W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX12W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX12W64-NEXT: s_wait_kmcnt 0x0
-; GFX12W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX12W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX12W64-NEXT: s_wait_loadcnt 0x0
; GFX12W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX12W64-NEXT: v_mov_b32_e32 v1, 0
@@ -2358,15 +2347,15 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s2, exec_lo
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB6_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: s_mul_i32 s2, s0, s2
; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -2388,29 +2377,28 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX13W64-LABEL: sub_i32_uniform:
; GFX13W64: ; %bb.0: ; %entry
-; GFX13W64-NEXT: s_load_b32 s2, s[4:5], 0x44 nv
+; GFX13W64-NEXT: s_load_b32 s6, s[4:5], 0x44 nv
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB6_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX13W64-NEXT: s_wait_kmcnt 0x0
-; GFX13W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX13W64-NEXT: s_mul_i32 s2, s6, s2
; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX13W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
; GFX13W64-NEXT: .LBB6_2:
; GFX13W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX13W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13W64-NEXT: s_wait_kmcnt 0x0
-; GFX13W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX13W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX13W64-NEXT: s_wait_loadcnt 0x0
; GFX13W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX13W64-NEXT: v_mov_b32_e32 v1, 0
@@ -2423,15 +2411,15 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: s_load_b32 s0, s[4:5], 0x44 nv
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s2, exec_lo
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB6_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: s_mul_i32 s2, s0, s2
; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll
index f7045fe08693a1..28e02c5434c6f8 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll
@@ -40,13 +40,13 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX7LESS-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x9
; GFX7LESS-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX7LESS-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX7LESS-NEXT: s_mov_b64 s[4:5], exec
-; GFX7LESS-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GFX7LESS-NEXT: s_mov_b64 s[6:7], exec
; GFX7LESS-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX7LESS-NEXT: ; implicit-def: $vgpr1
; GFX7LESS-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX7LESS-NEXT: s_cbranch_execz .LBB0_2
; GFX7LESS-NEXT: ; %bb.1:
+; GFX7LESS-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX7LESS-NEXT: s_mul_i32 s6, s6, 5
; GFX7LESS-NEXT: s_mov_b32 s11, 0xf000
; GFX7LESS-NEXT: s_mov_b32 s10, -1
@@ -72,13 +72,13 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX8-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[4:5], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GFX8-NEXT: s_mov_b64 s[6:7], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX8-NEXT: s_cbranch_execz .LBB0_2
; GFX8-NEXT: ; %bb.1:
+; GFX8-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX8-NEXT: s_mul_i32 s6, s6, 5
; GFX8-NEXT: s_mov_b32 s11, 0xf000
; GFX8-NEXT: s_mov_b32 s10, -1
@@ -104,13 +104,13 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GFX9-NEXT: s_mov_b64 s[6:7], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX9-NEXT: s_cbranch_execz .LBB0_2
; GFX9-NEXT: ; %bb.1:
+; GFX9-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX9-NEXT: s_mul_i32 s6, s6, 5
; GFX9-NEXT: s_mov_b32 s11, 0xf000
; GFX9-NEXT: s_mov_b32 s10, -1
@@ -135,18 +135,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1064: ; %bb.0: ; %entry
; GFX1064-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[4:5], exec
+; GFX1064-NEXT: s_mov_b64 s[6:7], exec
; GFX1064-NEXT: ; implicit-def: $vgpr1
-; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1064-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB0_2
; GFX1064-NEXT: ; %bb.1:
-; GFX1064-NEXT: s_mul_i32 s6, s6, 5
+; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1064-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1064-NEXT: v_mov_b32_e32 v1, s6
+; GFX1064-NEXT: s_mul_i32 s6, s6, 5
; GFX1064-NEXT: s_mov_b32 s10, -1
+; GFX1064-NEXT: v_mov_b32_e32 v1, s6
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: s_mov_b32 s8, s2
; GFX1064-NEXT: s_mov_b32 s9, s3
@@ -169,17 +169,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
+; GFX1032-NEXT: s_mov_b32 s5, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s4
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1032-NEXT: s_and_saveexec_b32 s4, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB0_2
; GFX1032-NEXT: ; %bb.1:
-; GFX1032-NEXT: s_mul_i32 s5, s5, 5
+; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s5
; GFX1032-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1032-NEXT: v_mov_b32_e32 v1, s5
+; GFX1032-NEXT: s_mul_i32 s5, s5, 5
; GFX1032-NEXT: s_mov_b32 s10, -1
+; GFX1032-NEXT: v_mov_b32_e32 v1, s5
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_mov_b32 s8, s2
; GFX1032-NEXT: s_mov_b32 s9, s3
@@ -202,20 +202,19 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164: ; %bb.0: ; %entry
; GFX1164-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[6:7], exec
; GFX1164-NEXT: s_mov_b64 s[4:5], exec
; GFX1164-NEXT: ; implicit-def: $vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1164-NEXT: s_mov_b64 s[4:5], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1164-NEXT: s_cbranch_execz .LBB0_2
; GFX1164-NEXT: ; %bb.1:
-; GFX1164-NEXT: s_mul_i32 s6, s6, 5
+; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1164-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1164-NEXT: v_mov_b32_e32 v1, s6
+; GFX1164-NEXT: s_mul_i32 s6, s6, 5
; GFX1164-NEXT: s_mov_b32 s10, -1
+; GFX1164-NEXT: v_mov_b32_e32 v1, s6
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-NEXT: s_mov_b32 s8, s2
; GFX1164-NEXT: s_mov_b32 s9, s3
@@ -237,18 +236,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s5, exec_lo
; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1132-NEXT: s_mov_b32 s4, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB0_2
; GFX1132-NEXT: ; %bb.1:
-; GFX1132-NEXT: s_mul_i32 s5, s5, 5
+; GFX1132-NEXT: s_bcnt1_i32_b32 s5, s5
; GFX1132-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1132-NEXT: v_mov_b32_e32 v1, s5
+; GFX1132-NEXT: s_mul_i32 s5, s5, 5
; GFX1132-NEXT: s_mov_b32 s10, -1
+; GFX1132-NEXT: v_mov_b32_e32 v1, s5
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_mov_b32 s8, s2
; GFX1132-NEXT: s_mov_b32 s9, s3
@@ -270,20 +269,19 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1264: ; %bb.0: ; %entry
; GFX1264-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1264-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1264-NEXT: s_mov_b64 s[6:7], exec
; GFX1264-NEXT: s_mov_b64 s[4:5], exec
; GFX1264-NEXT: ; implicit-def: $vgpr1
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1264-NEXT: s_mov_b64 s[4:5], exec
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1264-NEXT: s_cbranch_execz .LBB0_2
; GFX1264-NEXT: ; %bb.1:
-; GFX1264-NEXT: s_mul_i32 s6, s6, 5
+; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1264-NEXT: v_mov_b32_e32 v1, s6
+; GFX1264-NEXT: s_mul_i32 s6, s6, 5
; GFX1264-NEXT: s_mov_b32 s10, -1
+; GFX1264-NEXT: v_mov_b32_e32 v1, s6
; GFX1264-NEXT: s_wait_kmcnt 0x0
; GFX1264-NEXT: s_mov_b32 s8, s2
; GFX1264-NEXT: s_mov_b32 s9, s3
@@ -304,18 +302,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1232: ; %bb.0: ; %entry
; GFX1232-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1232-NEXT: s_mov_b32 s5, exec_lo
; GFX1232-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-NEXT: ; implicit-def: $vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1232-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1232-NEXT: s_mov_b32 s4, exec_lo
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1232-NEXT: s_cbranch_execz .LBB0_2
; GFX1232-NEXT: ; %bb.1:
-; GFX1232-NEXT: s_mul_i32 s5, s5, 5
+; GFX1232-NEXT: s_bcnt1_i32_b32 s5, s5
; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1232-NEXT: v_mov_b32_e32 v1, s5
+; GFX1232-NEXT: s_mul_i32 s5, s5, 5
; GFX1232-NEXT: s_mov_b32 s10, -1
+; GFX1232-NEXT: v_mov_b32_e32 v1, s5
; GFX1232-NEXT: s_wait_kmcnt 0x0
; GFX1232-NEXT: s_mov_b32 s8, s2
; GFX1232-NEXT: s_mov_b32 s9, s3
@@ -336,20 +334,19 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-NEXT: s_mov_b64 s[4:5], exec
; GFX1364-NEXT: ; implicit-def: $vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1364-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1364-NEXT: s_cbranch_execz .LBB0_2
; GFX1364-NEXT: ; %bb.1:
-; GFX1364-NEXT: s_mul_i32 s6, s6, 5
+; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1364-NEXT: v_mov_b32_e32 v1, s6
+; GFX1364-NEXT: s_mul_i32 s6, s6, 5
; GFX1364-NEXT: s_mov_b32 s10, -1
+; GFX1364-NEXT: v_mov_b32_e32 v1, s6
; GFX1364-NEXT: s_wait_kmcnt 0x0
; GFX1364-NEXT: s_mov_b32 s8, s2
; GFX1364-NEXT: s_mov_b32 s9, s3
@@ -373,18 +370,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s5, exec_lo
; GFX1332-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1332-NEXT: s_mov_b32 s4, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB0_2
; GFX1332-NEXT: ; %bb.1:
-; GFX1332-NEXT: s_mul_i32 s5, s5, 5
+; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s5
; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1332-NEXT: v_mov_b32_e32 v1, s5
+; GFX1332-NEXT: s_mul_i32 s5, s5, 5
; GFX1332-NEXT: s_mov_b32 s10, -1
+; GFX1332-NEXT: v_mov_b32_e32 v1, s5
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mov_b32 s8, s2
; GFX1332-NEXT: s_mov_b32 s9, s3
@@ -413,30 +410,30 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX7LESS-LABEL: add_i32_uniform:
; GFX7LESS: ; %bb.0: ; %entry
; GFX7LESS-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x9
-; GFX7LESS-NEXT: s_load_dword s6, s[4:5], 0xd
+; GFX7LESS-NEXT: s_load_dword s8, s[4:5], 0xd
; GFX7LESS-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX7LESS-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX7LESS-NEXT: s_mov_b64 s[4:5], exec
-; GFX7LESS-NEXT: s_bcnt1_i32_b64 s7, s[4:5]
+; GFX7LESS-NEXT: s_mov_b64 s[6:7], exec
; GFX7LESS-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX7LESS-NEXT: ; implicit-def: $vgpr1
; GFX7LESS-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX7LESS-NEXT: s_cbranch_execz .LBB1_2
; GFX7LESS-NEXT: ; %bb.1:
+; GFX7LESS-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: s_mul_i32 s7, s6, s7
-; GFX7LESS-NEXT: s_mov_b32 s11, 0xf000
-; GFX7LESS-NEXT: s_mov_b32 s10, -1
-; GFX7LESS-NEXT: s_mov_b32 s8, s2
-; GFX7LESS-NEXT: s_mov_b32 s9, s3
-; GFX7LESS-NEXT: v_mov_b32_e32 v1, s7
-; GFX7LESS-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
+; GFX7LESS-NEXT: s_mul_i32 s6, s8, s6
+; GFX7LESS-NEXT: s_mov_b32 s15, 0xf000
+; GFX7LESS-NEXT: s_mov_b32 s14, -1
+; GFX7LESS-NEXT: s_mov_b32 s12, s2
+; GFX7LESS-NEXT: s_mov_b32 s13, s3
+; GFX7LESS-NEXT: v_mov_b32_e32 v1, s6
+; GFX7LESS-NEXT: buffer_atomic_add v1, off, s[12:15], 0 glc
; GFX7LESS-NEXT: s_waitcnt vmcnt(0)
; GFX7LESS-NEXT: buffer_wbinvl1
; GFX7LESS-NEXT: .LBB1_2:
; GFX7LESS-NEXT: s_or_b64 exec, exec, s[4:5]
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: v_mul_lo_u32 v0, s6, v0
+; GFX7LESS-NEXT: v_mul_lo_u32 v0, s8, v0
; GFX7LESS-NEXT: v_readfirstlane_b32 s4, v1
; GFX7LESS-NEXT: s_mov_b32 s3, 0xf000
; GFX7LESS-NEXT: s_mov_b32 s2, -1
@@ -447,30 +444,30 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX8-LABEL: add_i32_uniform:
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
-; GFX8-NEXT: s_load_dword s6, s[4:5], 0x34
+; GFX8-NEXT: s_load_dword s8, s[4:5], 0x34
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[4:5], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s7, s[4:5]
+; GFX8-NEXT: s_mov_b64 s[6:7], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX8-NEXT: s_cbranch_execz .LBB1_2
; GFX8-NEXT: ; %bb.1:
+; GFX8-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_mul_i32 s7, s6, s7
-; GFX8-NEXT: s_mov_b32 s11, 0xf000
-; GFX8-NEXT: s_mov_b32 s10, -1
-; GFX8-NEXT: s_mov_b32 s8, s2
-; GFX8-NEXT: s_mov_b32 s9, s3
-; GFX8-NEXT: v_mov_b32_e32 v1, s7
-; GFX8-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
+; GFX8-NEXT: s_mul_i32 s6, s8, s6
+; GFX8-NEXT: s_mov_b32 s15, 0xf000
+; GFX8-NEXT: s_mov_b32 s14, -1
+; GFX8-NEXT: s_mov_b32 s12, s2
+; GFX8-NEXT: s_mov_b32 s13, s3
+; GFX8-NEXT: v_mov_b32_e32 v1, s6
+; GFX8-NEXT: buffer_atomic_add v1, off, s[12:15], 0 glc
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: buffer_wbinvl1_vol
; GFX8-NEXT: .LBB1_2:
; GFX8-NEXT: s_or_b64 exec, exec, s[4:5]
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mul_lo_u32 v0, s6, v0
+; GFX8-NEXT: v_mul_lo_u32 v0, s8, v0
; GFX8-NEXT: v_readfirstlane_b32 s4, v1
; GFX8-NEXT: s_mov_b32 s3, 0xf000
; GFX8-NEXT: s_mov_b32 s2, -1
@@ -481,30 +478,30 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX9-LABEL: add_i32_uniform:
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
-; GFX9-NEXT: s_load_dword s6, s[4:5], 0x34
+; GFX9-NEXT: s_load_dword s8, s[4:5], 0x34
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s7, s[4:5]
+; GFX9-NEXT: s_mov_b64 s[6:7], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX9-NEXT: s_cbranch_execz .LBB1_2
; GFX9-NEXT: ; %bb.1:
+; GFX9-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s7, s6, s7
-; GFX9-NEXT: s_mov_b32 s11, 0xf000
-; GFX9-NEXT: s_mov_b32 s10, -1
-; GFX9-NEXT: s_mov_b32 s8, s2
-; GFX9-NEXT: s_mov_b32 s9, s3
-; GFX9-NEXT: v_mov_b32_e32 v1, s7
-; GFX9-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
+; GFX9-NEXT: s_mul_i32 s6, s8, s6
+; GFX9-NEXT: s_mov_b32 s15, 0xf000
+; GFX9-NEXT: s_mov_b32 s14, -1
+; GFX9-NEXT: s_mov_b32 s12, s2
+; GFX9-NEXT: s_mov_b32 s13, s3
+; GFX9-NEXT: v_mov_b32_e32 v1, s6
+; GFX9-NEXT: buffer_atomic_add v1, off, s[12:15], 0 glc
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: buffer_wbinvl1_vol
; GFX9-NEXT: .LBB1_2:
; GFX9-NEXT: s_or_b64 exec, exec, s[4:5]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mul_lo_u32 v0, s6, v0
+; GFX9-NEXT: v_mul_lo_u32 v0, s8, v0
; GFX9-NEXT: v_readfirstlane_b32 s4, v1
; GFX9-NEXT: s_mov_b32 s3, 0xf000
; GFX9-NEXT: s_mov_b32 s2, -1
@@ -516,24 +513,24 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1064: ; %bb.0: ; %entry
; GFX1064-NEXT: s_clause 0x1
; GFX1064-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
-; GFX1064-NEXT: s_load_dword s6, s[4:5], 0x34
+; GFX1064-NEXT: s_load_dword s8, s[4:5], 0x34
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[4:5], exec
+; GFX1064-NEXT: s_mov_b64 s[6:7], exec
; GFX1064-NEXT: ; implicit-def: $vgpr1
-; GFX1064-NEXT: s_bcnt1_i32_b64 s7, s[4:5]
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1064-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB1_2
; GFX1064-NEXT: ; %bb.1:
+; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1064-NEXT: s_mov_b32 s15, 0x31016000
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1064-NEXT: s_mul_i32 s7, s6, s7
-; GFX1064-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1064-NEXT: v_mov_b32_e32 v1, s7
-; GFX1064-NEXT: s_mov_b32 s10, -1
-; GFX1064-NEXT: s_mov_b32 s8, s2
-; GFX1064-NEXT: s_mov_b32 s9, s3
-; GFX1064-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
+; GFX1064-NEXT: s_mul_i32 s6, s8, s6
+; GFX1064-NEXT: s_mov_b32 s14, -1
+; GFX1064-NEXT: v_mov_b32_e32 v1, s6
+; GFX1064-NEXT: s_mov_b32 s12, s2
+; GFX1064-NEXT: s_mov_b32 s13, s3
+; GFX1064-NEXT: buffer_atomic_add v1, off, s[12:15], 0 glc
; GFX1064-NEXT: s_waitcnt vmcnt(0)
; GFX1064-NEXT: buffer_gl1_inv
; GFX1064-NEXT: buffer_gl0_inv
@@ -542,7 +539,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1064-NEXT: s_or_b64 exec, exec, s[4:5]
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: v_readfirstlane_b32 s2, v1
-; GFX1064-NEXT: v_mad_u64_u32 v[0:1], s[2:3], s6, v0, s[2:3]
+; GFX1064-NEXT: v_mad_u64_u32 v[0:1], s[2:3], s8, v0, s[2:3]
; GFX1064-NEXT: s_mov_b32 s3, 0x31016000
; GFX1064-NEXT: s_mov_b32 s2, -1
; GFX1064-NEXT: buffer_store_dword v0, off, s[0:3], 0
@@ -554,18 +551,18 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: s_load_dword s6, s[4:5], 0x34
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
+; GFX1032-NEXT: s_mov_b32 s5, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s4
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1032-NEXT: s_and_saveexec_b32 s4, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB1_2
; GFX1032-NEXT: ; %bb.1:
+; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s5
+; GFX1032-NEXT: s_mov_b32 s11, 0x31016000
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_mul_i32 s5, s6, s5
-; GFX1032-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1032-NEXT: v_mov_b32_e32 v1, s5
; GFX1032-NEXT: s_mov_b32 s10, -1
+; GFX1032-NEXT: v_mov_b32_e32 v1, s5
; GFX1032-NEXT: s_mov_b32 s8, s2
; GFX1032-NEXT: s_mov_b32 s9, s3
; GFX1032-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
@@ -587,26 +584,25 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164: ; %bb.0: ; %entry
; GFX1164-NEXT: s_clause 0x1
; GFX1164-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX1164-NEXT: s_load_b32 s6, s[4:5], 0x34
+; GFX1164-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[6:7], exec
; GFX1164-NEXT: s_mov_b64 s[4:5], exec
; GFX1164-NEXT: ; implicit-def: $vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s7, s[4:5]
-; GFX1164-NEXT: s_mov_b64 s[4:5], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1164-NEXT: s_cbranch_execz .LBB1_2
; GFX1164-NEXT: ; %bb.1:
+; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1164-NEXT: s_mov_b32 s15, 0x31016000
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1164-NEXT: s_mul_i32 s7, s6, s7
-; GFX1164-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1164-NEXT: v_mov_b32_e32 v1, s7
-; GFX1164-NEXT: s_mov_b32 s10, -1
-; GFX1164-NEXT: s_mov_b32 s8, s2
-; GFX1164-NEXT: s_mov_b32 s9, s3
-; GFX1164-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], 0 glc
+; GFX1164-NEXT: s_mul_i32 s6, s8, s6
+; GFX1164-NEXT: s_mov_b32 s14, -1
+; GFX1164-NEXT: v_mov_b32_e32 v1, s6
+; GFX1164-NEXT: s_mov_b32 s12, s2
+; GFX1164-NEXT: s_mov_b32 s13, s3
+; GFX1164-NEXT: buffer_atomic_add_u32 v1, off, s[12:15], 0 glc
; GFX1164-NEXT: s_waitcnt vmcnt(0)
; GFX1164-NEXT: buffer_gl1_inv
; GFX1164-NEXT: buffer_gl0_inv
@@ -615,7 +611,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-NEXT: v_readfirstlane_b32 s2, v1
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1164-NEXT: v_mad_u64_u32 v[1:2], null, s6, v0, s[2:3]
+; GFX1164-NEXT: v_mad_u64_u32 v[1:2], null, s8, v0, s[2:3]
; GFX1164-NEXT: s_mov_b32 s3, 0x31016000
; GFX1164-NEXT: s_mov_b32 s2, -1
; GFX1164-NEXT: buffer_store_b32 v1, off, s[0:3], 0
@@ -627,19 +623,19 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-NEXT: s_load_b32 s4, s[4:5], 0x34
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s6, exec_lo
; GFX1132-NEXT: s_mov_b32 s5, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s6, s5
-; GFX1132-NEXT: s_mov_b32 s5, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB1_2
; GFX1132-NEXT: ; %bb.1:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s6, s6
+; GFX1132-NEXT: s_mov_b32 s11, 0x31016000
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_mul_i32 s6, s4, s6
-; GFX1132-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1132-NEXT: v_mov_b32_e32 v1, s6
; GFX1132-NEXT: s_mov_b32 s10, -1
+; GFX1132-NEXT: v_mov_b32_e32 v1, s6
; GFX1132-NEXT: s_mov_b32 s8, s2
; GFX1132-NEXT: s_mov_b32 s9, s3
; GFX1132-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], 0 glc
@@ -661,26 +657,25 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1264: ; %bb.0: ; %entry
; GFX1264-NEXT: s_clause 0x1
; GFX1264-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX1264-NEXT: s_load_b32 s6, s[4:5], 0x34
+; GFX1264-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1264-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1264-NEXT: s_mov_b64 s[6:7], exec
; GFX1264-NEXT: s_mov_b64 s[4:5], exec
; GFX1264-NEXT: ; implicit-def: $vgpr1
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1264-NEXT: s_bcnt1_i32_b64 s7, s[4:5]
-; GFX1264-NEXT: s_mov_b64 s[4:5], exec
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1264-NEXT: s_cbranch_execz .LBB1_2
; GFX1264-NEXT: ; %bb.1:
+; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1264-NEXT: s_mov_b32 s15, 0x31016000
; GFX1264-NEXT: s_wait_kmcnt 0x0
-; GFX1264-NEXT: s_mul_i32 s7, s6, s7
-; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1264-NEXT: v_mov_b32_e32 v1, s7
-; GFX1264-NEXT: s_mov_b32 s10, -1
-; GFX1264-NEXT: s_mov_b32 s8, s2
-; GFX1264-NEXT: s_mov_b32 s9, s3
-; GFX1264-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN scope:SCOPE_DEV
+; GFX1264-NEXT: s_mul_i32 s6, s8, s6
+; GFX1264-NEXT: s_mov_b32 s14, -1
+; GFX1264-NEXT: v_mov_b32_e32 v1, s6
+; GFX1264-NEXT: s_mov_b32 s12, s2
+; GFX1264-NEXT: s_mov_b32 s13, s3
+; GFX1264-NEXT: buffer_atomic_add_u32 v1, off, s[12:15], null th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX1264-NEXT: s_wait_loadcnt 0x0
; GFX1264-NEXT: global_inv scope:SCOPE_DEV
; GFX1264-NEXT: .LBB1_2:
@@ -688,7 +683,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1264-NEXT: s_wait_kmcnt 0x0
; GFX1264-NEXT: v_readfirstlane_b32 s2, v1
; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1264-NEXT: v_mad_co_u64_u32 v[0:1], null, s6, v0, s[2:3]
+; GFX1264-NEXT: v_mad_co_u64_u32 v[0:1], null, s8, v0, s[2:3]
; GFX1264-NEXT: s_mov_b32 s3, 0x31016000
; GFX1264-NEXT: s_mov_b32 s2, -1
; GFX1264-NEXT: buffer_store_b32 v0, off, s[0:3], null
@@ -700,19 +695,19 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1232-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-NEXT: s_load_b32 s4, s[4:5], 0x34
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1232-NEXT: s_mov_b32 s6, exec_lo
; GFX1232-NEXT: s_mov_b32 s5, exec_lo
; GFX1232-NEXT: ; implicit-def: $vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1232-NEXT: s_bcnt1_i32_b32 s6, s5
-; GFX1232-NEXT: s_mov_b32 s5, exec_lo
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1232-NEXT: s_cbranch_execz .LBB1_2
; GFX1232-NEXT: ; %bb.1:
+; GFX1232-NEXT: s_bcnt1_i32_b32 s6, s6
+; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
; GFX1232-NEXT: s_wait_kmcnt 0x0
; GFX1232-NEXT: s_mul_i32 s6, s4, s6
-; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1232-NEXT: v_mov_b32_e32 v1, s6
; GFX1232-NEXT: s_mov_b32 s10, -1
+; GFX1232-NEXT: v_mov_b32_e32 v1, s6
; GFX1232-NEXT: s_mov_b32 s8, s2
; GFX1232-NEXT: s_mov_b32 s9, s3
; GFX1232-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -733,28 +728,27 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: s_clause 0x1
; GFX1364-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX1364-NEXT: s_load_b32 s6, s[4:5], 0x34 nv
+; GFX1364-NEXT: s_load_b32 s8, s[4:5], 0x34 nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-NEXT: s_mov_b64 s[4:5], exec
; GFX1364-NEXT: ; implicit-def: $vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s7, s[4:5]
-; GFX1364-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1364-NEXT: s_cbranch_execz .LBB1_2
; GFX1364-NEXT: ; %bb.1:
+; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1364-NEXT: s_mov_b32 s15, 0x31016000
; GFX1364-NEXT: s_wait_kmcnt 0x0
-; GFX1364-NEXT: s_mul_i32 s7, s6, s7
-; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1364-NEXT: v_mov_b32_e32 v1, s7
-; GFX1364-NEXT: s_mov_b32 s10, -1
-; GFX1364-NEXT: s_mov_b32 s8, s2
-; GFX1364-NEXT: s_mov_b32 s9, s3
+; GFX1364-NEXT: s_mul_i32 s6, s8, s6
+; GFX1364-NEXT: s_mov_b32 s14, -1
+; GFX1364-NEXT: v_mov_b32_e32 v1, s6
+; GFX1364-NEXT: s_mov_b32 s12, s2
+; GFX1364-NEXT: s_mov_b32 s13, s3
; GFX1364-NEXT: global_wb scope:SCOPE_DEV
; GFX1364-NEXT: s_wait_storecnt 0x0
-; GFX1364-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN scope:SCOPE_DEV
+; GFX1364-NEXT: buffer_atomic_add_u32 v1, off, s[12:15], null th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX1364-NEXT: s_wait_loadcnt 0x0
; GFX1364-NEXT: global_inv scope:SCOPE_DEV
; GFX1364-NEXT: s_wait_loadcnt 0x0
@@ -763,7 +757,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: s_wait_kmcnt 0x0
; GFX1364-NEXT: v_readfirstlane_b32 s2, v1
; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX1364-NEXT: v_mad_co_u64_u32 v[0:1], null, s6, v0, s[2:3]
+; GFX1364-NEXT: v_mad_co_u64_u32 v[0:1], null, s8, v0, s[2:3]
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: buffer_store_b32 v0, off, s[0:3], null
@@ -775,19 +769,19 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-NEXT: s_load_b32 s4, s[4:5], 0x34 nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s6, exec_lo
; GFX1332-NEXT: s_mov_b32 s5, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s6, s5
-; GFX1332-NEXT: s_mov_b32 s5, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB1_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s6, s6
+; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mul_i32 s6, s4, s6
-; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1332-NEXT: v_mov_b32_e32 v1, s6
; GFX1332-NEXT: s_mov_b32 s10, -1
+; GFX1332-NEXT: v_mov_b32_e32 v1, s6
; GFX1332-NEXT: s_mov_b32 s8, s2
; GFX1332-NEXT: s_mov_b32 s9, s3
; GFX1332-NEXT: global_wb scope:SCOPE_DEV
@@ -1938,24 +1932,24 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GFX9-NEXT: s_mov_b64 s[6:7], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX9-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX9-NEXT: s_cbranch_execz .LBB3_2
; GFX9-NEXT: ; %bb.1:
-; GFX9-NEXT: s_mul_i32 s12, s6, 5
-; GFX9-NEXT: s_mul_hi_u32 s7, 5, s6
-; GFX9-NEXT: s_mul_i32 s6, s6, 0
-; GFX9-NEXT: s_add_u32 s13, s7, s6
+; GFX9-NEXT: s_bcnt1_i32_b64 s7, s[6:7]
+; GFX9-NEXT: s_mul_i32 s6, s7, 5
+; GFX9-NEXT: s_mul_hi_u32 s8, 5, s7
+; GFX9-NEXT: s_mul_i32 s7, s7, 0
+; GFX9-NEXT: s_add_u32 s7, s8, s7
; GFX9-NEXT: s_mov_b32 s11, 0xf000
; GFX9-NEXT: s_mov_b32 s10, -1
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-NEXT: s_mov_b32 s8, s2
; GFX9-NEXT: s_mov_b32 s9, s3
-; GFX9-NEXT: v_mov_b32_e32 v0, s12
-; GFX9-NEXT: v_mov_b32_e32 v1, s13
+; GFX9-NEXT: v_mov_b32_e32 v0, s6
+; GFX9-NEXT: v_mov_b32_e32 v1, s7
; GFX9-NEXT: buffer_atomic_add_x2 v[0:1], off, s[8:11], 0 glc
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: buffer_wbinvl1_vol
@@ -1977,21 +1971,21 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1064: ; %bb.0: ; %entry
; GFX1064-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[4:5], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GFX1064-NEXT: s_mov_b64 s[6:7], exec
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1064-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB3_2
; GFX1064-NEXT: ; %bb.1:
+; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1064-NEXT: s_mov_b32 s11, 0x31016000
; GFX1064-NEXT: s_mul_hi_u32 s7, 5, s6
; GFX1064-NEXT: s_mul_i32 s8, s6, 0
; GFX1064-NEXT: s_mul_i32 s6, s6, 5
; GFX1064-NEXT: s_add_u32 s7, s7, s8
; GFX1064-NEXT: v_mov_b32_e32 v0, s6
; GFX1064-NEXT: v_mov_b32_e32 v1, s7
-; GFX1064-NEXT: s_mov_b32 s11, 0x31016000
; GFX1064-NEXT: s_mov_b32 s10, -1
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: s_mov_b32 s8, s2
@@ -2016,20 +2010,20 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
+; GFX1032-NEXT: s_mov_b32 s5, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s4
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v2
; GFX1032-NEXT: s_and_saveexec_b32 s4, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB3_2
; GFX1032-NEXT: ; %bb.1:
+; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s5
+; GFX1032-NEXT: s_mov_b32 s11, 0x31016000
; GFX1032-NEXT: s_mul_hi_u32 s7, 5, s5
; GFX1032-NEXT: s_mul_i32 s8, s5, 0
; GFX1032-NEXT: s_mul_i32 s6, s5, 5
; GFX1032-NEXT: s_add_u32 s7, s7, s8
; GFX1032-NEXT: v_mov_b32_e32 v0, s6
; GFX1032-NEXT: v_mov_b32_e32 v1, s7
-; GFX1032-NEXT: s_mov_b32 s11, 0x31016000
; GFX1032-NEXT: s_mov_b32 s10, -1
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_mov_b32 s8, s2
@@ -2054,23 +2048,22 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164: ; %bb.0: ; %entry
; GFX1164-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[6:7], exec
; GFX1164-NEXT: s_mov_b64 s[4:5], exec
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1164-NEXT: s_mov_b64 s[4:5], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB3_2
; GFX1164-NEXT: ; %bb.1:
+; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1164-NEXT: s_mov_b32 s11, 0x31016000
; GFX1164-NEXT: s_mul_hi_u32 s7, 5, s6
; GFX1164-NEXT: s_mul_i32 s8, s6, 0
; GFX1164-NEXT: s_mul_i32 s6, s6, 5
; GFX1164-NEXT: s_add_u32 s7, s7, s8
; GFX1164-NEXT: v_mov_b32_e32 v0, s6
; GFX1164-NEXT: v_mov_b32_e32 v1, s7
-; GFX1164-NEXT: s_mov_b32 s11, 0x31016000
; GFX1164-NEXT: s_mov_b32 s10, -1
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-NEXT: s_mov_b32 s8, s2
@@ -2095,21 +2088,21 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s5, exec_lo
; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1132-NEXT: s_mov_b32 s4, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB3_2
; GFX1132-NEXT: ; %bb.1:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s5, s5
+; GFX1132-NEXT: s_mov_b32 s11, 0x31016000
; GFX1132-NEXT: s_mul_hi_u32 s7, 5, s5
; GFX1132-NEXT: s_mul_i32 s8, s5, 0
; GFX1132-NEXT: s_mul_i32 s6, s5, 5
; GFX1132-NEXT: s_add_u32 s7, s7, s8
; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
-; GFX1132-NEXT: s_mov_b32 s11, 0x31016000
; GFX1132-NEXT: s_mov_b32 s10, -1
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_mov_b32 s8, s2
@@ -2134,23 +2127,22 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1264: ; %bb.0: ; %entry
; GFX1264-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1264-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1264-NEXT: s_mov_b64 s[6:7], exec
; GFX1264-NEXT: s_mov_b64 s[4:5], exec
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1264-NEXT: s_mov_b64 s[4:5], exec
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1264-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1264-NEXT: s_cbranch_execz .LBB3_2
; GFX1264-NEXT: ; %bb.1:
+; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
; GFX1264-NEXT: s_mul_hi_u32 s7, 5, s6
; GFX1264-NEXT: s_mul_i32 s8, s6, 0
; GFX1264-NEXT: s_mul_i32 s6, s6, 5
; GFX1264-NEXT: s_add_co_u32 s7, s7, s8
; GFX1264-NEXT: v_mov_b32_e32 v0, s6
; GFX1264-NEXT: v_mov_b32_e32 v1, s7
-; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
; GFX1264-NEXT: s_mov_b32 s10, -1
; GFX1264-NEXT: s_wait_kmcnt 0x0
; GFX1264-NEXT: s_mov_b32 s8, s2
@@ -2174,21 +2166,21 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1232: ; %bb.0: ; %entry
; GFX1232-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1232-NEXT: s_mov_b32 s5, exec_lo
; GFX1232-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1232-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1232-NEXT: s_mov_b32 s4, exec_lo
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1232-NEXT: s_cbranch_execz .LBB3_2
; GFX1232-NEXT: ; %bb.1:
+; GFX1232-NEXT: s_bcnt1_i32_b32 s5, s5
+; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
; GFX1232-NEXT: s_mul_hi_u32 s7, 5, s5
; GFX1232-NEXT: s_mul_i32 s8, s5, 0
; GFX1232-NEXT: s_mul_i32 s6, s5, 5
; GFX1232-NEXT: s_add_co_u32 s7, s7, s8
; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
-; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
; GFX1232-NEXT: s_mov_b32 s10, -1
; GFX1232-NEXT: s_wait_kmcnt 0x0
; GFX1232-NEXT: s_mov_b32 s8, s2
@@ -2212,23 +2204,22 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-NEXT: s_mov_b64 s[4:5], exec
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1364-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1364-NEXT: s_cbranch_execz .LBB3_2
; GFX1364-NEXT: ; %bb.1:
+; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-NEXT: s_mul_hi_u32 s7, 5, s6
; GFX1364-NEXT: s_mul_i32 s8, s6, 0
; GFX1364-NEXT: s_mul_i32 s6, s6, 5
; GFX1364-NEXT: s_add_co_u32 s7, s7, s8
; GFX1364-NEXT: v_mov_b32_e32 v0, s6
; GFX1364-NEXT: v_mov_b32_e32 v1, s7
-; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-NEXT: s_mov_b32 s10, -1
; GFX1364-NEXT: s_wait_kmcnt 0x0
; GFX1364-NEXT: s_mov_b32 s8, s2
@@ -2255,21 +2246,21 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s5, exec_lo
; GFX1332-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1332-NEXT: s_mov_b32 s4, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB3_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s5
+; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-NEXT: s_mul_hi_u32 s7, 5, s5
; GFX1332-NEXT: s_mul_i32 s8, s5, 0
; GFX1332-NEXT: s_mul_i32 s6, s5, 5
; GFX1332-NEXT: s_add_co_u32 s7, s7, s8
; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
-; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-NEXT: s_mov_b32 s10, -1
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mov_b32 s8, s2
@@ -2394,13 +2385,13 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX9-NEXT: s_load_dwordx2 s[6:7], s[4:5], 0x34
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s8, s[4:5]
+; GFX9-NEXT: s_mov_b64 s[8:9], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX9-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX9-NEXT: s_cbranch_execz .LBB4_2
; GFX9-NEXT: ; %bb.1:
+; GFX9-NEXT: s_bcnt1_i32_b64 s8, s[8:9]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-NEXT: s_mul_i32 s12, s6, s8
; GFX9-NEXT: s_mul_hi_u32 s9, s6, s8
@@ -2436,14 +2427,15 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1064-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1064-NEXT: s_load_dwordx2 s[6:7], s[4:5], 0x34
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[4:5], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s8, s[4:5]
+; GFX1064-NEXT: s_mov_b64 s[8:9], exec
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1064-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB4_2
; GFX1064-NEXT: ; %bb.1:
+; GFX1064-NEXT: s_bcnt1_i32_b64 s8, s[8:9]
+; GFX1064-NEXT: s_mov_b32 s11, 0x31016000
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: s_mul_hi_u32 s9, s6, s8
; GFX1064-NEXT: s_mul_i32 s10, s7, s8
@@ -2451,7 +2443,6 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1064-NEXT: s_add_u32 s9, s9, s10
; GFX1064-NEXT: v_mov_b32_e32 v0, s8
; GFX1064-NEXT: v_mov_b32_e32 v1, s9
-; GFX1064-NEXT: s_mov_b32 s11, 0x31016000
; GFX1064-NEXT: s_mov_b32 s10, -1
; GFX1064-NEXT: s_mov_b32 s8, s2
; GFX1064-NEXT: s_mov_b32 s9, s3
@@ -2478,13 +2469,14 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: s_load_dwordx2 s[6:7], s[4:5], 0x34
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
+; GFX1032-NEXT: s_mov_b32 s5, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s4
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v2
; GFX1032-NEXT: s_and_saveexec_b32 s4, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB4_2
; GFX1032-NEXT: ; %bb.1:
+; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s5
+; GFX1032-NEXT: s_mov_b32 s11, 0x31016000
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_mul_hi_u32 s9, s6, s5
; GFX1032-NEXT: s_mul_i32 s10, s7, s5
@@ -2492,7 +2484,6 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1032-NEXT: s_add_u32 s9, s9, s10
; GFX1032-NEXT: v_mov_b32_e32 v0, s8
; GFX1032-NEXT: v_mov_b32_e32 v1, s9
-; GFX1032-NEXT: s_mov_b32 s11, 0x31016000
; GFX1032-NEXT: s_mov_b32 s10, -1
; GFX1032-NEXT: s_mov_b32 s8, s2
; GFX1032-NEXT: s_mov_b32 s9, s3
@@ -2519,16 +2510,16 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1164-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[8:9], exec
; GFX1164-NEXT: s_mov_b64 s[6:7], exec
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s8, s[6:7]
-; GFX1164-NEXT: s_mov_b64 s[6:7], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB4_2
; GFX1164-NEXT: ; %bb.1:
+; GFX1164-NEXT: s_bcnt1_i32_b64 s8, s[8:9]
+; GFX1164-NEXT: s_mov_b32 s11, 0x31016000
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-NEXT: s_mul_hi_u32 s9, s4, s8
; GFX1164-NEXT: s_mul_i32 s10, s5, s8
@@ -2536,7 +2527,6 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: s_add_u32 s9, s9, s10
; GFX1164-NEXT: v_mov_b32_e32 v0, s8
; GFX1164-NEXT: v_mov_b32_e32 v1, s9
-; GFX1164-NEXT: s_mov_b32 s11, 0x31016000
; GFX1164-NEXT: s_mov_b32 s10, -1
; GFX1164-NEXT: s_mov_b32 s8, s2
; GFX1164-NEXT: s_mov_b32 s9, s3
@@ -2564,14 +2554,15 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s7, exec_lo
; GFX1132-NEXT: s_mov_b32 s6, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s7, s6
-; GFX1132-NEXT: s_mov_b32 s6, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB4_2
; GFX1132-NEXT: ; %bb.1:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s7, s7
+; GFX1132-NEXT: s_mov_b32 s11, 0x31016000
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_mul_hi_u32 s9, s4, s7
; GFX1132-NEXT: s_mul_i32 s10, s5, s7
@@ -2579,7 +2570,6 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: s_add_u32 s9, s9, s10
; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: v_dual_mov_b32 v0, s8 :: v_dual_mov_b32 v1, s9
-; GFX1132-NEXT: s_mov_b32 s11, 0x31016000
; GFX1132-NEXT: s_mov_b32 s10, -1
; GFX1132-NEXT: s_mov_b32 s8, s2
; GFX1132-NEXT: s_mov_b32 s9, s3
@@ -2607,16 +2597,16 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1264-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1264-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1264-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1264-NEXT: s_mov_b64 s[8:9], exec
; GFX1264-NEXT: s_mov_b64 s[6:7], exec
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1264-NEXT: s_bcnt1_i32_b64 s8, s[6:7]
-; GFX1264-NEXT: s_mov_b64 s[6:7], exec
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1264-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1264-NEXT: s_cbranch_execz .LBB4_2
; GFX1264-NEXT: ; %bb.1:
+; GFX1264-NEXT: s_bcnt1_i32_b64 s8, s[8:9]
+; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
; GFX1264-NEXT: s_wait_kmcnt 0x0
; GFX1264-NEXT: s_mul_hi_u32 s9, s4, s8
; GFX1264-NEXT: s_mul_i32 s10, s5, s8
@@ -2624,7 +2614,6 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1264-NEXT: s_add_co_u32 s9, s9, s10
; GFX1264-NEXT: v_mov_b32_e32 v0, s8
; GFX1264-NEXT: v_mov_b32_e32 v1, s9
-; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
; GFX1264-NEXT: s_mov_b32 s10, -1
; GFX1264-NEXT: s_mov_b32 s8, s2
; GFX1264-NEXT: s_mov_b32 s9, s3
@@ -2650,14 +2639,15 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1232-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1232-NEXT: s_mov_b32 s7, exec_lo
; GFX1232-NEXT: s_mov_b32 s6, exec_lo
; GFX1232-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1232-NEXT: s_bcnt1_i32_b32 s7, s6
-; GFX1232-NEXT: s_mov_b32 s6, exec_lo
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1232-NEXT: s_cbranch_execz .LBB4_2
; GFX1232-NEXT: ; %bb.1:
+; GFX1232-NEXT: s_bcnt1_i32_b32 s7, s7
+; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
; GFX1232-NEXT: s_wait_kmcnt 0x0
; GFX1232-NEXT: s_mul_hi_u32 s9, s4, s7
; GFX1232-NEXT: s_mul_i32 s10, s5, s7
@@ -2665,7 +2655,6 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1232-NEXT: s_add_co_u32 s9, s9, s10
; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: v_dual_mov_b32 v0, s8 :: v_dual_mov_b32 v1, s9
-; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
; GFX1232-NEXT: s_mov_b32 s10, -1
; GFX1232-NEXT: s_mov_b32 s8, s2
; GFX1232-NEXT: s_mov_b32 s9, s3
@@ -2691,16 +2680,16 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-NEXT: s_load_b64 s[4:5], s[4:5], 0x34 nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[8:9], exec
; GFX1364-NEXT: s_mov_b64 s[6:7], exec
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s8, s[6:7]
-; GFX1364-NEXT: s_mov_b64 s[6:7], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1364-NEXT: s_cbranch_execz .LBB4_2
; GFX1364-NEXT: ; %bb.1:
+; GFX1364-NEXT: s_bcnt1_i32_b64 s8, s[8:9]
+; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-NEXT: s_wait_kmcnt 0x0
; GFX1364-NEXT: s_mul_hi_u32 s9, s4, s8
; GFX1364-NEXT: s_mul_i32 s10, s5, s8
@@ -2708,7 +2697,6 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: s_add_co_u32 s9, s9, s10
; GFX1364-NEXT: v_mov_b32_e32 v0, s8
; GFX1364-NEXT: v_mov_b32_e32 v1, s9
-; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-NEXT: s_mov_b32 s10, -1
; GFX1364-NEXT: s_mov_b32 s8, s2
; GFX1364-NEXT: s_mov_b32 s9, s3
@@ -2737,14 +2725,15 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-NEXT: s_load_b64 s[4:5], s[4:5], 0x34 nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s7, exec_lo
; GFX1332-NEXT: s_mov_b32 s6, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s7, s6
-; GFX1332-NEXT: s_mov_b32 s6, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB4_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s7, s7
+; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mul_hi_u32 s9, s4, s7
; GFX1332-NEXT: s_mul_i32 s10, s5, s7
@@ -2752,7 +2741,6 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: s_add_co_u32 s9, s9, s10
; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_dual_mov_b32 v0, s8 :: v_dual_mov_b32 v1, s9
-; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-NEXT: s_mov_b32 s10, -1
; GFX1332-NEXT: s_mov_b32 s8, s2
; GFX1332-NEXT: s_mov_b32 s9, s3
@@ -4240,20 +4228,20 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX7LESS-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX7LESS-NEXT: v_mbcnt_hi_u32_b32_e32 v2, exec_hi, v0
; GFX7LESS-NEXT: s_mov_b64 s[4:5], exec
-; GFX7LESS-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX7LESS-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX7LESS-NEXT: ; implicit-def: $vgpr0
; GFX7LESS-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX7LESS-NEXT: s_cbranch_execz .LBB6_4
; GFX7LESS-NEXT: ; %bb.1:
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: s_load_dword s5, s[2:3], 0x0
-; GFX7LESS-NEXT: s_mul_i32 s12, s4, 5
+; GFX7LESS-NEXT: s_load_dword s6, s[2:3], 0x0
+; GFX7LESS-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX7LESS-NEXT: s_mov_b64 s[10:11], 0
; GFX7LESS-NEXT: s_mov_b32 s7, 0xf000
-; GFX7LESS-NEXT: s_mov_b32 s6, -1
+; GFX7LESS-NEXT: s_mul_i32 s12, s4, 5
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: v_mov_b32_e32 v0, s5
+; GFX7LESS-NEXT: v_mov_b32_e32 v0, s6
+; GFX7LESS-NEXT: s_mov_b32 s6, -1
; GFX7LESS-NEXT: s_mov_b32 s4, s2
; GFX7LESS-NEXT: s_mov_b32 s5, s3
; GFX7LESS-NEXT: .LBB6_2: ; %atomicrmw.start
@@ -4290,20 +4278,20 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX8-NEXT: s_mov_b64 s[4:5], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX8-NEXT: ; implicit-def: $vgpr0
; GFX8-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX8-NEXT: s_cbranch_execz .LBB6_4
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_load_dword s5, s[2:3], 0x0
-; GFX8-NEXT: s_mul_i32 s12, s4, 5
+; GFX8-NEXT: s_load_dword s6, s[2:3], 0x0
+; GFX8-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX8-NEXT: s_mov_b64 s[10:11], 0
; GFX8-NEXT: s_mov_b32 s7, 0xf000
-; GFX8-NEXT: s_mov_b32 s6, -1
+; GFX8-NEXT: s_mul_i32 s12, s4, 5
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mov_b32_e32 v0, s5
+; GFX8-NEXT: v_mov_b32_e32 v0, s6
+; GFX8-NEXT: s_mov_b32 s6, -1
; GFX8-NEXT: s_mov_b32 s4, s2
; GFX8-NEXT: s_mov_b32 s5, s3
; GFX8-NEXT: .LBB6_2: ; %atomicrmw.start
@@ -4338,20 +4326,20 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX9-NEXT: ; implicit-def: $vgpr0
; GFX9-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX9-NEXT: s_cbranch_execz .LBB6_4
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_load_dword s5, s[2:3], 0x0
-; GFX9-NEXT: s_mul_i32 s12, s4, 5
+; GFX9-NEXT: s_load_dword s6, s[2:3], 0x0
+; GFX9-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX9-NEXT: s_mov_b64 s[10:11], 0
; GFX9-NEXT: s_mov_b32 s7, 0xf000
-; GFX9-NEXT: s_mov_b32 s6, -1
+; GFX9-NEXT: s_mul_i32 s12, s4, 5
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mov_b32_e32 v0, s5
+; GFX9-NEXT: v_mov_b32_e32 v0, s6
+; GFX9-NEXT: s_mov_b32 s6, -1
; GFX9-NEXT: s_mov_b32 s4, s2
; GFX9-NEXT: s_mov_b32 s5, s3
; GFX9-NEXT: .LBB6_2: ; %atomicrmw.start
@@ -4385,7 +4373,6 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1064-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1064-NEXT: s_mov_b64 s[4:5], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
@@ -4393,15 +4380,16 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1064-NEXT: s_cbranch_execz .LBB6_4
; GFX1064-NEXT: ; %bb.1:
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1064-NEXT: s_load_dword s5, s[2:3], 0x0
-; GFX1064-NEXT: s_mul_i32 s12, s4, 5
+; GFX1064-NEXT: s_load_dword s6, s[2:3], 0x0
+; GFX1064-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1064-NEXT: s_mov_b64 s[10:11], 0
+; GFX1064-NEXT: s_mul_i32 s12, s4, 5
; GFX1064-NEXT: s_mov_b32 s7, 0x31016000
-; GFX1064-NEXT: s_mov_b32 s6, -1
; GFX1064-NEXT: s_mov_b32 s4, s2
-; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1064-NEXT: v_mov_b32_e32 v0, s5
; GFX1064-NEXT: s_mov_b32 s5, s3
+; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
+; GFX1064-NEXT: v_mov_b32_e32 v0, s6
+; GFX1064-NEXT: s_mov_b32 s6, -1
; GFX1064-NEXT: .LBB6_2: ; %atomicrmw.start
; GFX1064-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX1064-NEXT: v_mov_b32_e32 v4, v0
@@ -4433,9 +4421,8 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
; GFX1032-NEXT: s_mov_b32 s9, 0
-; GFX1032-NEXT: s_bcnt1_i32_b32 s4, s4
+; GFX1032-NEXT: s_mov_b32 s4, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v2
; GFX1032-NEXT: s_and_saveexec_b32 s8, vcc_lo
@@ -4443,8 +4430,9 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1032-NEXT: ; %bb.1:
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_load_dword s5, s[2:3], 0x0
-; GFX1032-NEXT: s_mul_i32 s10, s4, 5
+; GFX1032-NEXT: s_bcnt1_i32_b32 s4, s4
; GFX1032-NEXT: s_mov_b32 s7, 0x31016000
+; GFX1032-NEXT: s_mul_i32 s10, s4, 5
; GFX1032-NEXT: s_mov_b32 s6, -1
; GFX1032-NEXT: s_mov_b32 s4, s2
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
@@ -4483,7 +4471,6 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1164-NEXT: s_mov_b64 s[4:5], exec
; GFX1164-NEXT: s_mov_b64 s[8:9], exec
-; GFX1164-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0
@@ -4491,15 +4478,16 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164-NEXT: s_cbranch_execz .LBB6_4
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1164-NEXT: s_load_b32 s5, s[2:3], 0x0
-; GFX1164-NEXT: s_mul_i32 s12, s4, 5
+; GFX1164-NEXT: s_load_b32 s6, s[2:3], 0x0
+; GFX1164-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1164-NEXT: s_mov_b64 s[10:11], 0
+; GFX1164-NEXT: s_mul_i32 s12, s4, 5
; GFX1164-NEXT: s_mov_b32 s7, 0x31016000
-; GFX1164-NEXT: s_mov_b32 s6, -1
; GFX1164-NEXT: s_mov_b32 s4, s2
-; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1164-NEXT: v_mov_b32_e32 v0, s5
; GFX1164-NEXT: s_mov_b32 s5, s3
+; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
+; GFX1164-NEXT: v_mov_b32_e32 v0, s6
+; GFX1164-NEXT: s_mov_b32 s6, -1
; GFX1164-NEXT: .LBB6_2: ; %atomicrmw.start
; GFX1164-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -4535,18 +4523,19 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
-; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: s_mov_b32 s9, 0
-; GFX1132-NEXT: s_bcnt1_i32_b32 s4, s4
+; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: s_mov_b32 s8, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB6_4
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_load_b32 s5, s[2:3], 0x0
-; GFX1132-NEXT: s_mul_i32 s10, s4, 5
+; GFX1132-NEXT: s_bcnt1_i32_b32 s4, s4
; GFX1132-NEXT: s_mov_b32 s7, 0x31016000
+; GFX1132-NEXT: s_mul_i32 s10, s4, 5
; GFX1132-NEXT: s_mov_b32 s6, -1
; GFX1132-NEXT: s_mov_b32 s4, s2
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
@@ -4586,20 +4575,19 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1264: ; %bb.0: ; %entry
; GFX1264-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1264-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1264-NEXT: s_mov_b64 s[6:7], exec
; GFX1264-NEXT: s_mov_b64 s[4:5], exec
; GFX1264-NEXT: ; implicit-def: $vgpr1
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1264-NEXT: s_mov_b64 s[4:5], exec
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1264-NEXT: s_cbranch_execz .LBB6_2
; GFX1264-NEXT: ; %bb.1:
-; GFX1264-NEXT: s_mul_i32 s6, s6, 5
+; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1264-NEXT: v_mov_b32_e32 v1, s6
+; GFX1264-NEXT: s_mul_i32 s6, s6, 5
; GFX1264-NEXT: s_mov_b32 s10, -1
+; GFX1264-NEXT: v_mov_b32_e32 v1, s6
; GFX1264-NEXT: s_wait_kmcnt 0x0
; GFX1264-NEXT: s_mov_b32 s8, s2
; GFX1264-NEXT: s_mov_b32 s9, s3
@@ -4622,18 +4610,18 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1232: ; %bb.0: ; %entry
; GFX1232-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1232-NEXT: s_mov_b32 s5, exec_lo
; GFX1232-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-NEXT: ; implicit-def: $vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1232-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1232-NEXT: s_mov_b32 s4, exec_lo
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1232-NEXT: s_cbranch_execz .LBB6_2
; GFX1232-NEXT: ; %bb.1:
-; GFX1232-NEXT: s_mul_i32 s5, s5, 5
+; GFX1232-NEXT: s_bcnt1_i32_b32 s5, s5
; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1232-NEXT: v_mov_b32_e32 v1, s5
+; GFX1232-NEXT: s_mul_i32 s5, s5, 5
; GFX1232-NEXT: s_mov_b32 s10, -1
+; GFX1232-NEXT: v_mov_b32_e32 v1, s5
; GFX1232-NEXT: s_wait_kmcnt 0x0
; GFX1232-NEXT: s_mov_b32 s8, s2
; GFX1232-NEXT: s_mov_b32 s9, s3
@@ -4656,20 +4644,19 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-NEXT: s_mov_b64 s[4:5], exec
; GFX1364-NEXT: ; implicit-def: $vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1364-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1364-NEXT: s_cbranch_execz .LBB6_2
; GFX1364-NEXT: ; %bb.1:
-; GFX1364-NEXT: s_mul_i32 s6, s6, 5
+; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1364-NEXT: v_mov_b32_e32 v1, s6
+; GFX1364-NEXT: s_mul_i32 s6, s6, 5
; GFX1364-NEXT: s_mov_b32 s10, -1
+; GFX1364-NEXT: v_mov_b32_e32 v1, s6
; GFX1364-NEXT: s_wait_kmcnt 0x0
; GFX1364-NEXT: s_mov_b32 s8, s2
; GFX1364-NEXT: s_mov_b32 s9, s3
@@ -4695,18 +4682,18 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s5, exec_lo
; GFX1332-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1332-NEXT: s_mov_b32 s4, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB6_2
; GFX1332-NEXT: ; %bb.1:
-; GFX1332-NEXT: s_mul_i32 s5, s5, 5
+; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s5
; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1332-NEXT: v_mov_b32_e32 v1, s5
+; GFX1332-NEXT: s_mul_i32 s5, s5, 5
; GFX1332-NEXT: s_mov_b32 s10, -1
+; GFX1332-NEXT: v_mov_b32_e32 v1, s5
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mov_b32 s8, s2
; GFX1332-NEXT: s_mov_b32 s9, s3
@@ -4741,20 +4728,20 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX7LESS-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX7LESS-NEXT: v_mbcnt_hi_u32_b32_e32 v2, exec_hi, v0
; GFX7LESS-NEXT: s_mov_b64 s[4:5], exec
-; GFX7LESS-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX7LESS-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX7LESS-NEXT: ; implicit-def: $vgpr0
; GFX7LESS-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX7LESS-NEXT: s_cbranch_execz .LBB7_4
; GFX7LESS-NEXT: ; %bb.1:
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: s_load_dword s5, s[2:3], 0x0
-; GFX7LESS-NEXT: s_mul_i32 s13, s12, s4
+; GFX7LESS-NEXT: s_load_dword s6, s[2:3], 0x0
+; GFX7LESS-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX7LESS-NEXT: s_mov_b64 s[10:11], 0
; GFX7LESS-NEXT: s_mov_b32 s7, 0xf000
-; GFX7LESS-NEXT: s_mov_b32 s6, -1
+; GFX7LESS-NEXT: s_mul_i32 s13, s12, s4
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: v_mov_b32_e32 v0, s5
+; GFX7LESS-NEXT: v_mov_b32_e32 v0, s6
+; GFX7LESS-NEXT: s_mov_b32 s6, -1
; GFX7LESS-NEXT: s_mov_b32 s4, s2
; GFX7LESS-NEXT: s_mov_b32 s5, s3
; GFX7LESS-NEXT: .LBB7_2: ; %atomicrmw.start
@@ -4791,20 +4778,20 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX8-NEXT: s_mov_b64 s[4:5], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX8-NEXT: ; implicit-def: $vgpr0
; GFX8-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX8-NEXT: s_cbranch_execz .LBB7_4
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_load_dword s5, s[2:3], 0x0
-; GFX8-NEXT: s_mul_i32 s13, s12, s4
+; GFX8-NEXT: s_load_dword s6, s[2:3], 0x0
+; GFX8-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX8-NEXT: s_mov_b64 s[10:11], 0
; GFX8-NEXT: s_mov_b32 s7, 0xf000
-; GFX8-NEXT: s_mov_b32 s6, -1
+; GFX8-NEXT: s_mul_i32 s13, s12, s4
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mov_b32_e32 v0, s5
+; GFX8-NEXT: v_mov_b32_e32 v0, s6
+; GFX8-NEXT: s_mov_b32 s6, -1
; GFX8-NEXT: s_mov_b32 s4, s2
; GFX8-NEXT: s_mov_b32 s5, s3
; GFX8-NEXT: .LBB7_2: ; %atomicrmw.start
@@ -4840,20 +4827,20 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX9-NEXT: ; implicit-def: $vgpr0
; GFX9-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX9-NEXT: s_cbranch_execz .LBB7_4
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_load_dword s5, s[2:3], 0x0
-; GFX9-NEXT: s_mul_i32 s13, s12, s4
+; GFX9-NEXT: s_load_dword s6, s[2:3], 0x0
+; GFX9-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX9-NEXT: s_mov_b64 s[10:11], 0
; GFX9-NEXT: s_mov_b32 s7, 0xf000
-; GFX9-NEXT: s_mov_b32 s6, -1
+; GFX9-NEXT: s_mul_i32 s13, s12, s4
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mov_b32_e32 v0, s5
+; GFX9-NEXT: v_mov_b32_e32 v0, s6
+; GFX9-NEXT: s_mov_b32 s6, -1
; GFX9-NEXT: s_mov_b32 s4, s2
; GFX9-NEXT: s_mov_b32 s5, s3
; GFX9-NEXT: .LBB7_2: ; %atomicrmw.start
@@ -4889,7 +4876,6 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1064-NEXT: s_load_dword s12, s[4:5], 0x34
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1064-NEXT: s_mov_b64 s[4:5], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
@@ -4897,15 +4883,16 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1064-NEXT: s_cbranch_execz .LBB7_4
; GFX1064-NEXT: ; %bb.1:
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1064-NEXT: s_load_dword s5, s[2:3], 0x0
-; GFX1064-NEXT: s_mul_i32 s13, s12, s4
+; GFX1064-NEXT: s_load_dword s6, s[2:3], 0x0
+; GFX1064-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1064-NEXT: s_mov_b64 s[10:11], 0
+; GFX1064-NEXT: s_mul_i32 s13, s12, s4
; GFX1064-NEXT: s_mov_b32 s7, 0x31016000
-; GFX1064-NEXT: s_mov_b32 s6, -1
; GFX1064-NEXT: s_mov_b32 s4, s2
-; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1064-NEXT: v_mov_b32_e32 v0, s5
; GFX1064-NEXT: s_mov_b32 s5, s3
+; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
+; GFX1064-NEXT: v_mov_b32_e32 v0, s6
+; GFX1064-NEXT: s_mov_b32 s6, -1
; GFX1064-NEXT: .LBB7_2: ; %atomicrmw.start
; GFX1064-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX1064-NEXT: v_mov_b32_e32 v4, v0
@@ -4939,9 +4926,8 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: s_load_dword s8, s[4:5], 0x34
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
; GFX1032-NEXT: s_mov_b32 s10, 0
-; GFX1032-NEXT: s_bcnt1_i32_b32 s4, s4
+; GFX1032-NEXT: s_mov_b32 s4, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v2
; GFX1032-NEXT: s_and_saveexec_b32 s9, vcc_lo
@@ -4949,8 +4935,9 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1032-NEXT: ; %bb.1:
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_load_dword s5, s[2:3], 0x0
-; GFX1032-NEXT: s_mul_i32 s11, s8, s4
+; GFX1032-NEXT: s_bcnt1_i32_b32 s4, s4
; GFX1032-NEXT: s_mov_b32 s7, 0x31016000
+; GFX1032-NEXT: s_mul_i32 s11, s8, s4
; GFX1032-NEXT: s_mov_b32 s6, -1
; GFX1032-NEXT: s_mov_b32 s4, s2
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
@@ -4991,7 +4978,6 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1164-NEXT: s_mov_b64 s[4:5], exec
; GFX1164-NEXT: s_mov_b64 s[8:9], exec
-; GFX1164-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0
@@ -4999,15 +4985,16 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: s_cbranch_execz .LBB7_4
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1164-NEXT: s_load_b32 s5, s[2:3], 0x0
-; GFX1164-NEXT: s_mul_i32 s13, s12, s4
+; GFX1164-NEXT: s_load_b32 s6, s[2:3], 0x0
+; GFX1164-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1164-NEXT: s_mov_b64 s[10:11], 0
+; GFX1164-NEXT: s_mul_i32 s13, s12, s4
; GFX1164-NEXT: s_mov_b32 s7, 0x31016000
-; GFX1164-NEXT: s_mov_b32 s6, -1
; GFX1164-NEXT: s_mov_b32 s4, s2
-; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1164-NEXT: v_mov_b32_e32 v0, s5
; GFX1164-NEXT: s_mov_b32 s5, s3
+; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
+; GFX1164-NEXT: v_mov_b32_e32 v0, s6
+; GFX1164-NEXT: s_mov_b32 s6, -1
; GFX1164-NEXT: .LBB7_2: ; %atomicrmw.start
; GFX1164-NEXT: ; =>This Inner Loop Header: Depth=1
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
@@ -5045,18 +5032,19 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
-; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: s_mov_b32 s10, 0
-; GFX1132-NEXT: s_bcnt1_i32_b32 s4, s4
+; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB7_4
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_load_b32 s5, s[2:3], 0x0
-; GFX1132-NEXT: s_mul_i32 s11, s8, s4
+; GFX1132-NEXT: s_bcnt1_i32_b32 s4, s4
; GFX1132-NEXT: s_mov_b32 s7, 0x31016000
+; GFX1132-NEXT: s_mul_i32 s11, s8, s4
; GFX1132-NEXT: s_mov_b32 s6, -1
; GFX1132-NEXT: s_mov_b32 s4, s2
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
@@ -5096,32 +5084,31 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1264: ; %bb.0: ; %entry
; GFX1264-NEXT: s_clause 0x1
; GFX1264-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX1264-NEXT: s_load_b32 s6, s[4:5], 0x34
+; GFX1264-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1264-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1264-NEXT: s_mov_b64 s[6:7], exec
; GFX1264-NEXT: s_mov_b64 s[4:5], exec
; GFX1264-NEXT: ; implicit-def: $vgpr1
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1264-NEXT: s_bcnt1_i32_b64 s7, s[4:5]
-; GFX1264-NEXT: s_mov_b64 s[4:5], exec
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1264-NEXT: s_cbranch_execz .LBB7_2
; GFX1264-NEXT: ; %bb.1:
+; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1264-NEXT: s_mov_b32 s15, 0x31016000
; GFX1264-NEXT: s_wait_kmcnt 0x0
-; GFX1264-NEXT: s_mul_i32 s7, s6, s7
-; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1264-NEXT: v_mov_b32_e32 v1, s7
-; GFX1264-NEXT: s_mov_b32 s10, -1
-; GFX1264-NEXT: s_mov_b32 s8, s2
-; GFX1264-NEXT: s_mov_b32 s9, s3
-; GFX1264-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN scope:SCOPE_DEV
+; GFX1264-NEXT: s_mul_i32 s6, s8, s6
+; GFX1264-NEXT: s_mov_b32 s14, -1
+; GFX1264-NEXT: v_mov_b32_e32 v1, s6
+; GFX1264-NEXT: s_mov_b32 s12, s2
+; GFX1264-NEXT: s_mov_b32 s13, s3
+; GFX1264-NEXT: buffer_atomic_sub_u32 v1, off, s[12:15], null th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX1264-NEXT: s_wait_loadcnt 0x0
; GFX1264-NEXT: global_inv scope:SCOPE_DEV
; GFX1264-NEXT: .LBB7_2:
; GFX1264-NEXT: s_or_b64 exec, exec, s[4:5]
; GFX1264-NEXT: s_wait_kmcnt 0x0
-; GFX1264-NEXT: v_mul_lo_u32 v0, s6, v0
+; GFX1264-NEXT: v_mul_lo_u32 v0, s8, v0
; GFX1264-NEXT: v_readfirstlane_b32 s2, v1
; GFX1264-NEXT: s_mov_b32 s3, 0x31016000
; GFX1264-NEXT: v_sub_nc_u32_e32 v0, s2, v0
@@ -5135,19 +5122,19 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1232-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-NEXT: s_load_b32 s4, s[4:5], 0x34
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1232-NEXT: s_mov_b32 s6, exec_lo
; GFX1232-NEXT: s_mov_b32 s5, exec_lo
; GFX1232-NEXT: ; implicit-def: $vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1232-NEXT: s_bcnt1_i32_b32 s6, s5
-; GFX1232-NEXT: s_mov_b32 s5, exec_lo
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1232-NEXT: s_cbranch_execz .LBB7_2
; GFX1232-NEXT: ; %bb.1:
+; GFX1232-NEXT: s_bcnt1_i32_b32 s6, s6
+; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
; GFX1232-NEXT: s_wait_kmcnt 0x0
; GFX1232-NEXT: s_mul_i32 s6, s4, s6
-; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1232-NEXT: v_mov_b32_e32 v1, s6
; GFX1232-NEXT: s_mov_b32 s10, -1
+; GFX1232-NEXT: v_mov_b32_e32 v1, s6
; GFX1232-NEXT: s_mov_b32 s8, s2
; GFX1232-NEXT: s_mov_b32 s9, s3
; GFX1232-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN scope:SCOPE_DEV
@@ -5168,35 +5155,34 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: s_clause 0x1
; GFX1364-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX1364-NEXT: s_load_b32 s6, s[4:5], 0x34 nv
+; GFX1364-NEXT: s_load_b32 s8, s[4:5], 0x34 nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-NEXT: s_mov_b64 s[4:5], exec
; GFX1364-NEXT: ; implicit-def: $vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s7, s[4:5]
-; GFX1364-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1364-NEXT: s_cbranch_execz .LBB7_2
; GFX1364-NEXT: ; %bb.1:
+; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1364-NEXT: s_mov_b32 s15, 0x31016000
; GFX1364-NEXT: s_wait_kmcnt 0x0
-; GFX1364-NEXT: s_mul_i32 s7, s6, s7
-; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1364-NEXT: v_mov_b32_e32 v1, s7
-; GFX1364-NEXT: s_mov_b32 s10, -1
-; GFX1364-NEXT: s_mov_b32 s8, s2
-; GFX1364-NEXT: s_mov_b32 s9, s3
+; GFX1364-NEXT: s_mul_i32 s6, s8, s6
+; GFX1364-NEXT: s_mov_b32 s14, -1
+; GFX1364-NEXT: v_mov_b32_e32 v1, s6
+; GFX1364-NEXT: s_mov_b32 s12, s2
+; GFX1364-NEXT: s_mov_b32 s13, s3
; GFX1364-NEXT: global_wb scope:SCOPE_DEV
; GFX1364-NEXT: s_wait_storecnt 0x0
-; GFX1364-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN scope:SCOPE_DEV
+; GFX1364-NEXT: buffer_atomic_sub_u32 v1, off, s[12:15], null th:TH_ATOMIC_RETURN scope:SCOPE_DEV
; GFX1364-NEXT: s_wait_loadcnt 0x0
; GFX1364-NEXT: global_inv scope:SCOPE_DEV
; GFX1364-NEXT: s_wait_loadcnt 0x0
; GFX1364-NEXT: .LBB7_2:
; GFX1364-NEXT: s_or_b64 exec, exec, s[4:5]
; GFX1364-NEXT: s_wait_kmcnt 0x0
-; GFX1364-NEXT: v_mul_lo_u32 v0, s6, v0
+; GFX1364-NEXT: v_mul_lo_u32 v0, s8, v0
; GFX1364-NEXT: v_readfirstlane_b32 s2, v1
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
; GFX1364-NEXT: v_sub_nc_u32_e32 v0, s2, v0
@@ -5210,19 +5196,19 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-NEXT: s_load_b32 s4, s[4:5], 0x34 nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s6, exec_lo
; GFX1332-NEXT: s_mov_b32 s5, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s6, s5
-; GFX1332-NEXT: s_mov_b32 s5, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB7_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s6, s6
+; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mul_i32 s6, s4, s6
-; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
-; GFX1332-NEXT: v_mov_b32_e32 v1, s6
; GFX1332-NEXT: s_mov_b32 s10, -1
+; GFX1332-NEXT: v_mov_b32_e32 v1, s6
; GFX1332-NEXT: s_mov_b32 s8, s2
; GFX1332-NEXT: s_mov_b32 s9, s3
; GFX1332-NEXT: global_wb scope:SCOPE_DEV
@@ -6626,23 +6612,23 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX9-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX9-NEXT: s_cbranch_execz .LBB9_4
; GFX9-NEXT: ; %bb.1:
+; GFX9-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_load_dwordx2 s[6:7], s[2:3], 0x0
-; GFX9-NEXT: s_mul_i32 s12, s4, 5
-; GFX9-NEXT: s_mul_hi_u32 s5, 5, s4
-; GFX9-NEXT: s_mul_i32 s4, s4, 0
-; GFX9-NEXT: s_add_u32 s4, s5, s4
+; GFX9-NEXT: s_load_dwordx2 s[4:5], s[2:3], 0x0
+; GFX9-NEXT: s_mul_i32 s12, s6, 5
+; GFX9-NEXT: s_mul_hi_u32 s7, 5, s6
+; GFX9-NEXT: s_mul_i32 s6, s6, 0
+; GFX9-NEXT: s_add_u32 s6, s7, s6
; GFX9-NEXT: s_mov_b64 s[10:11], 0
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mov_b32_e32 v0, s6
-; GFX9-NEXT: v_mov_b32_e32 v1, s7
-; GFX9-NEXT: v_mov_b32_e32 v5, s4
+; GFX9-NEXT: v_mov_b32_e32 v0, s4
+; GFX9-NEXT: v_mov_b32_e32 v1, s5
+; GFX9-NEXT: v_mov_b32_e32 v5, s6
; GFX9-NEXT: s_mov_b32 s7, 0xf000
; GFX9-NEXT: s_mov_b32 s6, -1
; GFX9-NEXT: s_mov_b32 s4, s2
@@ -6686,7 +6672,6 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1064-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1064-NEXT: s_mov_b64 s[4:5], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
@@ -6695,11 +6680,12 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1064-NEXT: ; %bb.1:
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: s_load_dwordx2 s[6:7], s[2:3], 0x0
-; GFX1064-NEXT: s_mul_i32 s12, s4, 5
-; GFX1064-NEXT: s_mul_hi_u32 s5, 5, s4
-; GFX1064-NEXT: s_mul_i32 s4, s4, 0
+; GFX1064-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1064-NEXT: s_mov_b64 s[10:11], 0
-; GFX1064-NEXT: s_add_u32 s13, s5, s4
+; GFX1064-NEXT: s_mul_hi_u32 s5, 5, s4
+; GFX1064-NEXT: s_mul_i32 s13, s4, 0
+; GFX1064-NEXT: s_mul_i32 s12, s4, 5
+; GFX1064-NEXT: s_add_u32 s13, s5, s13
; GFX1064-NEXT: s_mov_b32 s4, s2
; GFX1064-NEXT: s_mov_b32 s5, s3
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
@@ -6745,9 +6731,8 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
; GFX1032-NEXT: s_mov_b32 s9, 0
-; GFX1032-NEXT: s_bcnt1_i32_b32 s4, s4
+; GFX1032-NEXT: s_mov_b32 s4, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v4
; GFX1032-NEXT: s_and_saveexec_b32 s8, vcc_lo
@@ -6755,6 +6740,7 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1032-NEXT: ; %bb.1:
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_load_dwordx2 s[6:7], s[2:3], 0x0
+; GFX1032-NEXT: s_bcnt1_i32_b32 s4, s4
; GFX1032-NEXT: s_mul_hi_u32 s5, 5, s4
; GFX1032-NEXT: s_mul_i32 s11, s4, 0
; GFX1032-NEXT: s_mul_i32 s10, s4, 5
@@ -6806,7 +6792,6 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1164-NEXT: s_mov_b64 s[4:5], exec
; GFX1164-NEXT: s_mov_b64 s[8:9], exec
-; GFX1164-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
@@ -6815,11 +6800,12 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-NEXT: s_load_b64 s[6:7], s[2:3], 0x0
-; GFX1164-NEXT: s_mul_i32 s12, s4, 5
-; GFX1164-NEXT: s_mul_hi_u32 s5, 5, s4
-; GFX1164-NEXT: s_mul_i32 s4, s4, 0
+; GFX1164-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1164-NEXT: s_mov_b64 s[10:11], 0
-; GFX1164-NEXT: s_add_u32 s13, s5, s4
+; GFX1164-NEXT: s_mul_hi_u32 s5, 5, s4
+; GFX1164-NEXT: s_mul_i32 s13, s4, 0
+; GFX1164-NEXT: s_mul_i32 s12, s4, 5
+; GFX1164-NEXT: s_add_u32 s13, s5, s13
; GFX1164-NEXT: s_mov_b32 s4, s2
; GFX1164-NEXT: s_mov_b32 s5, s3
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
@@ -6872,16 +6858,18 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: s_mov_b32 s9, 0
-; GFX1132-NEXT: s_bcnt1_i32_b32 s4, s4
+; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: s_mov_b32 s8, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-NEXT: s_cbranch_execz .LBB9_4
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_load_b64 s[6:7], s[2:3], 0x0
+; GFX1132-NEXT: s_bcnt1_i32_b32 s4, s4
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: s_mul_hi_u32 s5, 5, s4
; GFX1132-NEXT: s_mul_i32 s11, s4, 0
; GFX1132-NEXT: s_mul_i32 s10, s4, 5
@@ -6932,23 +6920,22 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1264: ; %bb.0: ; %entry
; GFX1264-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1264-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1264-NEXT: s_mov_b64 s[6:7], exec
; GFX1264-NEXT: s_mov_b64 s[4:5], exec
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1264-NEXT: s_mov_b64 s[4:5], exec
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1264-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1264-NEXT: s_cbranch_execz .LBB9_2
; GFX1264-NEXT: ; %bb.1:
+; GFX1264-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
; GFX1264-NEXT: s_mul_hi_u32 s7, 5, s6
; GFX1264-NEXT: s_mul_i32 s8, s6, 0
; GFX1264-NEXT: s_mul_i32 s6, s6, 5
; GFX1264-NEXT: s_add_co_u32 s7, s7, s8
; GFX1264-NEXT: v_mov_b32_e32 v0, s6
; GFX1264-NEXT: v_mov_b32_e32 v1, s7
-; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
; GFX1264-NEXT: s_mov_b32 s10, -1
; GFX1264-NEXT: s_wait_kmcnt 0x0
; GFX1264-NEXT: s_mov_b32 s8, s2
@@ -6975,21 +6962,21 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1232: ; %bb.0: ; %entry
; GFX1232-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1232-NEXT: s_mov_b32 s5, exec_lo
; GFX1232-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1232-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1232-NEXT: s_mov_b32 s4, exec_lo
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1232-NEXT: s_cbranch_execz .LBB9_2
; GFX1232-NEXT: ; %bb.1:
+; GFX1232-NEXT: s_bcnt1_i32_b32 s5, s5
+; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
; GFX1232-NEXT: s_mul_hi_u32 s7, 5, s5
; GFX1232-NEXT: s_mul_i32 s8, s5, 0
; GFX1232-NEXT: s_mul_i32 s6, s5, 5
; GFX1232-NEXT: s_add_co_u32 s7, s7, s8
; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
-; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
; GFX1232-NEXT: s_mov_b32 s10, -1
; GFX1232-NEXT: s_wait_kmcnt 0x0
; GFX1232-NEXT: s_mov_b32 s8, s2
@@ -7016,23 +7003,22 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-NEXT: s_mov_b64 s[4:5], exec
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1364-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1364-NEXT: s_cbranch_execz .LBB9_2
; GFX1364-NEXT: ; %bb.1:
+; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-NEXT: s_mul_hi_u32 s7, 5, s6
; GFX1364-NEXT: s_mul_i32 s8, s6, 0
; GFX1364-NEXT: s_mul_i32 s6, s6, 5
; GFX1364-NEXT: s_add_co_u32 s7, s7, s8
; GFX1364-NEXT: v_mov_b32_e32 v0, s6
; GFX1364-NEXT: v_mov_b32_e32 v1, s7
-; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-NEXT: s_mov_b32 s10, -1
; GFX1364-NEXT: s_wait_kmcnt 0x0
; GFX1364-NEXT: s_mov_b32 s8, s2
@@ -7062,21 +7048,21 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out, ptr addrspace
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s5, exec_lo
; GFX1332-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1332-NEXT: s_mov_b32 s4, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB9_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s5
+; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-NEXT: s_mul_hi_u32 s7, 5, s5
; GFX1332-NEXT: s_mul_i32 s8, s5, 0
; GFX1332-NEXT: s_mul_i32 s6, s5, 5
; GFX1332-NEXT: s_add_co_u32 s7, s7, s8
; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_dual_mov_b32 v0, s6 :: v_dual_mov_b32 v1, s7
-; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-NEXT: s_mov_b32 s10, -1
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mov_b32 s8, s2
@@ -7246,23 +7232,23 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX9-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[10:11], vcc
; GFX9-NEXT: s_cbranch_execz .LBB10_4
; GFX9-NEXT: ; %bb.1:
+; GFX9-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_load_dwordx2 s[6:7], s[2:3], 0x0
-; GFX9-NEXT: s_mul_i32 s14, s8, s4
-; GFX9-NEXT: s_mul_hi_u32 s5, s8, s4
-; GFX9-NEXT: s_mul_i32 s4, s9, s4
-; GFX9-NEXT: s_add_u32 s4, s5, s4
+; GFX9-NEXT: s_load_dwordx2 s[4:5], s[2:3], 0x0
+; GFX9-NEXT: s_mul_i32 s14, s8, s6
+; GFX9-NEXT: s_mul_hi_u32 s7, s8, s6
+; GFX9-NEXT: s_mul_i32 s6, s9, s6
+; GFX9-NEXT: s_add_u32 s6, s7, s6
; GFX9-NEXT: s_mov_b64 s[12:13], 0
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mov_b32_e32 v0, s6
-; GFX9-NEXT: v_mov_b32_e32 v1, s7
-; GFX9-NEXT: v_mov_b32_e32 v5, s4
+; GFX9-NEXT: v_mov_b32_e32 v0, s4
+; GFX9-NEXT: v_mov_b32_e32 v1, s5
+; GFX9-NEXT: v_mov_b32_e32 v5, s6
; GFX9-NEXT: s_mov_b32 s7, 0xf000
; GFX9-NEXT: s_mov_b32 s6, -1
; GFX9-NEXT: s_mov_b32 s4, s2
@@ -7308,7 +7294,6 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1064-NEXT: s_load_dwordx2 s[8:9], s[4:5], 0x34
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1064-NEXT: s_mov_b64 s[4:5], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
@@ -7317,11 +7302,12 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1064-NEXT: ; %bb.1:
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: s_load_dwordx2 s[6:7], s[2:3], 0x0
-; GFX1064-NEXT: s_mul_i32 s14, s8, s4
-; GFX1064-NEXT: s_mul_hi_u32 s5, s8, s4
-; GFX1064-NEXT: s_mul_i32 s4, s9, s4
+; GFX1064-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1064-NEXT: s_mov_b64 s[12:13], 0
-; GFX1064-NEXT: s_add_u32 s15, s5, s4
+; GFX1064-NEXT: s_mul_hi_u32 s5, s8, s4
+; GFX1064-NEXT: s_mul_i32 s15, s9, s4
+; GFX1064-NEXT: s_mul_i32 s14, s8, s4
+; GFX1064-NEXT: s_add_u32 s15, s5, s15
; GFX1064-NEXT: s_mov_b32 s4, s2
; GFX1064-NEXT: s_mov_b32 s5, s3
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
@@ -7369,9 +7355,8 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: s_load_dwordx2 s[8:9], s[4:5], 0x34
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
; GFX1032-NEXT: s_mov_b32 s11, 0
-; GFX1032-NEXT: s_bcnt1_i32_b32 s4, s4
+; GFX1032-NEXT: s_mov_b32 s4, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v4
; GFX1032-NEXT: s_and_saveexec_b32 s10, vcc_lo
@@ -7379,6 +7364,7 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1032-NEXT: ; %bb.1:
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_load_dwordx2 s[6:7], s[2:3], 0x0
+; GFX1032-NEXT: s_bcnt1_i32_b32 s4, s4
; GFX1032-NEXT: s_mul_hi_u32 s5, s8, s4
; GFX1032-NEXT: s_mul_i32 s13, s9, s4
; GFX1032-NEXT: s_mul_i32 s12, s8, s4
@@ -7432,7 +7418,6 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX1164-NEXT: s_mov_b64 s[4:5], exec
; GFX1164-NEXT: s_mov_b64 s[10:11], exec
-; GFX1164-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
@@ -7441,11 +7426,12 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-NEXT: s_load_b64 s[6:7], s[2:3], 0x0
-; GFX1164-NEXT: s_mul_i32 s14, s8, s4
-; GFX1164-NEXT: s_mul_hi_u32 s5, s8, s4
-; GFX1164-NEXT: s_mul_i32 s4, s9, s4
+; GFX1164-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX1164-NEXT: s_mov_b64 s[12:13], 0
-; GFX1164-NEXT: s_add_u32 s15, s5, s4
+; GFX1164-NEXT: s_mul_hi_u32 s5, s8, s4
+; GFX1164-NEXT: s_mul_i32 s15, s9, s4
+; GFX1164-NEXT: s_mul_i32 s14, s8, s4
+; GFX1164-NEXT: s_add_u32 s15, s5, s15
; GFX1164-NEXT: s_mov_b32 s4, s2
; GFX1164-NEXT: s_mov_b32 s5, s3
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
@@ -7500,16 +7486,18 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1132-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-NEXT: s_load_b64 s[8:9], s[4:5], 0x34
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: s_mov_b32 s11, 0
-; GFX1132-NEXT: s_bcnt1_i32_b32 s4, s4
+; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: s_mov_b32 s10, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-NEXT: s_cbranch_execz .LBB10_4
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_load_b64 s[6:7], s[2:3], 0x0
+; GFX1132-NEXT: s_bcnt1_i32_b32 s4, s4
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: s_mul_hi_u32 s5, s8, s4
; GFX1132-NEXT: s_mul_i32 s13, s9, s4
; GFX1132-NEXT: s_mul_i32 s12, s8, s4
@@ -7562,16 +7550,16 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1264-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1264-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1264-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1264-NEXT: s_mov_b64 s[8:9], exec
; GFX1264-NEXT: s_mov_b64 s[6:7], exec
-; GFX1264-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1264-NEXT: s_bcnt1_i32_b64 s8, s[6:7]
-; GFX1264-NEXT: s_mov_b64 s[6:7], exec
+; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1264-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1264-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1264-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1264-NEXT: s_cbranch_execz .LBB10_2
; GFX1264-NEXT: ; %bb.1:
+; GFX1264-NEXT: s_bcnt1_i32_b64 s8, s[8:9]
+; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
; GFX1264-NEXT: s_wait_kmcnt 0x0
; GFX1264-NEXT: s_mul_hi_u32 s9, s4, s8
; GFX1264-NEXT: s_mul_i32 s10, s5, s8
@@ -7579,7 +7567,6 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1264-NEXT: s_add_co_u32 s9, s9, s10
; GFX1264-NEXT: v_mov_b32_e32 v0, s8
; GFX1264-NEXT: v_mov_b32_e32 v1, s9
-; GFX1264-NEXT: s_mov_b32 s11, 0x31016000
; GFX1264-NEXT: s_mov_b32 s10, -1
; GFX1264-NEXT: s_mov_b32 s8, s2
; GFX1264-NEXT: s_mov_b32 s9, s3
@@ -7607,14 +7594,15 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1232-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-NEXT: s_load_b64 s[4:5], s[4:5], 0x34
; GFX1232-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1232-NEXT: s_mov_b32 s7, exec_lo
; GFX1232-NEXT: s_mov_b32 s6, exec_lo
; GFX1232-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1232-NEXT: s_bcnt1_i32_b32 s7, s6
-; GFX1232-NEXT: s_mov_b32 s6, exec_lo
+; GFX1232-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1232-NEXT: s_cbranch_execz .LBB10_2
; GFX1232-NEXT: ; %bb.1:
+; GFX1232-NEXT: s_bcnt1_i32_b32 s7, s7
+; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
; GFX1232-NEXT: s_wait_kmcnt 0x0
; GFX1232-NEXT: s_mul_hi_u32 s9, s4, s7
; GFX1232-NEXT: s_mul_i32 s10, s5, s7
@@ -7622,7 +7610,6 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1232-NEXT: s_add_co_u32 s9, s9, s10
; GFX1232-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1232-NEXT: v_dual_mov_b32 v0, s8 :: v_dual_mov_b32 v1, s9
-; GFX1232-NEXT: s_mov_b32 s11, 0x31016000
; GFX1232-NEXT: s_mov_b32 s10, -1
; GFX1232-NEXT: s_mov_b32 s8, s2
; GFX1232-NEXT: s_mov_b32 s9, s3
@@ -7650,16 +7637,16 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-NEXT: s_load_b64 s[4:5], s[4:5], 0x34 nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[8:9], exec
; GFX1364-NEXT: s_mov_b64 s[6:7], exec
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s8, s[6:7]
-; GFX1364-NEXT: s_mov_b64 s[6:7], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1364-NEXT: s_cbranch_execz .LBB10_2
; GFX1364-NEXT: ; %bb.1:
+; GFX1364-NEXT: s_bcnt1_i32_b64 s8, s[8:9]
+; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-NEXT: s_wait_kmcnt 0x0
; GFX1364-NEXT: s_mul_hi_u32 s9, s4, s8
; GFX1364-NEXT: s_mul_i32 s10, s5, s8
@@ -7667,7 +7654,6 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1364-NEXT: s_add_co_u32 s9, s9, s10
; GFX1364-NEXT: v_mov_b32_e32 v0, s8
; GFX1364-NEXT: v_mov_b32_e32 v1, s9
-; GFX1364-NEXT: s_mov_b32 s11, 0x31016000
; GFX1364-NEXT: s_mov_b32 s10, -1
; GFX1364-NEXT: s_mov_b32 s8, s2
; GFX1364-NEXT: s_mov_b32 s9, s3
@@ -7698,14 +7684,15 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-NEXT: s_load_b64 s[4:5], s[4:5], 0x34 nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s7, exec_lo
; GFX1332-NEXT: s_mov_b32 s6, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s7, s6
-; GFX1332-NEXT: s_mov_b32 s6, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB10_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s7, s7
+; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mul_hi_u32 s9, s4, s7
; GFX1332-NEXT: s_mul_i32 s10, s5, s7
@@ -7713,7 +7700,6 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX1332-NEXT: s_add_co_u32 s9, s9, s10
; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_dual_mov_b32 v0, s8 :: v_dual_mov_b32 v1, s9
-; GFX1332-NEXT: s_mov_b32 s11, 0x31016000
; GFX1332-NEXT: s_mov_b32 s10, -1
; GFX1332-NEXT: s_mov_b32 s8, s2
; GFX1332-NEXT: s_mov_b32 s9, s3
@@ -10402,15 +10388,15 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX7LESS-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX7LESS-NEXT: v_mbcnt_hi_u32_b32_e32 v4, exec_hi, v0
; GFX7LESS-NEXT: s_mov_b64 s[4:5], exec
-; GFX7LESS-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX7LESS-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX7LESS-NEXT: ; implicit-def: $vgpr0
; GFX7LESS-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX7LESS-NEXT: s_cbranch_execz .LBB13_4
; GFX7LESS-NEXT: ; %bb.1:
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: s_sext_i32_i8 s5, s12
-; GFX7LESS-NEXT: s_mul_i32 s6, s5, s4
+; GFX7LESS-NEXT: s_sext_i32_i8 s6, s12
+; GFX7LESS-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
+; GFX7LESS-NEXT: s_mul_i32 s6, s6, s4
; GFX7LESS-NEXT: s_and_b32 s4, s2, -4
; GFX7LESS-NEXT: s_mov_b32 s5, s3
; GFX7LESS-NEXT: s_load_dword s5, s[4:5], 0x0
@@ -10464,15 +10450,15 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX8-NEXT: s_mov_b64 s[4:5], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX8-NEXT: ; implicit-def: $vgpr0
; GFX8-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX8-NEXT: s_cbranch_execz .LBB13_4
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_sext_i32_i8 s5, s12
-; GFX8-NEXT: s_mul_i32 s6, s5, s4
+; GFX8-NEXT: s_sext_i32_i8 s6, s12
+; GFX8-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
+; GFX8-NEXT: s_mul_i32 s6, s6, s4
; GFX8-NEXT: s_and_b32 s4, s2, -4
; GFX8-NEXT: s_mov_b32 s5, s3
; GFX8-NEXT: s_load_dword s5, s[4:5], 0x0
@@ -10524,15 +10510,15 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX9-NEXT: ; implicit-def: $vgpr0
; GFX9-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX9-NEXT: s_cbranch_execz .LBB13_4
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_sext_i32_i8 s5, s12
-; GFX9-NEXT: s_mul_i32 s6, s5, s4
+; GFX9-NEXT: s_sext_i32_i8 s6, s12
+; GFX9-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
+; GFX9-NEXT: s_mul_i32 s6, s6, s4
; GFX9-NEXT: s_and_b32 s4, s2, -4
; GFX9-NEXT: s_mov_b32 s5, s3
; GFX9-NEXT: s_load_dword s5, s[4:5], 0x0
@@ -10582,8 +10568,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1064-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1064-NEXT: s_load_dword s12, s[4:5], 0x34
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[4:5], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GFX1064-NEXT: s_mov_b64 s[6:7], exec
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
@@ -10593,13 +10578,14 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: s_and_b32 s4, s2, -4
; GFX1064-NEXT: s_mov_b32 s5, s3
-; GFX1064-NEXT: s_and_b32 s2, s2, 3
+; GFX1064-NEXT: s_sext_i32_i8 s10, s12
; GFX1064-NEXT: s_load_dword s5, s[4:5], 0x0
-; GFX1064-NEXT: s_sext_i32_i8 s7, s12
+; GFX1064-NEXT: s_and_b32 s2, s2, 3
+; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1064-NEXT: s_lshl_b32 s2, s2, 3
-; GFX1064-NEXT: s_mul_i32 s7, s7, s6
+; GFX1064-NEXT: s_mul_i32 s10, s10, s6
; GFX1064-NEXT: s_lshl_b32 s13, 0xff, s2
-; GFX1064-NEXT: s_and_b32 s6, s7, 0xff
+; GFX1064-NEXT: s_and_b32 s6, s10, 0xff
; GFX1064-NEXT: s_not_b32 s14, s13
; GFX1064-NEXT: s_lshl_b32 s15, s6, s2
; GFX1064-NEXT: s_mov_b64 s[10:11], 0
@@ -10641,9 +10627,8 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: s_load_dword s8, s[4:5], 0x34
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
; GFX1032-NEXT: s_mov_b32 s10, 0
-; GFX1032-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1032-NEXT: s_mov_b32 s6, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v4
; GFX1032-NEXT: s_and_saveexec_b32 s9, vcc_lo
@@ -10652,9 +10637,10 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_and_b32 s4, s2, -4
; GFX1032-NEXT: s_mov_b32 s5, s3
-; GFX1032-NEXT: s_and_b32 s2, s2, 3
-; GFX1032-NEXT: s_load_dword s5, s[4:5], 0x0
; GFX1032-NEXT: s_sext_i32_i8 s7, s8
+; GFX1032-NEXT: s_load_dword s5, s[4:5], 0x0
+; GFX1032-NEXT: s_and_b32 s2, s2, 3
+; GFX1032-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1032-NEXT: s_lshl_b32 s2, s2, 3
; GFX1032-NEXT: s_mul_i32 s7, s7, s6
; GFX1032-NEXT: s_lshl_b32 s11, 0xff, s2
@@ -10699,9 +10685,8 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1164-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1164-TRUE16-NEXT: s_load_b32 s12, s[4:5], 0x34
; GFX1164-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1164-TRUE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1164-TRUE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1164-TRUE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1164-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1164-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
@@ -10711,13 +10696,14 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1164-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-TRUE16-NEXT: s_and_b32 s4, s2, -4
; GFX1164-TRUE16-NEXT: s_mov_b32 s5, s3
-; GFX1164-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1164-TRUE16-NEXT: s_sext_i32_i8 s10, s12
; GFX1164-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
-; GFX1164-TRUE16-NEXT: s_sext_i32_i8 s7, s12
+; GFX1164-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1164-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1164-TRUE16-NEXT: s_lshl_b32 s2, s2, 3
-; GFX1164-TRUE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1164-TRUE16-NEXT: s_mul_i32 s10, s10, s6
; GFX1164-TRUE16-NEXT: s_lshl_b32 s13, 0xff, s2
-; GFX1164-TRUE16-NEXT: s_and_b32 s6, s7, 0xff
+; GFX1164-TRUE16-NEXT: s_and_b32 s6, s10, 0xff
; GFX1164-TRUE16-NEXT: s_not_b32 s14, s13
; GFX1164-TRUE16-NEXT: s_lshl_b32 s15, s6, s2
; GFX1164-TRUE16-NEXT: s_mov_b64 s[10:11], 0
@@ -10764,9 +10750,8 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1164-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1164-FAKE16-NEXT: s_load_b32 s12, s[4:5], 0x34
; GFX1164-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1164-FAKE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1164-FAKE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1164-FAKE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1164-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1164-FAKE16-NEXT: ; implicit-def: $vgpr0
@@ -10776,13 +10761,14 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1164-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-FAKE16-NEXT: s_and_b32 s4, s2, -4
; GFX1164-FAKE16-NEXT: s_mov_b32 s5, s3
-; GFX1164-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1164-FAKE16-NEXT: s_sext_i32_i8 s10, s12
; GFX1164-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
-; GFX1164-FAKE16-NEXT: s_sext_i32_i8 s7, s12
+; GFX1164-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1164-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1164-FAKE16-NEXT: s_lshl_b32 s2, s2, 3
-; GFX1164-FAKE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1164-FAKE16-NEXT: s_mul_i32 s10, s10, s6
; GFX1164-FAKE16-NEXT: s_lshl_b32 s13, 0xff, s2
-; GFX1164-FAKE16-NEXT: s_and_b32 s6, s7, 0xff
+; GFX1164-FAKE16-NEXT: s_and_b32 s6, s10, 0xff
; GFX1164-FAKE16-NEXT: s_not_b32 s14, s13
; GFX1164-FAKE16-NEXT: s_lshl_b32 s15, s6, s2
; GFX1164-FAKE16-NEXT: s_mov_b64 s[10:11], 0
@@ -10829,20 +10815,21 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1132-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-TRUE16-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1132-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1132-TRUE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-TRUE16-NEXT: s_mov_b32 s10, 0
-; GFX1132-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1132-TRUE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1132-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-TRUE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1132-TRUE16-NEXT: ; %bb.1:
; GFX1132-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-TRUE16-NEXT: s_and_b32 s4, s2, -4
; GFX1132-TRUE16-NEXT: s_mov_b32 s5, s3
-; GFX1132-TRUE16-NEXT: s_and_b32 s2, s2, 3
-; GFX1132-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
; GFX1132-TRUE16-NEXT: s_sext_i32_i8 s7, s8
+; GFX1132-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
+; GFX1132-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1132-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1132-TRUE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1132-TRUE16-NEXT: s_mul_i32 s7, s7, s6
; GFX1132-TRUE16-NEXT: s_lshl_b32 s11, 0xff, s2
@@ -10890,20 +10877,21 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1132-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-FAKE16-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1132-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1132-FAKE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-FAKE16-NEXT: s_mov_b32 s10, 0
-; GFX1132-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1132-FAKE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1132-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-FAKE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1132-FAKE16-NEXT: ; %bb.1:
; GFX1132-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-FAKE16-NEXT: s_and_b32 s4, s2, -4
; GFX1132-FAKE16-NEXT: s_mov_b32 s5, s3
-; GFX1132-FAKE16-NEXT: s_and_b32 s2, s2, 3
-; GFX1132-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
; GFX1132-FAKE16-NEXT: s_sext_i32_i8 s7, s8
+; GFX1132-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
+; GFX1132-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1132-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1132-FAKE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1132-FAKE16-NEXT: s_mul_i32 s7, s7, s6
; GFX1132-FAKE16-NEXT: s_lshl_b32 s11, 0xff, s2
@@ -10951,9 +10939,8 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1264-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1264-TRUE16-NEXT: s_load_b32 s12, s[4:5], 0x34
; GFX1264-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1264-TRUE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1264-TRUE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1264-TRUE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1264-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1264-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
@@ -10963,13 +10950,14 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1264-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1264-TRUE16-NEXT: s_and_b32 s4, s2, -4
; GFX1264-TRUE16-NEXT: s_mov_b32 s5, s3
-; GFX1264-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1264-TRUE16-NEXT: s_sext_i32_i8 s10, s12
; GFX1264-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
-; GFX1264-TRUE16-NEXT: s_sext_i32_i8 s7, s12
+; GFX1264-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1264-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1264-TRUE16-NEXT: s_lshl_b32 s2, s2, 3
-; GFX1264-TRUE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1264-TRUE16-NEXT: s_mul_i32 s10, s10, s6
; GFX1264-TRUE16-NEXT: s_lshl_b32 s13, 0xff, s2
-; GFX1264-TRUE16-NEXT: s_and_b32 s6, s7, 0xff
+; GFX1264-TRUE16-NEXT: s_and_b32 s6, s10, 0xff
; GFX1264-TRUE16-NEXT: s_not_b32 s14, s13
; GFX1264-TRUE16-NEXT: s_lshl_b32 s15, s6, s2
; GFX1264-TRUE16-NEXT: s_mov_b64 s[10:11], 0
@@ -11016,9 +11004,8 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1264-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1264-FAKE16-NEXT: s_load_b32 s12, s[4:5], 0x34
; GFX1264-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1264-FAKE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1264-FAKE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1264-FAKE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1264-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1264-FAKE16-NEXT: ; implicit-def: $vgpr0
@@ -11028,13 +11015,14 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1264-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1264-FAKE16-NEXT: s_and_b32 s4, s2, -4
; GFX1264-FAKE16-NEXT: s_mov_b32 s5, s3
-; GFX1264-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1264-FAKE16-NEXT: s_sext_i32_i8 s10, s12
; GFX1264-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
-; GFX1264-FAKE16-NEXT: s_sext_i32_i8 s7, s12
+; GFX1264-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1264-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1264-FAKE16-NEXT: s_lshl_b32 s2, s2, 3
-; GFX1264-FAKE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1264-FAKE16-NEXT: s_mul_i32 s10, s10, s6
; GFX1264-FAKE16-NEXT: s_lshl_b32 s13, 0xff, s2
-; GFX1264-FAKE16-NEXT: s_and_b32 s6, s7, 0xff
+; GFX1264-FAKE16-NEXT: s_and_b32 s6, s10, 0xff
; GFX1264-FAKE16-NEXT: s_not_b32 s14, s13
; GFX1264-FAKE16-NEXT: s_lshl_b32 s15, s6, s2
; GFX1264-FAKE16-NEXT: s_mov_b64 s[10:11], 0
@@ -11081,20 +11069,21 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1232-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-TRUE16-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1232-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1232-TRUE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-TRUE16-NEXT: s_mov_b32 s10, 0
-; GFX1232-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1232-TRUE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1232-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1232-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1232-TRUE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1232-TRUE16-NEXT: ; %bb.1:
; GFX1232-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1232-TRUE16-NEXT: s_and_b32 s4, s2, -4
; GFX1232-TRUE16-NEXT: s_mov_b32 s5, s3
-; GFX1232-TRUE16-NEXT: s_and_b32 s2, s2, 3
-; GFX1232-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
; GFX1232-TRUE16-NEXT: s_sext_i32_i8 s7, s8
+; GFX1232-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
+; GFX1232-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1232-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1232-TRUE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1232-TRUE16-NEXT: s_mul_i32 s7, s7, s6
; GFX1232-TRUE16-NEXT: s_lshl_b32 s11, 0xff, s2
@@ -11143,20 +11132,21 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1232-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-FAKE16-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1232-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1232-FAKE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-FAKE16-NEXT: s_mov_b32 s10, 0
-; GFX1232-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1232-FAKE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1232-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1232-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1232-FAKE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1232-FAKE16-NEXT: ; %bb.1:
; GFX1232-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1232-FAKE16-NEXT: s_and_b32 s4, s2, -4
; GFX1232-FAKE16-NEXT: s_mov_b32 s5, s3
-; GFX1232-FAKE16-NEXT: s_and_b32 s2, s2, 3
-; GFX1232-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
; GFX1232-FAKE16-NEXT: s_sext_i32_i8 s7, s8
+; GFX1232-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
+; GFX1232-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1232-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1232-FAKE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1232-FAKE16-NEXT: s_mul_i32 s7, s7, s6
; GFX1232-FAKE16-NEXT: s_lshl_b32 s11, 0xff, s2
@@ -11205,9 +11195,8 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1364-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-TRUE16-NEXT: s_load_b32 s10, s[4:5], 0x34 nv
; GFX1364-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1364-TRUE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-TRUE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-TRUE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1364-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1364-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
@@ -11216,13 +11205,14 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1364-TRUE16-NEXT: ; %bb.1:
; GFX1364-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1364-TRUE16-NEXT: s_and_b64 s[4:5], s[2:3], -4
-; GFX1364-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1364-TRUE16-NEXT: s_sext_i32_i8 s12, s10
; GFX1364-TRUE16-NEXT: s_load_b32 s3, s[4:5], 0x0
-; GFX1364-TRUE16-NEXT: s_sext_i32_i8 s7, s10
+; GFX1364-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1364-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1364-TRUE16-NEXT: s_lshl_b32 s11, s2, 3
-; GFX1364-TRUE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1364-TRUE16-NEXT: s_mul_i32 s2, s12, s6
; GFX1364-TRUE16-NEXT: s_lshl_b32 s12, 0xff, s11
-; GFX1364-TRUE16-NEXT: s_and_b32 s2, s7, 0xff
+; GFX1364-TRUE16-NEXT: s_and_b32 s2, s2, 0xff
; GFX1364-TRUE16-NEXT: s_not_b32 s13, s12
; GFX1364-TRUE16-NEXT: s_lshl_b32 s14, s2, s11
; GFX1364-TRUE16-NEXT: s_mov_b32 s7, 0x31016000
@@ -11267,9 +11257,8 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1364-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-FAKE16-NEXT: s_load_b32 s10, s[4:5], 0x34 nv
; GFX1364-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1364-FAKE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-FAKE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-FAKE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1364-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1364-FAKE16-NEXT: ; implicit-def: $vgpr0
@@ -11278,13 +11267,14 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1364-FAKE16-NEXT: ; %bb.1:
; GFX1364-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1364-FAKE16-NEXT: s_and_b64 s[4:5], s[2:3], -4
-; GFX1364-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1364-FAKE16-NEXT: s_sext_i32_i8 s12, s10
; GFX1364-FAKE16-NEXT: s_load_b32 s3, s[4:5], 0x0
-; GFX1364-FAKE16-NEXT: s_sext_i32_i8 s7, s10
+; GFX1364-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1364-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1364-FAKE16-NEXT: s_lshl_b32 s11, s2, 3
-; GFX1364-FAKE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1364-FAKE16-NEXT: s_mul_i32 s2, s12, s6
; GFX1364-FAKE16-NEXT: s_lshl_b32 s12, 0xff, s11
-; GFX1364-FAKE16-NEXT: s_and_b32 s2, s7, 0xff
+; GFX1364-FAKE16-NEXT: s_and_b32 s2, s2, 0xff
; GFX1364-FAKE16-NEXT: s_not_b32 s13, s12
; GFX1364-FAKE16-NEXT: s_lshl_b32 s14, s2, s11
; GFX1364-FAKE16-NEXT: s_mov_b32 s7, 0x31016000
@@ -11329,11 +11319,11 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1332-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-TRUE16-NEXT: s_load_b32 s8, s[4:5], 0x34 nv
; GFX1332-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1332-TRUE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-TRUE16-NEXT: s_mov_b32 s10, 0
-; GFX1332-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1332-TRUE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1332-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1332-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1332-TRUE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1332-TRUE16-NEXT: ; %bb.1:
@@ -11342,6 +11332,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1332-TRUE16-NEXT: s_and_b32 s2, s2, 3
; GFX1332-TRUE16-NEXT: s_load_b32 s7, s[4:5], 0x0
; GFX1332-TRUE16-NEXT: s_sext_i32_i8 s11, s8
+; GFX1332-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1332-TRUE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1332-TRUE16-NEXT: s_mul_i32 s6, s11, s6
; GFX1332-TRUE16-NEXT: s_lshl_b32 s3, 0xff, s2
@@ -11388,11 +11379,11 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1332-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-FAKE16-NEXT: s_load_b32 s8, s[4:5], 0x34 nv
; GFX1332-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1332-FAKE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-FAKE16-NEXT: s_mov_b32 s10, 0
-; GFX1332-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1332-FAKE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1332-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1332-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1332-FAKE16-NEXT: s_cbranch_execz .LBB13_4
; GFX1332-FAKE16-NEXT: ; %bb.1:
@@ -11401,6 +11392,7 @@ define amdgpu_kernel void @uniform_add_i8(ptr addrspace(1) %result, ptr addrspac
; GFX1332-FAKE16-NEXT: s_and_b32 s2, s2, 3
; GFX1332-FAKE16-NEXT: s_load_b32 s7, s[4:5], 0x0
; GFX1332-FAKE16-NEXT: s_sext_i32_i8 s11, s8
+; GFX1332-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1332-FAKE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1332-FAKE16-NEXT: s_mul_i32 s6, s11, s6
; GFX1332-FAKE16-NEXT: s_lshl_b32 s3, 0xff, s2
@@ -12840,21 +12832,21 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX7LESS-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX7LESS-NEXT: v_mbcnt_hi_u32_b32_e32 v4, exec_hi, v0
; GFX7LESS-NEXT: s_mov_b64 s[4:5], exec
-; GFX7LESS-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX7LESS-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX7LESS-NEXT: ; implicit-def: $vgpr0
; GFX7LESS-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX7LESS-NEXT: s_cbranch_execz .LBB16_4
; GFX7LESS-NEXT: ; %bb.1:
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: s_sext_i32_i16 s5, s12
-; GFX7LESS-NEXT: s_mul_i32 s5, s5, s4
-; GFX7LESS-NEXT: s_and_b32 s6, s5, 0xffff
+; GFX7LESS-NEXT: s_sext_i32_i16 s6, s12
+; GFX7LESS-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
+; GFX7LESS-NEXT: s_mul_i32 s6, s6, s4
; GFX7LESS-NEXT: s_and_b32 s4, s2, -4
; GFX7LESS-NEXT: s_mov_b32 s5, s3
; GFX7LESS-NEXT: s_load_dword s5, s[4:5], 0x0
; GFX7LESS-NEXT: s_and_b32 s2, s2, 3
; GFX7LESS-NEXT: s_lshl_b32 s2, s2, 3
+; GFX7LESS-NEXT: s_and_b32 s6, s6, 0xffff
; GFX7LESS-NEXT: s_lshl_b32 s13, 0xffff, s2
; GFX7LESS-NEXT: s_not_b32 s14, s13
; GFX7LESS-NEXT: s_lshl_b32 s15, s6, s2
@@ -12902,15 +12894,15 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX8-NEXT: s_mov_b64 s[4:5], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX8-NEXT: ; implicit-def: $vgpr0
; GFX8-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX8-NEXT: s_cbranch_execz .LBB16_4
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_sext_i32_i16 s5, s12
-; GFX8-NEXT: s_mul_i32 s6, s5, s4
+; GFX8-NEXT: s_sext_i32_i16 s6, s12
+; GFX8-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
+; GFX8-NEXT: s_mul_i32 s6, s6, s4
; GFX8-NEXT: s_and_b32 s4, s2, -4
; GFX8-NEXT: s_mov_b32 s5, s3
; GFX8-NEXT: s_load_dword s5, s[4:5], 0x0
@@ -12962,15 +12954,15 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
; GFX9-NEXT: ; implicit-def: $vgpr0
; GFX9-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GFX9-NEXT: s_cbranch_execz .LBB16_4
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_sext_i32_i16 s5, s12
-; GFX9-NEXT: s_mul_i32 s6, s5, s4
+; GFX9-NEXT: s_sext_i32_i16 s6, s12
+; GFX9-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
+; GFX9-NEXT: s_mul_i32 s6, s6, s4
; GFX9-NEXT: s_and_b32 s4, s2, -4
; GFX9-NEXT: s_mov_b32 s5, s3
; GFX9-NEXT: s_load_dword s5, s[4:5], 0x0
@@ -13020,8 +13012,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1064-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1064-NEXT: s_load_dword s12, s[4:5], 0x34
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[4:5], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GFX1064-NEXT: s_mov_b64 s[6:7], exec
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v4
@@ -13031,13 +13022,14 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: s_and_b32 s4, s2, -4
; GFX1064-NEXT: s_mov_b32 s5, s3
-; GFX1064-NEXT: s_and_b32 s2, s2, 3
+; GFX1064-NEXT: s_sext_i32_i16 s10, s12
; GFX1064-NEXT: s_load_dword s5, s[4:5], 0x0
-; GFX1064-NEXT: s_sext_i32_i16 s7, s12
+; GFX1064-NEXT: s_and_b32 s2, s2, 3
+; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1064-NEXT: s_lshl_b32 s2, s2, 3
-; GFX1064-NEXT: s_mul_i32 s7, s7, s6
+; GFX1064-NEXT: s_mul_i32 s10, s10, s6
; GFX1064-NEXT: s_lshl_b32 s13, 0xffff, s2
-; GFX1064-NEXT: s_and_b32 s6, 0xffff, s7
+; GFX1064-NEXT: s_and_b32 s6, 0xffff, s10
; GFX1064-NEXT: s_not_b32 s14, s13
; GFX1064-NEXT: s_lshl_b32 s15, s6, s2
; GFX1064-NEXT: s_mov_b64 s[10:11], 0
@@ -13079,9 +13071,8 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: s_load_dword s8, s[4:5], 0x34
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
; GFX1032-NEXT: s_mov_b32 s10, 0
-; GFX1032-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1032-NEXT: s_mov_b32 s6, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v4
; GFX1032-NEXT: s_and_saveexec_b32 s9, vcc_lo
@@ -13090,9 +13081,10 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_and_b32 s4, s2, -4
; GFX1032-NEXT: s_mov_b32 s5, s3
-; GFX1032-NEXT: s_and_b32 s2, s2, 3
-; GFX1032-NEXT: s_load_dword s5, s[4:5], 0x0
; GFX1032-NEXT: s_sext_i32_i16 s7, s8
+; GFX1032-NEXT: s_load_dword s5, s[4:5], 0x0
+; GFX1032-NEXT: s_and_b32 s2, s2, 3
+; GFX1032-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1032-NEXT: s_lshl_b32 s2, s2, 3
; GFX1032-NEXT: s_mul_i32 s7, s7, s6
; GFX1032-NEXT: s_lshl_b32 s11, 0xffff, s2
@@ -13137,9 +13129,8 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1164-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1164-TRUE16-NEXT: s_load_b32 s12, s[4:5], 0x34
; GFX1164-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1164-TRUE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1164-TRUE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1164-TRUE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1164-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1164-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1164-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
@@ -13149,13 +13140,14 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1164-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-TRUE16-NEXT: s_and_b32 s4, s2, -4
; GFX1164-TRUE16-NEXT: s_mov_b32 s5, s3
-; GFX1164-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1164-TRUE16-NEXT: s_sext_i32_i16 s10, s12
; GFX1164-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
-; GFX1164-TRUE16-NEXT: s_sext_i32_i16 s7, s12
+; GFX1164-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1164-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1164-TRUE16-NEXT: s_lshl_b32 s2, s2, 3
-; GFX1164-TRUE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1164-TRUE16-NEXT: s_mul_i32 s10, s10, s6
; GFX1164-TRUE16-NEXT: s_lshl_b32 s13, 0xffff, s2
-; GFX1164-TRUE16-NEXT: s_and_b32 s6, 0xffff, s7
+; GFX1164-TRUE16-NEXT: s_and_b32 s6, 0xffff, s10
; GFX1164-TRUE16-NEXT: s_not_b32 s14, s13
; GFX1164-TRUE16-NEXT: s_lshl_b32 s15, s6, s2
; GFX1164-TRUE16-NEXT: s_mov_b64 s[10:11], 0
@@ -13202,9 +13194,8 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1164-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1164-FAKE16-NEXT: s_load_b32 s12, s[4:5], 0x34
; GFX1164-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1164-FAKE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1164-FAKE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1164-FAKE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1164-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1164-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1164-FAKE16-NEXT: ; implicit-def: $vgpr0
@@ -13214,13 +13205,14 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1164-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-FAKE16-NEXT: s_and_b32 s4, s2, -4
; GFX1164-FAKE16-NEXT: s_mov_b32 s5, s3
-; GFX1164-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1164-FAKE16-NEXT: s_sext_i32_i16 s10, s12
; GFX1164-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
-; GFX1164-FAKE16-NEXT: s_sext_i32_i16 s7, s12
+; GFX1164-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1164-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1164-FAKE16-NEXT: s_lshl_b32 s2, s2, 3
-; GFX1164-FAKE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1164-FAKE16-NEXT: s_mul_i32 s10, s10, s6
; GFX1164-FAKE16-NEXT: s_lshl_b32 s13, 0xffff, s2
-; GFX1164-FAKE16-NEXT: s_and_b32 s6, 0xffff, s7
+; GFX1164-FAKE16-NEXT: s_and_b32 s6, 0xffff, s10
; GFX1164-FAKE16-NEXT: s_not_b32 s14, s13
; GFX1164-FAKE16-NEXT: s_lshl_b32 s15, s6, s2
; GFX1164-FAKE16-NEXT: s_mov_b64 s[10:11], 0
@@ -13267,20 +13259,21 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1132-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-TRUE16-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1132-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1132-TRUE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-TRUE16-NEXT: s_mov_b32 s10, 0
-; GFX1132-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1132-TRUE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1132-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1132-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-TRUE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1132-TRUE16-NEXT: ; %bb.1:
; GFX1132-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-TRUE16-NEXT: s_and_b32 s4, s2, -4
; GFX1132-TRUE16-NEXT: s_mov_b32 s5, s3
-; GFX1132-TRUE16-NEXT: s_and_b32 s2, s2, 3
-; GFX1132-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
; GFX1132-TRUE16-NEXT: s_sext_i32_i16 s7, s8
+; GFX1132-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
+; GFX1132-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1132-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1132-TRUE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1132-TRUE16-NEXT: s_mul_i32 s7, s7, s6
; GFX1132-TRUE16-NEXT: s_lshl_b32 s11, 0xffff, s2
@@ -13328,20 +13321,21 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1132-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-FAKE16-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1132-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1132-FAKE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-FAKE16-NEXT: s_mov_b32 s10, 0
-; GFX1132-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1132-FAKE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1132-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1132-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1132-FAKE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1132-FAKE16-NEXT: ; %bb.1:
; GFX1132-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-FAKE16-NEXT: s_and_b32 s4, s2, -4
; GFX1132-FAKE16-NEXT: s_mov_b32 s5, s3
-; GFX1132-FAKE16-NEXT: s_and_b32 s2, s2, 3
-; GFX1132-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
; GFX1132-FAKE16-NEXT: s_sext_i32_i16 s7, s8
+; GFX1132-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
+; GFX1132-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1132-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1132-FAKE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1132-FAKE16-NEXT: s_mul_i32 s7, s7, s6
; GFX1132-FAKE16-NEXT: s_lshl_b32 s11, 0xffff, s2
@@ -13389,9 +13383,8 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1264-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1264-TRUE16-NEXT: s_load_b32 s12, s[4:5], 0x34
; GFX1264-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1264-TRUE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1264-TRUE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1264-TRUE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1264-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1264-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1264-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
@@ -13401,13 +13394,14 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1264-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1264-TRUE16-NEXT: s_and_b32 s4, s2, -4
; GFX1264-TRUE16-NEXT: s_mov_b32 s5, s3
-; GFX1264-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1264-TRUE16-NEXT: s_sext_i32_i16 s10, s12
; GFX1264-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
-; GFX1264-TRUE16-NEXT: s_sext_i32_i16 s7, s12
+; GFX1264-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1264-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1264-TRUE16-NEXT: s_lshl_b32 s2, s2, 3
-; GFX1264-TRUE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1264-TRUE16-NEXT: s_mul_i32 s10, s10, s6
; GFX1264-TRUE16-NEXT: s_lshl_b32 s13, 0xffff, s2
-; GFX1264-TRUE16-NEXT: s_and_b32 s6, 0xffff, s7
+; GFX1264-TRUE16-NEXT: s_and_b32 s6, 0xffff, s10
; GFX1264-TRUE16-NEXT: s_not_b32 s14, s13
; GFX1264-TRUE16-NEXT: s_lshl_b32 s15, s6, s2
; GFX1264-TRUE16-NEXT: s_mov_b64 s[10:11], 0
@@ -13454,9 +13448,8 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1264-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1264-FAKE16-NEXT: s_load_b32 s12, s[4:5], 0x34
; GFX1264-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1264-FAKE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1264-FAKE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1264-FAKE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1264-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1264-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1264-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1264-FAKE16-NEXT: ; implicit-def: $vgpr0
@@ -13466,13 +13459,14 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1264-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1264-FAKE16-NEXT: s_and_b32 s4, s2, -4
; GFX1264-FAKE16-NEXT: s_mov_b32 s5, s3
-; GFX1264-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1264-FAKE16-NEXT: s_sext_i32_i16 s10, s12
; GFX1264-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
-; GFX1264-FAKE16-NEXT: s_sext_i32_i16 s7, s12
+; GFX1264-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1264-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1264-FAKE16-NEXT: s_lshl_b32 s2, s2, 3
-; GFX1264-FAKE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1264-FAKE16-NEXT: s_mul_i32 s10, s10, s6
; GFX1264-FAKE16-NEXT: s_lshl_b32 s13, 0xffff, s2
-; GFX1264-FAKE16-NEXT: s_and_b32 s6, 0xffff, s7
+; GFX1264-FAKE16-NEXT: s_and_b32 s6, 0xffff, s10
; GFX1264-FAKE16-NEXT: s_not_b32 s14, s13
; GFX1264-FAKE16-NEXT: s_lshl_b32 s15, s6, s2
; GFX1264-FAKE16-NEXT: s_mov_b64 s[10:11], 0
@@ -13519,20 +13513,21 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1232-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-TRUE16-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1232-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1232-TRUE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-TRUE16-NEXT: s_mov_b32 s10, 0
-; GFX1232-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1232-TRUE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1232-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1232-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1232-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1232-TRUE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1232-TRUE16-NEXT: ; %bb.1:
; GFX1232-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1232-TRUE16-NEXT: s_and_b32 s4, s2, -4
; GFX1232-TRUE16-NEXT: s_mov_b32 s5, s3
-; GFX1232-TRUE16-NEXT: s_and_b32 s2, s2, 3
-; GFX1232-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
; GFX1232-TRUE16-NEXT: s_sext_i32_i16 s7, s8
+; GFX1232-TRUE16-NEXT: s_load_b32 s5, s[4:5], 0x0
+; GFX1232-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1232-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1232-TRUE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1232-TRUE16-NEXT: s_mul_i32 s7, s7, s6
; GFX1232-TRUE16-NEXT: s_lshl_b32 s11, 0xffff, s2
@@ -13581,20 +13576,21 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1232-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1232-FAKE16-NEXT: s_load_b32 s8, s[4:5], 0x34
; GFX1232-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1232-FAKE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1232-FAKE16-NEXT: s_mov_b32 s10, 0
-; GFX1232-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1232-FAKE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1232-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1232-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1232-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1232-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1232-FAKE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1232-FAKE16-NEXT: ; %bb.1:
; GFX1232-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1232-FAKE16-NEXT: s_and_b32 s4, s2, -4
; GFX1232-FAKE16-NEXT: s_mov_b32 s5, s3
-; GFX1232-FAKE16-NEXT: s_and_b32 s2, s2, 3
-; GFX1232-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
; GFX1232-FAKE16-NEXT: s_sext_i32_i16 s7, s8
+; GFX1232-FAKE16-NEXT: s_load_b32 s5, s[4:5], 0x0
+; GFX1232-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1232-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1232-FAKE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1232-FAKE16-NEXT: s_mul_i32 s7, s7, s6
; GFX1232-FAKE16-NEXT: s_lshl_b32 s11, 0xffff, s2
@@ -13643,9 +13639,8 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1364-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-TRUE16-NEXT: s_load_b32 s10, s[4:5], 0x34 nv
; GFX1364-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1364-TRUE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-TRUE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-TRUE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1364-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1364-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-TRUE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1364-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
@@ -13654,13 +13649,14 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1364-TRUE16-NEXT: ; %bb.1:
; GFX1364-TRUE16-NEXT: s_wait_kmcnt 0x0
; GFX1364-TRUE16-NEXT: s_and_b64 s[4:5], s[2:3], -4
-; GFX1364-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1364-TRUE16-NEXT: s_sext_i32_i16 s12, s10
; GFX1364-TRUE16-NEXT: s_load_b32 s3, s[4:5], 0x0
-; GFX1364-TRUE16-NEXT: s_sext_i32_i16 s7, s10
+; GFX1364-TRUE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1364-TRUE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1364-TRUE16-NEXT: s_lshl_b32 s11, s2, 3
-; GFX1364-TRUE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1364-TRUE16-NEXT: s_mul_i32 s2, s12, s6
; GFX1364-TRUE16-NEXT: s_lshl_b32 s12, 0xffff, s11
-; GFX1364-TRUE16-NEXT: s_and_b32 s2, 0xffff, s7
+; GFX1364-TRUE16-NEXT: s_and_b32 s2, 0xffff, s2
; GFX1364-TRUE16-NEXT: s_not_b32 s13, s12
; GFX1364-TRUE16-NEXT: s_lshl_b32 s14, s2, s11
; GFX1364-TRUE16-NEXT: s_mov_b32 s7, 0x31016000
@@ -13705,9 +13701,8 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1364-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-FAKE16-NEXT: s_load_b32 s10, s[4:5], 0x34 nv
; GFX1364-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1364-FAKE16-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-FAKE16-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-FAKE16-NEXT: s_mov_b64 s[8:9], exec
-; GFX1364-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
; GFX1364-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-FAKE16-NEXT: v_mbcnt_hi_u32_b32 v4, exec_hi, v0
; GFX1364-FAKE16-NEXT: ; implicit-def: $vgpr0
@@ -13716,13 +13711,14 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1364-FAKE16-NEXT: ; %bb.1:
; GFX1364-FAKE16-NEXT: s_wait_kmcnt 0x0
; GFX1364-FAKE16-NEXT: s_and_b64 s[4:5], s[2:3], -4
-; GFX1364-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1364-FAKE16-NEXT: s_sext_i32_i16 s12, s10
; GFX1364-FAKE16-NEXT: s_load_b32 s3, s[4:5], 0x0
-; GFX1364-FAKE16-NEXT: s_sext_i32_i16 s7, s10
+; GFX1364-FAKE16-NEXT: s_and_b32 s2, s2, 3
+; GFX1364-FAKE16-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GFX1364-FAKE16-NEXT: s_lshl_b32 s11, s2, 3
-; GFX1364-FAKE16-NEXT: s_mul_i32 s7, s7, s6
+; GFX1364-FAKE16-NEXT: s_mul_i32 s2, s12, s6
; GFX1364-FAKE16-NEXT: s_lshl_b32 s12, 0xffff, s11
-; GFX1364-FAKE16-NEXT: s_and_b32 s2, 0xffff, s7
+; GFX1364-FAKE16-NEXT: s_and_b32 s2, 0xffff, s2
; GFX1364-FAKE16-NEXT: s_not_b32 s13, s12
; GFX1364-FAKE16-NEXT: s_lshl_b32 s14, s2, s11
; GFX1364-FAKE16-NEXT: s_mov_b32 s7, 0x31016000
@@ -13767,11 +13763,11 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1332-TRUE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-TRUE16-NEXT: s_load_b32 s8, s[4:5], 0x34 nv
; GFX1332-TRUE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1332-TRUE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-TRUE16-NEXT: s_mov_b32 s10, 0
-; GFX1332-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1332-TRUE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1332-TRUE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1332-TRUE16-NEXT: ; implicit-def: $vgpr0_lo16
+; GFX1332-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-TRUE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1332-TRUE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1332-TRUE16-NEXT: ; %bb.1:
@@ -13780,6 +13776,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1332-TRUE16-NEXT: s_and_b32 s2, s2, 3
; GFX1332-TRUE16-NEXT: s_load_b32 s7, s[4:5], 0x0
; GFX1332-TRUE16-NEXT: s_sext_i32_i16 s11, s8
+; GFX1332-TRUE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1332-TRUE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1332-TRUE16-NEXT: s_mul_i32 s6, s11, s6
; GFX1332-TRUE16-NEXT: s_lshl_b32 s3, 0xffff, s2
@@ -13826,11 +13823,11 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1332-FAKE16-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-FAKE16-NEXT: s_load_b32 s8, s[4:5], 0x34 nv
; GFX1332-FAKE16-NEXT: v_mbcnt_lo_u32_b32 v4, exec_lo, 0
-; GFX1332-FAKE16-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-FAKE16-NEXT: s_mov_b32 s10, 0
-; GFX1332-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s4
+; GFX1332-FAKE16-NEXT: s_mov_b32 s6, exec_lo
; GFX1332-FAKE16-NEXT: s_mov_b32 s9, exec_lo
; GFX1332-FAKE16-NEXT: ; implicit-def: $vgpr0
+; GFX1332-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-FAKE16-NEXT: v_cmpx_eq_u32_e32 0, v4
; GFX1332-FAKE16-NEXT: s_cbranch_execz .LBB16_4
; GFX1332-FAKE16-NEXT: ; %bb.1:
@@ -13839,6 +13836,7 @@ define amdgpu_kernel void @uniform_add_i16(ptr addrspace(1) %result, ptr addrspa
; GFX1332-FAKE16-NEXT: s_and_b32 s2, s2, 3
; GFX1332-FAKE16-NEXT: s_load_b32 s7, s[4:5], 0x0
; GFX1332-FAKE16-NEXT: s_sext_i32_i16 s11, s8
+; GFX1332-FAKE16-NEXT: s_bcnt1_i32_b32 s6, s6
; GFX1332-FAKE16-NEXT: s_lshl_b32 s2, s2, 3
; GFX1332-FAKE16-NEXT: s_mul_i32 s6, s11, s6
; GFX1332-FAKE16-NEXT: s_lshl_b32 s3, 0xffff, s2
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll
index 6f7b956857742c..4b16a4a371c183 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll
@@ -28,13 +28,13 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX7LESS: ; %bb.0: ; %entry
; GFX7LESS-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX7LESS-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX7LESS-NEXT: s_mov_b64 s[0:1], exec
-; GFX7LESS-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX7LESS-NEXT: s_mov_b64 s[2:3], exec
; GFX7LESS-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX7LESS-NEXT: ; implicit-def: $vgpr1
; GFX7LESS-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX7LESS-NEXT: s_cbranch_execz .LBB0_2
; GFX7LESS-NEXT: ; %bb.1:
+; GFX7LESS-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX7LESS-NEXT: s_mul_i32 s2, s2, 5
; GFX7LESS-NEXT: v_mov_b32_e32 v1, 0
; GFX7LESS-NEXT: v_mov_b32_e32 v2, s2
@@ -56,13 +56,13 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB0_2
; GFX8-NEXT: ; %bb.1:
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_mul_i32 s2, s2, 5
; GFX8-NEXT: v_mov_b32_e32 v1, 0
; GFX8-NEXT: v_mov_b32_e32 v2, s2
@@ -84,13 +84,13 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB0_2
; GFX9-NEXT: ; %bb.1:
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_mul_i32 s2, s2, 5
; GFX9-NEXT: v_mov_b32_e32 v1, 0
; GFX9-NEXT: v_mov_b32_e32 v2, s2
@@ -110,16 +110,16 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX1064-LABEL: add_i32_constant:
; GFX1064: ; %bb.0: ; %entry
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[0:1], exec
+; GFX1064-NEXT: s_mov_b64 s[2:3], exec
; GFX1064-NEXT: ; implicit-def: $vgpr1
-; GFX1064-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1064-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB0_2
; GFX1064-NEXT: ; %bb.1:
-; GFX1064-NEXT: s_mul_i32 s2, s2, 5
+; GFX1064-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1064-NEXT: v_mov_b32_e32 v1, 0
+; GFX1064-NEXT: s_mul_i32 s2, s2, 5
; GFX1064-NEXT: v_mov_b32_e32 v2, s2
; GFX1064-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
@@ -139,15 +139,15 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX1032-LABEL: add_i32_constant:
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s0, exec_lo
+; GFX1032-NEXT: s_mov_b32 s1, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1032-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB0_2
; GFX1032-NEXT: ; %bb.1:
-; GFX1032-NEXT: s_mul_i32 s1, s1, 5
+; GFX1032-NEXT: s_bcnt1_i32_b32 s1, s1
; GFX1032-NEXT: v_mov_b32_e32 v1, 0
+; GFX1032-NEXT: s_mul_i32 s1, s1, 5
; GFX1032-NEXT: v_mov_b32_e32 v2, s1
; GFX1032-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
@@ -167,18 +167,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX1164-LABEL: add_i32_constant:
; GFX1164: ; %bb.0: ; %entry
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[2:3], exec
; GFX1164-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-NEXT: ; implicit-def: $vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX1164-NEXT: s_mov_b64 s[0:1], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1164-NEXT: s_cbranch_execz .LBB0_2
; GFX1164-NEXT: ; %bb.1:
-; GFX1164-NEXT: s_mul_i32 s2, s2, 5
+; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v1, 0
+; GFX1164-NEXT: s_mul_i32 s2, s2, 5
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: v_mov_b32_e32 v2, s2
; GFX1164-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
@@ -197,16 +197,16 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX1132-LABEL: add_i32_constant:
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX1132-NEXT: s_mov_b32 s0, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB0_2
; GFX1132-NEXT: ; %bb.1:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_mul_i32 s1, s1, 5
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v2, s1
; GFX1132-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
@@ -225,18 +225,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX1364-LABEL: add_i32_constant:
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[2:3], exec
; GFX1364-NEXT: s_mov_b64 s[0:1], exec
; GFX1364-NEXT: ; implicit-def: $vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX1364-NEXT: s_mov_b64 s[0:1], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1364-NEXT: s_cbranch_execz .LBB0_2
; GFX1364-NEXT: ; %bb.1:
-; GFX1364-NEXT: s_mul_i32 s2, s2, 5
+; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1364-NEXT: v_mov_b32_e32 v1, 0
+; GFX1364-NEXT: s_mul_i32 s2, s2, 5
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_mov_b32_e32 v2, s2
; GFX1364-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX1364-NEXT: s_wait_dscnt 0x0
@@ -255,16 +255,16 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out) {
; GFX1332-LABEL: add_i32_constant:
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s1, exec_lo
; GFX1332-NEXT: s_mov_b32 s0, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX1332-NEXT: s_mov_b32 s0, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB0_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: s_mul_i32 s1, s1, 5
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v2, s1
; GFX1332-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX1332-NEXT: s_wait_dscnt 0x0
@@ -288,20 +288,20 @@ entry:
define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive) {
; GFX7LESS-LABEL: add_i32_uniform:
; GFX7LESS: ; %bb.0: ; %entry
-; GFX7LESS-NEXT: s_load_dword s2, s[4:5], 0xb
+; GFX7LESS-NEXT: s_load_dword s6, s[4:5], 0xb
; GFX7LESS-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX7LESS-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX7LESS-NEXT: s_mov_b64 s[0:1], exec
-; GFX7LESS-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX7LESS-NEXT: s_mov_b64 s[2:3], exec
; GFX7LESS-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX7LESS-NEXT: ; implicit-def: $vgpr1
; GFX7LESS-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX7LESS-NEXT: s_cbranch_execz .LBB1_2
; GFX7LESS-NEXT: ; %bb.1:
+; GFX7LESS-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: s_mul_i32 s3, s2, s3
+; GFX7LESS-NEXT: s_mul_i32 s2, s6, s2
; GFX7LESS-NEXT: v_mov_b32_e32 v1, 0
-; GFX7LESS-NEXT: v_mov_b32_e32 v2, s3
+; GFX7LESS-NEXT: v_mov_b32_e32 v2, s2
; GFX7LESS-NEXT: s_mov_b32 m0, -1
; GFX7LESS-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
@@ -309,7 +309,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
; GFX7LESS-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX7LESS-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX7LESS-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX7LESS-NEXT: v_readfirstlane_b32 s4, v1
; GFX7LESS-NEXT: s_mov_b32 s3, 0xf000
; GFX7LESS-NEXT: s_mov_b32 s2, -1
@@ -319,20 +319,20 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
;
; GFX8-LABEL: add_i32_uniform:
; GFX8: ; %bb.0: ; %entry
-; GFX8-NEXT: s_load_dword s2, s[4:5], 0x2c
+; GFX8-NEXT: s_load_dword s6, s[4:5], 0x2c
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB1_2
; GFX8-NEXT: ; %bb.1:
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_mul_i32 s3, s2, s3
+; GFX8-NEXT: s_mul_i32 s2, s6, s2
; GFX8-NEXT: v_mov_b32_e32 v1, 0
-; GFX8-NEXT: v_mov_b32_e32 v2, s3
+; GFX8-NEXT: v_mov_b32_e32 v2, s2
; GFX8-NEXT: s_mov_b32 m0, -1
; GFX8-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
@@ -340,7 +340,7 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
; GFX8-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX8-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX8-NEXT: v_readfirstlane_b32 s4, v1
; GFX8-NEXT: s_mov_b32 s3, 0xf000
; GFX8-NEXT: s_mov_b32 s2, -1
@@ -350,27 +350,27 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
;
; GFX9-LABEL: add_i32_uniform:
; GFX9: ; %bb.0: ; %entry
-; GFX9-NEXT: s_load_dword s2, s[4:5], 0x2c
+; GFX9-NEXT: s_load_dword s6, s[4:5], 0x2c
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB1_2
; GFX9-NEXT: ; %bb.1:
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s3, s2, s3
+; GFX9-NEXT: s_mul_i32 s2, s6, s2
; GFX9-NEXT: v_mov_b32_e32 v1, 0
-; GFX9-NEXT: v_mov_b32_e32 v2, s3
+; GFX9-NEXT: v_mov_b32_e32 v2, s2
; GFX9-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-NEXT: .LBB1_2:
; GFX9-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX9-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX9-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX9-NEXT: v_readfirstlane_b32 s4, v1
; GFX9-NEXT: s_mov_b32 s3, 0xf000
; GFX9-NEXT: s_mov_b32 s2, -1
@@ -380,20 +380,20 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
;
; GFX1064-LABEL: add_i32_uniform:
; GFX1064: ; %bb.0: ; %entry
-; GFX1064-NEXT: s_load_dword s2, s[4:5], 0x2c
+; GFX1064-NEXT: s_load_dword s6, s[4:5], 0x2c
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[0:1], exec
+; GFX1064-NEXT: s_mov_b64 s[2:3], exec
; GFX1064-NEXT: ; implicit-def: $vgpr1
-; GFX1064-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1064-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB1_2
; GFX1064-NEXT: ; %bb.1:
-; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1064-NEXT: s_mul_i32 s3, s2, s3
+; GFX1064-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1064-NEXT: v_mov_b32_e32 v1, 0
-; GFX1064-NEXT: v_mov_b32_e32 v2, s3
+; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
+; GFX1064-NEXT: s_mul_i32 s2, s6, s2
+; GFX1064-NEXT: v_mov_b32_e32 v2, s2
; GFX1064-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: buffer_gl0_inv
@@ -401,10 +401,9 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
; GFX1064-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)
; GFX1064-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX1064-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
-; GFX1064-NEXT: s_mov_b32 null, 0
-; GFX1064-NEXT: v_readfirstlane_b32 s4, v1
+; GFX1064-NEXT: v_readfirstlane_b32 s2, v1
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1064-NEXT: v_mad_u64_u32 v[0:1], s[2:3], s2, v0, s[4:5]
+; GFX1064-NEXT: v_mad_u64_u32 v[0:1], s[2:3], s6, v0, s[2:3]
; GFX1064-NEXT: s_mov_b32 s3, 0x31016000
; GFX1064-NEXT: s_mov_b32 s2, -1
; GFX1064-NEXT: buffer_store_dword v0, off, s[0:3], 0
@@ -414,16 +413,16 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: s_load_dword s0, s[4:5], 0x2c
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s1, exec_lo
+; GFX1032-NEXT: s_mov_b32 s2, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s2, s1
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1032-NEXT: s_and_saveexec_b32 s1, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB1_2
; GFX1032-NEXT: ; %bb.1:
+; GFX1032-NEXT: s_bcnt1_i32_b32 s2, s2
+; GFX1032-NEXT: v_mov_b32_e32 v1, 0
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_mul_i32 s2, s0, s2
-; GFX1032-NEXT: v_mov_b32_e32 v1, 0
; GFX1032-NEXT: v_mov_b32_e32 v2, s2
; GFX1032-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
@@ -442,32 +441,33 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
;
; GFX1164-LABEL: add_i32_uniform:
; GFX1164: ; %bb.0: ; %entry
-; GFX1164-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX1164-NEXT: s_load_b32 s6, s[4:5], 0x2c
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[2:3], exec
; GFX1164-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-NEXT: ; implicit-def: $vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX1164-NEXT: s_mov_b64 s[0:1], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1164-NEXT: s_cbranch_execz .LBB1_2
; GFX1164-NEXT: ; %bb.1:
-; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1164-NEXT: s_mul_i32 s3, s2, s3
+; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v1, 0
-; GFX1164-NEXT: v_mov_b32_e32 v2, s3
+; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
+; GFX1164-NEXT: s_mul_i32 s2, s6, s2
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: v_mov_b32_e32 v2, s2
; GFX1164-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-NEXT: buffer_gl0_inv
; GFX1164-NEXT: .LBB1_2:
; GFX1164-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX1164-NEXT: v_readfirstlane_b32 s4, v1
-; GFX1164-NEXT: s_mov_b32 s3, 0x31016000
+; GFX1164-NEXT: v_readfirstlane_b32 s2, v1
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1164-NEXT: v_mad_u64_u32 v[1:2], null, s2, v0, s[4:5]
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1164-NEXT: v_mad_u64_u32 v[1:2], null, s6, v0, s[2:3]
+; GFX1164-NEXT: s_mov_b32 s3, 0x31016000
; GFX1164-NEXT: s_mov_b32 s2, -1
; GFX1164-NEXT: buffer_store_b32 v1, off, s[0:3], 0
; GFX1164-NEXT: s_endpgm
@@ -476,14 +476,14 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s2, exec_lo
; GFX1132-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX1132-NEXT: s_mov_b32 s1, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB1_2
; GFX1132-NEXT: ; %bb.1:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_mul_i32 s2, s0, s2
; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -504,32 +504,33 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
;
; GFX1364-LABEL: add_i32_uniform:
; GFX1364: ; %bb.0: ; %entry
-; GFX1364-NEXT: s_load_b32 s2, s[4:5], 0x2c nv
+; GFX1364-NEXT: s_load_b32 s6, s[4:5], 0x2c nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[2:3], exec
; GFX1364-NEXT: s_mov_b64 s[0:1], exec
; GFX1364-NEXT: ; implicit-def: $vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX1364-NEXT: s_mov_b64 s[0:1], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1364-NEXT: s_cbranch_execz .LBB1_2
; GFX1364-NEXT: ; %bb.1:
-; GFX1364-NEXT: s_wait_kmcnt 0x0
-; GFX1364-NEXT: s_mul_i32 s3, s2, s3
+; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1364-NEXT: v_mov_b32_e32 v1, 0
-; GFX1364-NEXT: v_mov_b32_e32 v2, s3
+; GFX1364-NEXT: s_wait_kmcnt 0x0
+; GFX1364-NEXT: s_mul_i32 s2, s6, s2
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-NEXT: v_mov_b32_e32 v2, s2
; GFX1364-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX1364-NEXT: s_wait_dscnt 0x0
; GFX1364-NEXT: global_inv scope:SCOPE_SE
; GFX1364-NEXT: .LBB1_2:
; GFX1364-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX1364-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1364-NEXT: v_readfirstlane_b32 s4, v1
-; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
+; GFX1364-NEXT: v_readfirstlane_b32 s2, v1
; GFX1364-NEXT: s_wait_kmcnt 0x0
-; GFX1364-NEXT: v_mad_co_u64_u32 v[0:1], null, s2, v0, s[4:5]
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX1364-NEXT: v_mad_co_u64_u32 v[0:1], null, s6, v0, s[2:3]
+; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
; GFX1364-NEXT: s_mov_b32 s2, -1
; GFX1364-NEXT: buffer_store_b32 v0, off, s[0:3], null
; GFX1364-NEXT: s_endpgm
@@ -538,14 +539,14 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, i32 %additive)
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: s_load_b32 s0, s[4:5], 0x2c nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s2, exec_lo
; GFX1332-NEXT: s_mov_b32 s1, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX1332-NEXT: s_mov_b32 s1, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB1_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mul_i32 s2, s0, s2
; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -1839,20 +1840,20 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out) {
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX9-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB4_2
; GFX9-NEXT: ; %bb.1:
-; GFX9-NEXT: s_mul_i32 s6, s2, 5
-; GFX9-NEXT: s_mul_hi_u32 s3, 5, s2
-; GFX9-NEXT: s_mul_i32 s2, s2, 0
-; GFX9-NEXT: s_add_u32 s7, s3, s2
+; GFX9-NEXT: s_bcnt1_i32_b64 s3, s[2:3]
+; GFX9-NEXT: s_mul_i32 s2, s3, 5
+; GFX9-NEXT: s_mul_hi_u32 s6, 5, s3
+; GFX9-NEXT: s_mul_i32 s3, s3, 0
+; GFX9-NEXT: s_add_u32 s3, s6, s3
; GFX9-NEXT: v_mov_b32_e32 v3, 0
-; GFX9-NEXT: v_mov_b32_e32 v0, s6
-; GFX9-NEXT: v_mov_b32_e32 v1, s7
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
; GFX9-NEXT: ds_add_rtn_u64 v[0:1], v3, v[0:1]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-NEXT: .LBB4_2:
@@ -1873,19 +1874,19 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out) {
; GFX1064-LABEL: add_i64_constant:
; GFX1064: ; %bb.0: ; %entry
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[0:1], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX1064-NEXT: s_mov_b64 s[2:3], exec
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1064-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB4_2
; GFX1064-NEXT: ; %bb.1:
+; GFX1064-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX1064-NEXT: v_mov_b32_e32 v3, 0
; GFX1064-NEXT: s_mul_hi_u32 s3, 5, s2
; GFX1064-NEXT: s_mul_i32 s6, s2, 0
; GFX1064-NEXT: s_mul_i32 s2, s2, 5
; GFX1064-NEXT: s_add_u32 s3, s3, s6
-; GFX1064-NEXT: v_mov_b32_e32 v3, 0
; GFX1064-NEXT: v_mov_b32_e32 v0, s2
; GFX1064-NEXT: v_mov_b32_e32 v1, s3
; GFX1064-NEXT: ds_add_rtn_u64 v[0:1], v3, v[0:1]
@@ -1907,18 +1908,18 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out) {
; GFX1032-LABEL: add_i64_constant:
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s0, exec_lo
+; GFX1032-NEXT: s_mov_b32 s1, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v2
; GFX1032-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB4_2
; GFX1032-NEXT: ; %bb.1:
+; GFX1032-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX1032-NEXT: v_mov_b32_e32 v3, 0
; GFX1032-NEXT: s_mul_hi_u32 s3, 5, s1
; GFX1032-NEXT: s_mul_i32 s6, s1, 0
; GFX1032-NEXT: s_mul_i32 s2, s1, 5
; GFX1032-NEXT: s_add_u32 s3, s3, s6
-; GFX1032-NEXT: v_mov_b32_e32 v3, 0
; GFX1032-NEXT: v_mov_b32_e32 v0, s2
; GFX1032-NEXT: v_mov_b32_e32 v1, s3
; GFX1032-NEXT: ds_add_rtn_u64 v[0:1], v3, v[0:1]
@@ -1940,21 +1941,20 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out) {
; GFX1164-LABEL: add_i64_constant:
; GFX1164: ; %bb.0: ; %entry
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[2:3], exec
; GFX1164-NEXT: s_mov_b64 s[0:1], exec
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX1164-NEXT: s_mov_b64 s[0:1], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB4_2
; GFX1164-NEXT: ; %bb.1:
+; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX1164-NEXT: v_mov_b32_e32 v3, 0
; GFX1164-NEXT: s_mul_hi_u32 s3, 5, s2
; GFX1164-NEXT: s_mul_i32 s6, s2, 0
; GFX1164-NEXT: s_mul_i32 s2, s2, 5
; GFX1164-NEXT: s_add_u32 s3, s3, s6
-; GFX1164-NEXT: v_mov_b32_e32 v3, 0
; GFX1164-NEXT: v_mov_b32_e32 v0, s2
; GFX1164-NEXT: v_mov_b32_e32 v1, s3
; GFX1164-NEXT: ds_add_rtn_u64 v[0:1], v3, v[0:1]
@@ -1976,14 +1976,15 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out) {
; GFX1132-LABEL: add_i64_constant:
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX1132-NEXT: s_mov_b32 s0, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB4_2
; GFX1132-NEXT: ; %bb.1:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: s_mul_hi_u32 s3, 5, s1
; GFX1132-NEXT: s_mul_i32 s6, s1, 0
; GFX1132-NEXT: s_mul_i32 s2, s1, 5
@@ -2009,21 +2010,20 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out) {
; GFX1364-LABEL: add_i64_constant:
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[2:3], exec
; GFX1364-NEXT: s_mov_b64 s[0:1], exec
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX1364-NEXT: s_mov_b64 s[0:1], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1364-NEXT: s_cbranch_execz .LBB4_2
; GFX1364-NEXT: ; %bb.1:
+; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX1364-NEXT: v_mov_b32_e32 v3, 0
; GFX1364-NEXT: s_mul_hi_u32 s3, 5, s2
; GFX1364-NEXT: s_mul_i32 s6, s2, 0
; GFX1364-NEXT: s_mul_i32 s2, s2, 5
; GFX1364-NEXT: s_add_co_u32 s3, s3, s6
-; GFX1364-NEXT: v_mov_b32_e32 v3, 0
; GFX1364-NEXT: v_mov_b32_e32 v0, s2
; GFX1364-NEXT: v_mov_b32_e32 v1, s3
; GFX1364-NEXT: ds_add_rtn_u64 v[0:1], v3, v[0:1]
@@ -2045,14 +2045,15 @@ define amdgpu_kernel void @add_i64_constant(ptr addrspace(1) %out) {
; GFX1332-LABEL: add_i64_constant:
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s1, exec_lo
; GFX1332-NEXT: s_mov_b32 s0, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX1332-NEXT: s_mov_b32 s0, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB4_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: s_mul_hi_u32 s3, 5, s1
; GFX1332-NEXT: s_mul_i32 s6, s1, 0
; GFX1332-NEXT: s_mul_i32 s2, s1, 5
@@ -2170,21 +2171,21 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GFX9-NEXT: s_mov_b64 s[6:7], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX9-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX9-NEXT: s_cbranch_execz .LBB5_2
; GFX9-NEXT: ; %bb.1:
+; GFX9-NEXT: s_bcnt1_i32_b64 s7, s[6:7]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s8, s2, s6
-; GFX9-NEXT: s_mul_hi_u32 s7, s2, s6
-; GFX9-NEXT: s_mul_i32 s6, s3, s6
-; GFX9-NEXT: s_add_u32 s9, s7, s6
+; GFX9-NEXT: s_mul_i32 s6, s2, s7
+; GFX9-NEXT: s_mul_hi_u32 s8, s2, s7
+; GFX9-NEXT: s_mul_i32 s7, s3, s7
+; GFX9-NEXT: s_add_u32 s7, s8, s7
; GFX9-NEXT: v_mov_b32_e32 v3, 0
-; GFX9-NEXT: v_mov_b32_e32 v0, s8
-; GFX9-NEXT: v_mov_b32_e32 v1, s9
+; GFX9-NEXT: v_mov_b32_e32 v0, s6
+; GFX9-NEXT: v_mov_b32_e32 v1, s7
; GFX9-NEXT: ds_add_rtn_u64 v[0:1], v3, v[0:1]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-NEXT: .LBB5_2:
@@ -2207,20 +2208,20 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1064: ; %bb.0: ; %entry
; GFX1064-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[4:5], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GFX1064-NEXT: s_mov_b64 s[6:7], exec
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1064-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB5_2
; GFX1064-NEXT: ; %bb.1:
+; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1064-NEXT: v_mov_b32_e32 v3, 0
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: s_mul_hi_u32 s7, s2, s6
; GFX1064-NEXT: s_mul_i32 s8, s3, s6
; GFX1064-NEXT: s_mul_i32 s6, s2, s6
; GFX1064-NEXT: s_add_u32 s7, s7, s8
-; GFX1064-NEXT: v_mov_b32_e32 v3, 0
; GFX1064-NEXT: v_mov_b32_e32 v0, s6
; GFX1064-NEXT: v_mov_b32_e32 v1, s7
; GFX1064-NEXT: ds_add_rtn_u64 v[0:1], v3, v[0:1]
@@ -2243,19 +2244,19 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
+; GFX1032-NEXT: s_mov_b32 s5, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s4
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v2
; GFX1032-NEXT: s_and_saveexec_b32 s4, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB5_2
; GFX1032-NEXT: ; %bb.1:
+; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s5
+; GFX1032-NEXT: v_mov_b32_e32 v3, 0
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_mul_hi_u32 s7, s2, s5
; GFX1032-NEXT: s_mul_i32 s8, s3, s5
; GFX1032-NEXT: s_mul_i32 s6, s2, s5
; GFX1032-NEXT: s_add_u32 s7, s7, s8
-; GFX1032-NEXT: v_mov_b32_e32 v3, 0
; GFX1032-NEXT: v_mov_b32_e32 v0, s6
; GFX1032-NEXT: v_mov_b32_e32 v1, s7
; GFX1032-NEXT: ds_add_rtn_u64 v[0:1], v3, v[0:1]
@@ -2278,22 +2279,21 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1164: ; %bb.0: ; %entry
; GFX1164-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[6:7], exec
; GFX1164-NEXT: s_mov_b64 s[4:5], exec
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1164-NEXT: s_mov_b64 s[4:5], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB5_2
; GFX1164-NEXT: ; %bb.1:
+; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1164-NEXT: v_mov_b32_e32 v3, 0
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-NEXT: s_mul_hi_u32 s7, s2, s6
; GFX1164-NEXT: s_mul_i32 s8, s3, s6
; GFX1164-NEXT: s_mul_i32 s6, s2, s6
; GFX1164-NEXT: s_add_u32 s7, s7, s8
-; GFX1164-NEXT: v_mov_b32_e32 v3, 0
; GFX1164-NEXT: v_mov_b32_e32 v0, s6
; GFX1164-NEXT: v_mov_b32_e32 v1, s7
; GFX1164-NEXT: ds_add_rtn_u64 v[0:1], v3, v[0:1]
@@ -2317,14 +2317,14 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s5, exec_lo
; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1132-NEXT: s_mov_b32 s4, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB5_2
; GFX1132-NEXT: ; %bb.1:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s5, s5
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_mul_hi_u32 s7, s2, s5
; GFX1132-NEXT: s_mul_i32 s8, s3, s5
@@ -2353,22 +2353,21 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-NEXT: s_mov_b64 s[4:5], exec
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1364-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1364-NEXT: s_cbranch_execz .LBB5_2
; GFX1364-NEXT: ; %bb.1:
+; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1364-NEXT: v_mov_b32_e32 v3, 0
; GFX1364-NEXT: s_wait_kmcnt 0x0
; GFX1364-NEXT: s_mul_hi_u32 s7, s2, s6
; GFX1364-NEXT: s_mul_i32 s8, s3, s6
; GFX1364-NEXT: s_mul_i32 s6, s2, s6
; GFX1364-NEXT: s_add_co_u32 s7, s7, s8
-; GFX1364-NEXT: v_mov_b32_e32 v3, 0
; GFX1364-NEXT: v_mov_b32_e32 v0, s6
; GFX1364-NEXT: v_mov_b32_e32 v1, s7
; GFX1364-NEXT: ds_add_rtn_u64 v[0:1], v3, v[0:1]
@@ -2391,14 +2390,14 @@ define amdgpu_kernel void @add_i64_uniform(ptr addrspace(1) %out, i64 %additive)
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s5, exec_lo
; GFX1332-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1332-NEXT: s_mov_b32 s4, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB5_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s5
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mul_hi_u32 s7, s2, s5
; GFX1332-NEXT: s_mul_i32 s8, s3, s5
@@ -4258,13 +4257,13 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX7LESS: ; %bb.0: ; %entry
; GFX7LESS-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX7LESS-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX7LESS-NEXT: s_mov_b64 s[0:1], exec
-; GFX7LESS-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX7LESS-NEXT: s_mov_b64 s[2:3], exec
; GFX7LESS-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX7LESS-NEXT: ; implicit-def: $vgpr1
; GFX7LESS-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX7LESS-NEXT: s_cbranch_execz .LBB8_2
; GFX7LESS-NEXT: ; %bb.1:
+; GFX7LESS-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX7LESS-NEXT: s_mul_i32 s2, s2, 5
; GFX7LESS-NEXT: v_mov_b32_e32 v1, 0
; GFX7LESS-NEXT: v_mov_b32_e32 v2, s2
@@ -4287,13 +4286,13 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB8_2
; GFX8-NEXT: ; %bb.1:
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_mul_i32 s2, s2, 5
; GFX8-NEXT: v_mov_b32_e32 v1, 0
; GFX8-NEXT: v_mov_b32_e32 v2, s2
@@ -4316,13 +4315,13 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB8_2
; GFX9-NEXT: ; %bb.1:
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_mul_i32 s2, s2, 5
; GFX9-NEXT: v_mov_b32_e32 v1, 0
; GFX9-NEXT: v_mov_b32_e32 v2, s2
@@ -4343,16 +4342,16 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX1064-LABEL: sub_i32_constant:
; GFX1064: ; %bb.0: ; %entry
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[0:1], exec
+; GFX1064-NEXT: s_mov_b64 s[2:3], exec
; GFX1064-NEXT: ; implicit-def: $vgpr1
-; GFX1064-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1064-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB8_2
; GFX1064-NEXT: ; %bb.1:
-; GFX1064-NEXT: s_mul_i32 s2, s2, 5
+; GFX1064-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1064-NEXT: v_mov_b32_e32 v1, 0
+; GFX1064-NEXT: s_mul_i32 s2, s2, 5
; GFX1064-NEXT: v_mov_b32_e32 v2, s2
; GFX1064-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
@@ -4373,15 +4372,15 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX1032-LABEL: sub_i32_constant:
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s0, exec_lo
+; GFX1032-NEXT: s_mov_b32 s1, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1032-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB8_2
; GFX1032-NEXT: ; %bb.1:
-; GFX1032-NEXT: s_mul_i32 s1, s1, 5
+; GFX1032-NEXT: s_bcnt1_i32_b32 s1, s1
; GFX1032-NEXT: v_mov_b32_e32 v1, 0
+; GFX1032-NEXT: s_mul_i32 s1, s1, 5
; GFX1032-NEXT: v_mov_b32_e32 v2, s1
; GFX1032-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
@@ -4402,18 +4401,18 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX1164-LABEL: sub_i32_constant:
; GFX1164: ; %bb.0: ; %entry
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[2:3], exec
; GFX1164-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-NEXT: ; implicit-def: $vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX1164-NEXT: s_mov_b64 s[0:1], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1164-NEXT: s_cbranch_execz .LBB8_2
; GFX1164-NEXT: ; %bb.1:
-; GFX1164-NEXT: s_mul_i32 s2, s2, 5
+; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v1, 0
+; GFX1164-NEXT: s_mul_i32 s2, s2, 5
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: v_mov_b32_e32 v2, s2
; GFX1164-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
@@ -4434,16 +4433,16 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX1132-LABEL: sub_i32_constant:
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX1132-NEXT: s_mov_b32 s0, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB8_2
; GFX1132-NEXT: ; %bb.1:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_mul_i32 s1, s1, 5
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v2, s1
; GFX1132-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
@@ -4464,18 +4463,18 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX1364-LABEL: sub_i32_constant:
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[2:3], exec
; GFX1364-NEXT: s_mov_b64 s[0:1], exec
; GFX1364-NEXT: ; implicit-def: $vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX1364-NEXT: s_mov_b64 s[0:1], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1364-NEXT: s_cbranch_execz .LBB8_2
; GFX1364-NEXT: ; %bb.1:
-; GFX1364-NEXT: s_mul_i32 s2, s2, 5
+; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1364-NEXT: v_mov_b32_e32 v1, 0
+; GFX1364-NEXT: s_mul_i32 s2, s2, 5
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_mov_b32_e32 v2, s2
; GFX1364-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX1364-NEXT: s_wait_dscnt 0x0
@@ -4496,16 +4495,16 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out) {
; GFX1332-LABEL: sub_i32_constant:
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s1, exec_lo
; GFX1332-NEXT: s_mov_b32 s0, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX1332-NEXT: s_mov_b32 s0, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB8_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: s_mul_i32 s1, s1, 5
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v2, s1
; GFX1332-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX1332-NEXT: s_wait_dscnt 0x0
@@ -4531,20 +4530,20 @@ entry:
define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive) {
; GFX7LESS-LABEL: sub_i32_uniform:
; GFX7LESS: ; %bb.0: ; %entry
-; GFX7LESS-NEXT: s_load_dword s2, s[4:5], 0xb
+; GFX7LESS-NEXT: s_load_dword s6, s[4:5], 0xb
; GFX7LESS-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX7LESS-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX7LESS-NEXT: s_mov_b64 s[0:1], exec
-; GFX7LESS-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX7LESS-NEXT: s_mov_b64 s[2:3], exec
; GFX7LESS-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX7LESS-NEXT: ; implicit-def: $vgpr1
; GFX7LESS-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX7LESS-NEXT: s_cbranch_execz .LBB9_2
; GFX7LESS-NEXT: ; %bb.1:
+; GFX7LESS-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: s_mul_i32 s3, s2, s3
+; GFX7LESS-NEXT: s_mul_i32 s2, s6, s2
; GFX7LESS-NEXT: v_mov_b32_e32 v1, 0
-; GFX7LESS-NEXT: v_mov_b32_e32 v2, s3
+; GFX7LESS-NEXT: v_mov_b32_e32 v2, s2
; GFX7LESS-NEXT: s_mov_b32 m0, -1
; GFX7LESS-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
@@ -4552,7 +4551,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX7LESS-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX7LESS-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GFX7LESS-NEXT: s_waitcnt lgkmcnt(0)
-; GFX7LESS-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX7LESS-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX7LESS-NEXT: v_readfirstlane_b32 s4, v1
; GFX7LESS-NEXT: s_mov_b32 s3, 0xf000
; GFX7LESS-NEXT: s_mov_b32 s2, -1
@@ -4562,20 +4561,20 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
;
; GFX8-LABEL: sub_i32_uniform:
; GFX8: ; %bb.0: ; %entry
-; GFX8-NEXT: s_load_dword s2, s[4:5], 0x2c
+; GFX8-NEXT: s_load_dword s6, s[4:5], 0x2c
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB9_2
; GFX8-NEXT: ; %bb.1:
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_mul_i32 s3, s2, s3
+; GFX8-NEXT: s_mul_i32 s2, s6, s2
; GFX8-NEXT: v_mov_b32_e32 v1, 0
-; GFX8-NEXT: v_mov_b32_e32 v2, s3
+; GFX8-NEXT: v_mov_b32_e32 v2, s2
; GFX8-NEXT: s_mov_b32 m0, -1
; GFX8-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
@@ -4583,7 +4582,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX8-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX8-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX8-NEXT: v_readfirstlane_b32 s4, v1
; GFX8-NEXT: s_mov_b32 s3, 0xf000
; GFX8-NEXT: s_mov_b32 s2, -1
@@ -4593,27 +4592,27 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
;
; GFX9-LABEL: sub_i32_uniform:
; GFX9: ; %bb.0: ; %entry
-; GFX9-NEXT: s_load_dword s2, s[4:5], 0x2c
+; GFX9-NEXT: s_load_dword s6, s[4:5], 0x2c
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB9_2
; GFX9-NEXT: ; %bb.1:
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s3, s2, s3
+; GFX9-NEXT: s_mul_i32 s2, s6, s2
; GFX9-NEXT: v_mov_b32_e32 v1, 0
-; GFX9-NEXT: v_mov_b32_e32 v2, s3
+; GFX9-NEXT: v_mov_b32_e32 v2, s2
; GFX9-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-NEXT: .LBB9_2:
; GFX9-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX9-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX9-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX9-NEXT: v_readfirstlane_b32 s4, v1
; GFX9-NEXT: s_mov_b32 s3, 0xf000
; GFX9-NEXT: s_mov_b32 s2, -1
@@ -4623,20 +4622,20 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
;
; GFX1064-LABEL: sub_i32_uniform:
; GFX1064: ; %bb.0: ; %entry
-; GFX1064-NEXT: s_load_dword s2, s[4:5], 0x2c
+; GFX1064-NEXT: s_load_dword s6, s[4:5], 0x2c
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[0:1], exec
+; GFX1064-NEXT: s_mov_b64 s[2:3], exec
; GFX1064-NEXT: ; implicit-def: $vgpr1
-; GFX1064-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1064-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB9_2
; GFX1064-NEXT: ; %bb.1:
-; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1064-NEXT: s_mul_i32 s3, s2, s3
+; GFX1064-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1064-NEXT: v_mov_b32_e32 v1, 0
-; GFX1064-NEXT: v_mov_b32_e32 v2, s3
+; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
+; GFX1064-NEXT: s_mul_i32 s2, s6, s2
+; GFX1064-NEXT: v_mov_b32_e32 v2, s2
; GFX1064-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: buffer_gl0_inv
@@ -4645,7 +4644,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX1064-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX1064-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1064-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX1064-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX1064-NEXT: v_readfirstlane_b32 s2, v1
; GFX1064-NEXT: s_mov_b32 s3, 0x31016000
; GFX1064-NEXT: v_sub_nc_u32_e32 v0, s2, v0
@@ -4657,16 +4656,16 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: s_load_dword s0, s[4:5], 0x2c
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s1, exec_lo
+; GFX1032-NEXT: s_mov_b32 s2, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s2, s1
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1032-NEXT: s_and_saveexec_b32 s1, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB9_2
; GFX1032-NEXT: ; %bb.1:
+; GFX1032-NEXT: s_bcnt1_i32_b32 s2, s2
+; GFX1032-NEXT: v_mov_b32_e32 v1, 0
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_mul_i32 s2, s0, s2
-; GFX1032-NEXT: v_mov_b32_e32 v1, 0
; GFX1032-NEXT: v_mov_b32_e32 v2, s2
; GFX1032-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
@@ -4686,22 +4685,22 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
;
; GFX1164-LABEL: sub_i32_uniform:
; GFX1164: ; %bb.0: ; %entry
-; GFX1164-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX1164-NEXT: s_load_b32 s6, s[4:5], 0x2c
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[2:3], exec
; GFX1164-NEXT: s_mov_b64 s[0:1], exec
; GFX1164-NEXT: ; implicit-def: $vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX1164-NEXT: s_mov_b64 s[0:1], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1164-NEXT: s_cbranch_execz .LBB9_2
; GFX1164-NEXT: ; %bb.1:
-; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1164-NEXT: s_mul_i32 s3, s2, s3
+; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1164-NEXT: v_mov_b32_e32 v1, 0
-; GFX1164-NEXT: v_mov_b32_e32 v2, s3
+; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
+; GFX1164-NEXT: s_mul_i32 s2, s6, s2
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1164-NEXT: v_mov_b32_e32 v2, s2
; GFX1164-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-NEXT: buffer_gl0_inv
@@ -4709,7 +4708,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX1164-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX1164-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
-; GFX1164-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX1164-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX1164-NEXT: v_readfirstlane_b32 s2, v1
; GFX1164-NEXT: s_mov_b32 s3, 0x31016000
; GFX1164-NEXT: v_sub_nc_u32_e32 v0, s2, v0
@@ -4721,14 +4720,14 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: s_load_b32 s0, s[4:5], 0x2c
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s2, exec_lo
; GFX1132-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX1132-NEXT: s_mov_b32 s1, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB9_2
; GFX1132-NEXT: ; %bb.1:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_mul_i32 s2, s0, s2
; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -4750,22 +4749,22 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
;
; GFX1364-LABEL: sub_i32_uniform:
; GFX1364: ; %bb.0: ; %entry
-; GFX1364-NEXT: s_load_b32 s2, s[4:5], 0x2c nv
+; GFX1364-NEXT: s_load_b32 s6, s[4:5], 0x2c nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[2:3], exec
; GFX1364-NEXT: s_mov_b64 s[0:1], exec
; GFX1364-NEXT: ; implicit-def: $vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX1364-NEXT: s_mov_b64 s[0:1], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1364-NEXT: s_cbranch_execz .LBB9_2
; GFX1364-NEXT: ; %bb.1:
-; GFX1364-NEXT: s_wait_kmcnt 0x0
-; GFX1364-NEXT: s_mul_i32 s3, s2, s3
+; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX1364-NEXT: v_mov_b32_e32 v1, 0
-; GFX1364-NEXT: v_mov_b32_e32 v2, s3
+; GFX1364-NEXT: s_wait_kmcnt 0x0
+; GFX1364-NEXT: s_mul_i32 s2, s6, s2
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1364-NEXT: v_mov_b32_e32 v2, s2
; GFX1364-NEXT: ds_sub_rtn_u32 v1, v1, v2
; GFX1364-NEXT: s_wait_dscnt 0x0
; GFX1364-NEXT: global_inv scope:SCOPE_SE
@@ -4773,7 +4772,7 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX1364-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX1364-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX1364-NEXT: s_wait_kmcnt 0x0
-; GFX1364-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX1364-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX1364-NEXT: v_readfirstlane_b32 s2, v1
; GFX1364-NEXT: s_mov_b32 s3, 0x31016000
; GFX1364-NEXT: v_sub_nc_u32_e32 v0, s2, v0
@@ -4785,14 +4784,14 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, i32 %subitive)
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: s_load_b32 s0, s[4:5], 0x2c nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s2, exec_lo
; GFX1332-NEXT: s_mov_b32 s1, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX1332-NEXT: s_mov_b32 s1, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB9_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mul_i32 s2, s0, s2
; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -6087,20 +6086,20 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX9-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB12_2
; GFX9-NEXT: ; %bb.1:
-; GFX9-NEXT: s_mul_i32 s6, s2, 5
-; GFX9-NEXT: s_mul_hi_u32 s3, 5, s2
-; GFX9-NEXT: s_mul_i32 s2, s2, 0
-; GFX9-NEXT: s_add_u32 s7, s3, s2
+; GFX9-NEXT: s_bcnt1_i32_b64 s3, s[2:3]
+; GFX9-NEXT: s_mul_i32 s2, s3, 5
+; GFX9-NEXT: s_mul_hi_u32 s6, 5, s3
+; GFX9-NEXT: s_mul_i32 s3, s3, 0
+; GFX9-NEXT: s_add_u32 s3, s6, s3
; GFX9-NEXT: v_mov_b32_e32 v3, 0
-; GFX9-NEXT: v_mov_b32_e32 v0, s6
-; GFX9-NEXT: v_mov_b32_e32 v1, s7
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
; GFX9-NEXT: ds_sub_rtn_u64 v[0:1], v3, v[0:1]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-NEXT: .LBB12_2:
@@ -6122,19 +6121,19 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1064-LABEL: sub_i64_constant:
; GFX1064: ; %bb.0: ; %entry
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[0:1], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX1064-NEXT: s_mov_b64 s[2:3], exec
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1064-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB12_2
; GFX1064-NEXT: ; %bb.1:
+; GFX1064-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX1064-NEXT: v_mov_b32_e32 v3, 0
; GFX1064-NEXT: s_mul_hi_u32 s3, 5, s2
; GFX1064-NEXT: s_mul_i32 s6, s2, 0
; GFX1064-NEXT: s_mul_i32 s2, s2, 5
; GFX1064-NEXT: s_add_u32 s3, s3, s6
-; GFX1064-NEXT: v_mov_b32_e32 v3, 0
; GFX1064-NEXT: v_mov_b32_e32 v0, s2
; GFX1064-NEXT: v_mov_b32_e32 v1, s3
; GFX1064-NEXT: ds_sub_rtn_u64 v[0:1], v3, v[0:1]
@@ -6159,18 +6158,18 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1032-LABEL: sub_i64_constant:
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s0, exec_lo
+; GFX1032-NEXT: s_mov_b32 s1, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v2
; GFX1032-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB12_2
; GFX1032-NEXT: ; %bb.1:
+; GFX1032-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX1032-NEXT: v_mov_b32_e32 v3, 0
; GFX1032-NEXT: s_mul_hi_u32 s3, 5, s1
; GFX1032-NEXT: s_mul_i32 s6, s1, 0
; GFX1032-NEXT: s_mul_i32 s2, s1, 5
; GFX1032-NEXT: s_add_u32 s3, s3, s6
-; GFX1032-NEXT: v_mov_b32_e32 v3, 0
; GFX1032-NEXT: v_mov_b32_e32 v0, s2
; GFX1032-NEXT: v_mov_b32_e32 v1, s3
; GFX1032-NEXT: ds_sub_rtn_u64 v[0:1], v3, v[0:1]
@@ -6195,21 +6194,20 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1164-LABEL: sub_i64_constant:
; GFX1164: ; %bb.0: ; %entry
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[2:3], exec
; GFX1164-NEXT: s_mov_b64 s[0:1], exec
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX1164-NEXT: s_mov_b64 s[0:1], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB12_2
; GFX1164-NEXT: ; %bb.1:
+; GFX1164-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX1164-NEXT: v_mov_b32_e32 v3, 0
; GFX1164-NEXT: s_mul_hi_u32 s3, 5, s2
; GFX1164-NEXT: s_mul_i32 s6, s2, 0
; GFX1164-NEXT: s_mul_i32 s2, s2, 5
; GFX1164-NEXT: s_add_u32 s3, s3, s6
-; GFX1164-NEXT: v_mov_b32_e32 v3, 0
; GFX1164-NEXT: v_mov_b32_e32 v0, s2
; GFX1164-NEXT: v_mov_b32_e32 v1, s3
; GFX1164-NEXT: ds_sub_rtn_u64 v[0:1], v3, v[0:1]
@@ -6234,14 +6232,15 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1132-LABEL: sub_i64_constant:
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s1, exec_lo
; GFX1132-NEXT: s_mov_b32 s0, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX1132-NEXT: s_mov_b32 s0, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB12_2
; GFX1132-NEXT: ; %bb.1:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: s_mul_hi_u32 s3, 5, s1
; GFX1132-NEXT: s_mul_i32 s6, s1, 0
; GFX1132-NEXT: s_mul_i32 s2, s1, 5
@@ -6270,21 +6269,20 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1364-LABEL: sub_i64_constant:
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[2:3], exec
; GFX1364-NEXT: s_mov_b64 s[0:1], exec
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX1364-NEXT: s_mov_b64 s[0:1], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1364-NEXT: s_cbranch_execz .LBB12_2
; GFX1364-NEXT: ; %bb.1:
+; GFX1364-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX1364-NEXT: v_mov_b32_e32 v3, 0
; GFX1364-NEXT: s_mul_hi_u32 s3, 5, s2
; GFX1364-NEXT: s_mul_i32 s6, s2, 0
; GFX1364-NEXT: s_mul_i32 s2, s2, 5
; GFX1364-NEXT: s_add_co_u32 s3, s3, s6
-; GFX1364-NEXT: v_mov_b32_e32 v3, 0
; GFX1364-NEXT: v_mov_b32_e32 v0, s2
; GFX1364-NEXT: v_mov_b32_e32 v1, s3
; GFX1364-NEXT: ds_sub_rtn_u64 v[0:1], v3, v[0:1]
@@ -6309,14 +6307,15 @@ define amdgpu_kernel void @sub_i64_constant(ptr addrspace(1) %out) {
; GFX1332-LABEL: sub_i64_constant:
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s1, exec_lo
; GFX1332-NEXT: s_mov_b32 s0, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX1332-NEXT: s_mov_b32 s0, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB12_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: s_mul_hi_u32 s3, 5, s1
; GFX1332-NEXT: s_mul_i32 s6, s1, 0
; GFX1332-NEXT: s_mul_i32 s2, s1, 5
@@ -6438,21 +6437,21 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, i64 %subitive)
; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[4:5], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GFX9-NEXT: s_mov_b64 s[6:7], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX9-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX9-NEXT: s_cbranch_execz .LBB13_2
; GFX9-NEXT: ; %bb.1:
+; GFX9-NEXT: s_bcnt1_i32_b64 s7, s[6:7]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s8, s2, s6
-; GFX9-NEXT: s_mul_hi_u32 s7, s2, s6
-; GFX9-NEXT: s_mul_i32 s6, s3, s6
-; GFX9-NEXT: s_add_u32 s9, s7, s6
+; GFX9-NEXT: s_mul_i32 s6, s2, s7
+; GFX9-NEXT: s_mul_hi_u32 s8, s2, s7
+; GFX9-NEXT: s_mul_i32 s7, s3, s7
+; GFX9-NEXT: s_add_u32 s7, s8, s7
; GFX9-NEXT: v_mov_b32_e32 v3, 0
-; GFX9-NEXT: v_mov_b32_e32 v0, s8
-; GFX9-NEXT: v_mov_b32_e32 v1, s9
+; GFX9-NEXT: v_mov_b32_e32 v0, s6
+; GFX9-NEXT: v_mov_b32_e32 v1, s7
; GFX9-NEXT: ds_sub_rtn_u64 v[0:1], v3, v[0:1]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-NEXT: .LBB13_2:
@@ -6476,20 +6475,20 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, i64 %subitive)
; GFX1064: ; %bb.0: ; %entry
; GFX1064-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[4:5], exec
-; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GFX1064-NEXT: s_mov_b64 s[6:7], exec
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1064-NEXT: ; implicit-def: $vgpr0_vgpr1
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v2
; GFX1064-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB13_2
; GFX1064-NEXT: ; %bb.1:
+; GFX1064-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1064-NEXT: v_mov_b32_e32 v3, 0
; GFX1064-NEXT: s_waitcnt lgkmcnt(0)
; GFX1064-NEXT: s_mul_hi_u32 s7, s2, s6
; GFX1064-NEXT: s_mul_i32 s8, s3, s6
; GFX1064-NEXT: s_mul_i32 s6, s2, s6
; GFX1064-NEXT: s_add_u32 s7, s7, s8
-; GFX1064-NEXT: v_mov_b32_e32 v3, 0
; GFX1064-NEXT: v_mov_b32_e32 v0, s6
; GFX1064-NEXT: v_mov_b32_e32 v1, s7
; GFX1064-NEXT: ds_sub_rtn_u64 v[0:1], v3, v[0:1]
@@ -6514,19 +6513,19 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, i64 %subitive)
; GFX1032: ; %bb.0: ; %entry
; GFX1032-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s4, exec_lo
+; GFX1032-NEXT: s_mov_b32 s5, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s4
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v2
; GFX1032-NEXT: s_and_saveexec_b32 s4, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB13_2
; GFX1032-NEXT: ; %bb.1:
+; GFX1032-NEXT: s_bcnt1_i32_b32 s5, s5
+; GFX1032-NEXT: v_mov_b32_e32 v3, 0
; GFX1032-NEXT: s_waitcnt lgkmcnt(0)
; GFX1032-NEXT: s_mul_hi_u32 s7, s2, s5
; GFX1032-NEXT: s_mul_i32 s8, s3, s5
; GFX1032-NEXT: s_mul_i32 s6, s2, s5
; GFX1032-NEXT: s_add_u32 s7, s7, s8
-; GFX1032-NEXT: v_mov_b32_e32 v3, 0
; GFX1032-NEXT: v_mov_b32_e32 v0, s6
; GFX1032-NEXT: v_mov_b32_e32 v1, s7
; GFX1032-NEXT: ds_sub_rtn_u64 v[0:1], v3, v[0:1]
@@ -6551,22 +6550,21 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, i64 %subitive)
; GFX1164: ; %bb.0: ; %entry
; GFX1164-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[6:7], exec
; GFX1164-NEXT: s_mov_b64 s[4:5], exec
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1164-NEXT: s_mov_b64 s[4:5], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1164-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1164-NEXT: s_cbranch_execz .LBB13_2
; GFX1164-NEXT: ; %bb.1:
+; GFX1164-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1164-NEXT: v_mov_b32_e32 v3, 0
; GFX1164-NEXT: s_waitcnt lgkmcnt(0)
; GFX1164-NEXT: s_mul_hi_u32 s7, s2, s6
; GFX1164-NEXT: s_mul_i32 s8, s3, s6
; GFX1164-NEXT: s_mul_i32 s6, s2, s6
; GFX1164-NEXT: s_add_u32 s7, s7, s8
-; GFX1164-NEXT: v_mov_b32_e32 v3, 0
; GFX1164-NEXT: v_mov_b32_e32 v0, s6
; GFX1164-NEXT: v_mov_b32_e32 v1, s7
; GFX1164-NEXT: ds_sub_rtn_u64 v[0:1], v3, v[0:1]
@@ -6591,14 +6589,14 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, i64 %subitive)
; GFX1132: ; %bb.0: ; %entry
; GFX1132-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s5, exec_lo
; GFX1132-NEXT: s_mov_b32 s4, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1132-NEXT: s_mov_b32 s4, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1132-NEXT: s_cbranch_execz .LBB13_2
; GFX1132-NEXT: ; %bb.1:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s5, s5
; GFX1132-NEXT: s_waitcnt lgkmcnt(0)
; GFX1132-NEXT: s_mul_hi_u32 s7, s2, s5
; GFX1132-NEXT: s_mul_i32 s8, s3, s5
@@ -6628,22 +6626,21 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, i64 %subitive)
; GFX1364: ; %bb.0: ; %entry
; GFX1364-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[6:7], exec
; GFX1364-NEXT: s_mov_b64 s[4:5], exec
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
-; GFX1364-NEXT: s_mov_b64 s[4:5], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v2, exec_hi, v0
; GFX1364-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1364-NEXT: s_cbranch_execz .LBB13_2
; GFX1364-NEXT: ; %bb.1:
+; GFX1364-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
+; GFX1364-NEXT: v_mov_b32_e32 v3, 0
; GFX1364-NEXT: s_wait_kmcnt 0x0
; GFX1364-NEXT: s_mul_hi_u32 s7, s2, s6
; GFX1364-NEXT: s_mul_i32 s8, s3, s6
; GFX1364-NEXT: s_mul_i32 s6, s2, s6
; GFX1364-NEXT: s_add_co_u32 s7, s7, s8
-; GFX1364-NEXT: v_mov_b32_e32 v3, 0
; GFX1364-NEXT: v_mov_b32_e32 v0, s6
; GFX1364-NEXT: v_mov_b32_e32 v1, s7
; GFX1364-NEXT: ds_sub_rtn_u64 v[0:1], v3, v[0:1]
@@ -6668,14 +6665,14 @@ define amdgpu_kernel void @sub_i64_uniform(ptr addrspace(1) %out, i64 %subitive)
; GFX1332: ; %bb.0: ; %entry
; GFX1332-NEXT: s_load_b128 s[0:3], s[4:5], 0x24 nv
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v2, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s5, exec_lo
; GFX1332-NEXT: s_mov_b32 s4, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr0_vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s4
-; GFX1332-NEXT: s_mov_b32 s4, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v2
; GFX1332-NEXT: s_cbranch_execz .LBB13_2
; GFX1332-NEXT: ; %bb.1:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s5, s5
; GFX1332-NEXT: s_wait_kmcnt 0x0
; GFX1332-NEXT: s_mul_hi_u32 s7, s2, s5
; GFX1332-NEXT: s_mul_i32 s8, s3, s5
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_mul_one.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_mul_one.ll
index 7cdf2562336a9a..1f2c437e66fe5e 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_mul_one.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_mul_one.ll
@@ -37,11 +37,11 @@ define amdgpu_cs void @atomic_add_i32_constant_1(<4 x i32> inreg %arg) {
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: s_and_saveexec_b64 s[6:7], vcc
; GCN-NEXT: s_cbranch_execz .LBB0_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: v_mov_b32_e32 v0, s4
; GCN-NEXT: v_mov_b32_e32 v1, 0
; GCN-NEXT: buffer_atomic_add v0, v1, s[0:3], 0 idxen
@@ -111,13 +111,13 @@ define amdgpu_cs void @atomic_add_i64_constant_1(<4 x i32> inreg %arg) {
; GCN: ; %bb.0: ; %.entry
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: s_mov_b32 s5, 0
-; GCN-NEXT: s_bcnt1_i32_b64 s4, s[6:7]
+; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
-; GCN-NEXT: s_and_saveexec_b64 s[6:7], vcc
+; GCN-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GCN-NEXT: s_cbranch_execz .LBB2_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s4, s[6:7]
; GCN-NEXT: v_mov_b32_e32 v0, s4
; GCN-NEXT: v_mov_b32_e32 v1, s5
; GCN-NEXT: v_mov_b32_e32 v2, 0
@@ -194,13 +194,13 @@ define amdgpu_cs void @atomic_add_and_format(<4 x i32> inreg %arg) {
; GCN: ; %bb.0: ; %.entry
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: ; implicit-def: $vgpr1
; GCN-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GCN-NEXT: s_cbranch_execz .LBB4_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GCN-NEXT: v_mov_b32_e32 v1, s6
; GCN-NEXT: v_mov_b32_e32 v2, 0
; GCN-NEXT: buffer_atomic_add v1, v2, s[0:3], 0 idxen glc
@@ -246,11 +246,11 @@ define amdgpu_cs void @atomic_sub_i32_constant_1(<4 x i32> inreg %arg) {
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: s_and_saveexec_b64 s[6:7], vcc
; GCN-NEXT: s_cbranch_execz .LBB5_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: v_mov_b32_e32 v0, s4
; GCN-NEXT: v_mov_b32_e32 v1, 0
; GCN-NEXT: buffer_atomic_sub v0, v1, s[0:3], 0 idxen
@@ -320,13 +320,13 @@ define amdgpu_cs void @atomic_sub_i64_constant_1(<4 x i32> inreg %arg) {
; GCN: ; %bb.0: ; %.entry
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: s_mov_b32 s5, 0
-; GCN-NEXT: s_bcnt1_i32_b64 s4, s[6:7]
+; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
-; GCN-NEXT: s_and_saveexec_b64 s[6:7], vcc
+; GCN-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GCN-NEXT: s_cbranch_execz .LBB7_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s4, s[6:7]
; GCN-NEXT: v_mov_b32_e32 v0, s4
; GCN-NEXT: v_mov_b32_e32 v1, s5
; GCN-NEXT: v_mov_b32_e32 v2, 0
@@ -403,13 +403,13 @@ define amdgpu_cs void @atomic_sub_and_format(<4 x i32> inreg %arg) {
; GCN: ; %bb.0: ; %.entry
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: ; implicit-def: $vgpr1
; GCN-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GCN-NEXT: s_cbranch_execz .LBB9_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GCN-NEXT: v_mov_b32_e32 v1, s6
; GCN-NEXT: v_mov_b32_e32 v2, 0
; GCN-NEXT: buffer_atomic_sub v1, v2, s[0:3], 0 idxen glc
@@ -455,11 +455,11 @@ define amdgpu_cs void @atomic_xor_i32_constant_1(<4 x i32> inreg %arg) {
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: s_and_saveexec_b64 s[6:7], vcc
; GCN-NEXT: s_cbranch_execz .LBB10_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: s_and_b32 s4, s4, 1
; GCN-NEXT: v_mov_b32_e32 v0, s4
; GCN-NEXT: v_mov_b32_e32 v1, 0
@@ -568,13 +568,13 @@ define amdgpu_cs void @atomic_xor_i64_constant_1(<4 x i32> inreg %arg) {
; GCN: ; %bb.0: ; %.entry
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: s_mov_b32 s5, 0
-; GCN-NEXT: s_bcnt1_i32_b64 s4, s[6:7]
+; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
-; GCN-NEXT: s_and_saveexec_b64 s[6:7], vcc
+; GCN-NEXT: s_and_saveexec_b64 s[8:9], vcc
; GCN-NEXT: s_cbranch_execz .LBB13_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s4, s[6:7]
; GCN-NEXT: s_and_b32 s4, s4, 1
; GCN-NEXT: v_mov_b32_e32 v0, s4
; GCN-NEXT: v_mov_b32_e32 v1, s5
@@ -615,13 +615,13 @@ define amdgpu_cs void @atomic_xor_and_format(<4 x i32> inreg %arg) {
; GCN: ; %bb.0: ; %.entry
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: ; implicit-def: $vgpr1
; GCN-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GCN-NEXT: s_cbranch_execz .LBB14_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GCN-NEXT: s_and_b32 s6, s6, 1
; GCN-NEXT: v_mov_b32_e32 v1, s6
; GCN-NEXT: v_mov_b32_e32 v2, 0
@@ -669,11 +669,11 @@ define amdgpu_cs void @atomic_ptr_add(ptr addrspace(8) inreg %arg) {
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: s_and_saveexec_b64 s[6:7], vcc
; GCN-NEXT: s_cbranch_execz .LBB15_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: v_mov_b32_e32 v0, s4
; GCN-NEXT: v_mov_b32_e32 v1, 0
; GCN-NEXT: buffer_atomic_add v0, v1, s[0:3], 0 idxen
@@ -713,13 +713,13 @@ define amdgpu_cs void @atomic_ptr_add_and_format(ptr addrspace(8) inreg %arg) {
; GCN: ; %bb.0: ; %.entry
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: ; implicit-def: $vgpr1
; GCN-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GCN-NEXT: s_cbranch_execz .LBB16_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GCN-NEXT: v_mov_b32_e32 v1, s6
; GCN-NEXT: v_mov_b32_e32 v2, 0
; GCN-NEXT: buffer_atomic_add v1, v2, s[0:3], 0 idxen glc
@@ -767,11 +767,11 @@ define amdgpu_cs void @atomic_ptr_sub(ptr addrspace(8) inreg %arg) {
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: s_and_saveexec_b64 s[6:7], vcc
; GCN-NEXT: s_cbranch_execz .LBB17_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: v_mov_b32_e32 v0, s4
; GCN-NEXT: v_mov_b32_e32 v1, 0
; GCN-NEXT: buffer_atomic_sub v0, v1, s[0:3], 0 idxen
@@ -811,13 +811,13 @@ define amdgpu_cs void @atomic_ptr_sub_and_format(ptr addrspace(8) inreg %arg) {
; GCN: ; %bb.0: ; %.entry
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: ; implicit-def: $vgpr1
; GCN-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GCN-NEXT: s_cbranch_execz .LBB18_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GCN-NEXT: v_mov_b32_e32 v1, s6
; GCN-NEXT: v_mov_b32_e32 v2, 0
; GCN-NEXT: buffer_atomic_sub v1, v2, s[0:3], 0 idxen glc
@@ -865,11 +865,11 @@ define amdgpu_cs void @atomic_ptr_xor(ptr addrspace(8) inreg %arg) {
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: s_and_saveexec_b64 s[6:7], vcc
; GCN-NEXT: s_cbranch_execz .LBB19_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s4, s[4:5]
; GCN-NEXT: s_and_b32 s4, s4, 1
; GCN-NEXT: v_mov_b32_e32 v0, s4
; GCN-NEXT: v_mov_b32_e32 v1, 0
@@ -911,13 +911,13 @@ define amdgpu_cs void @atomic_ptr_xor_and_format(ptr addrspace(8) inreg %arg) {
; GCN: ; %bb.0: ; %.entry
; GCN-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GCN-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GCN-NEXT: s_mov_b64 s[4:5], exec
-; GCN-NEXT: s_bcnt1_i32_b64 s6, s[4:5]
+; GCN-NEXT: s_mov_b64 s[6:7], exec
; GCN-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GCN-NEXT: ; implicit-def: $vgpr1
; GCN-NEXT: s_and_saveexec_b64 s[4:5], vcc
; GCN-NEXT: s_cbranch_execz .LBB20_2
; GCN-NEXT: ; %bb.1:
+; GCN-NEXT: s_bcnt1_i32_b64 s6, s[6:7]
; GCN-NEXT: s_and_b32 s6, s6, 1
; GCN-NEXT: v_mov_b32_e32 v1, s6
; GCN-NEXT: v_mov_b32_e32 v2, 0
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_pixelshader.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_pixelshader.ll
index d3d7e8131a0c19..1f8ccf2e486a26 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_pixelshader.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_pixelshader.ll
@@ -25,13 +25,13 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX7-NEXT: ; %bb.1:
; GFX7-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX7-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX7-NEXT: s_mov_b64 s[10:11], exec
-; GFX7-NEXT: s_bcnt1_i32_b64 s12, s[10:11]
+; GFX7-NEXT: s_mov_b64 s[12:13], exec
; GFX7-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX7-NEXT: ; implicit-def: $vgpr1
; GFX7-NEXT: s_and_saveexec_b64 s[10:11], vcc
; GFX7-NEXT: s_cbranch_execz .LBB0_3
; GFX7-NEXT: ; %bb.2:
+; GFX7-NEXT: s_bcnt1_i32_b64 s12, s[12:13]
; GFX7-NEXT: s_mul_i32 s12, s12, 5
; GFX7-NEXT: v_mov_b32_e32 v1, s12
; GFX7-NEXT: buffer_atomic_add v1, off, s[4:7], 0 glc
@@ -62,13 +62,13 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX89-NEXT: ; %bb.1:
; GFX89-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX89-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX89-NEXT: s_mov_b64 s[10:11], exec
-; GFX89-NEXT: s_bcnt1_i32_b64 s12, s[10:11]
+; GFX89-NEXT: s_mov_b64 s[12:13], exec
; GFX89-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX89-NEXT: ; implicit-def: $vgpr1
; GFX89-NEXT: s_and_saveexec_b64 s[10:11], vcc
; GFX89-NEXT: s_cbranch_execz .LBB0_3
; GFX89-NEXT: ; %bb.2:
+; GFX89-NEXT: s_bcnt1_i32_b64 s12, s[12:13]
; GFX89-NEXT: s_mul_i32 s12, s12, 5
; GFX89-NEXT: v_mov_b32_e32 v1, s12
; GFX89-NEXT: buffer_atomic_add v1, off, s[4:7], 0 glc
@@ -98,14 +98,14 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1064-NEXT: s_cbranch_execz .LBB0_4
; GFX1064-NEXT: ; %bb.1:
; GFX1064-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1064-NEXT: s_mov_b64 s[10:11], exec
+; GFX1064-NEXT: s_mov_b64 s[12:13], exec
; GFX1064-NEXT: ; implicit-def: $vgpr1
-; GFX1064-NEXT: s_bcnt1_i32_b64 s12, s[10:11]
; GFX1064-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX1064-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX1064-NEXT: s_and_saveexec_b64 s[10:11], vcc
; GFX1064-NEXT: s_cbranch_execz .LBB0_3
; GFX1064-NEXT: ; %bb.2:
+; GFX1064-NEXT: s_bcnt1_i32_b64 s12, s[12:13]
; GFX1064-NEXT: s_mul_i32 s12, s12, 5
; GFX1064-NEXT: v_mov_b32_e32 v1, s12
; GFX1064-NEXT: buffer_atomic_add v1, off, s[4:7], 0 glc
@@ -136,13 +136,13 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1032-NEXT: s_cbranch_execz .LBB0_4
; GFX1032-NEXT: ; %bb.1:
; GFX1032-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX1032-NEXT: s_mov_b32 s9, exec_lo
+; GFX1032-NEXT: s_mov_b32 s10, exec_lo
; GFX1032-NEXT: ; implicit-def: $vgpr1
-; GFX1032-NEXT: s_bcnt1_i32_b32 s10, s9
; GFX1032-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX1032-NEXT: s_and_saveexec_b32 s9, vcc_lo
; GFX1032-NEXT: s_cbranch_execz .LBB0_3
; GFX1032-NEXT: ; %bb.2:
+; GFX1032-NEXT: s_bcnt1_i32_b32 s10, s10
; GFX1032-NEXT: s_mul_i32 s10, s10, 5
; GFX1032-NEXT: v_mov_b32_e32 v1, s10
; GFX1032-NEXT: buffer_atomic_add v1, off, s[4:7], 0 glc
@@ -174,18 +174,17 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1164-NEXT: s_cbranch_execz .LBB0_4
; GFX1164-NEXT: ; %bb.1:
; GFX1164-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1164-NEXT: s_mov_b64 s[12:13], exec
; GFX1164-NEXT: s_mov_b64 s[10:11], exec
; GFX1164-NEXT: ; implicit-def: $vgpr1
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1164-NEXT: s_bcnt1_i32_b64 s12, s[10:11]
-; GFX1164-NEXT: s_mov_b64 s[10:11], exec
+; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1164-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1164-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1164-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1164-NEXT: s_cbranch_execz .LBB0_3
; GFX1164-NEXT: ; %bb.2:
+; GFX1164-NEXT: s_bcnt1_i32_b64 s12, s[12:13]
+; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1164-NEXT: s_mul_i32 s12, s12, 5
-; GFX1164-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1164-NEXT: v_mov_b32_e32 v1, s12
; GFX1164-NEXT: buffer_atomic_add_u32 v1, off, s[4:7], 0 glc
; GFX1164-NEXT: .LBB0_3:
@@ -218,16 +217,16 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1132-NEXT: s_cbranch_execz .LBB0_4
; GFX1132-NEXT: ; %bb.1:
; GFX1132-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1132-NEXT: s_mov_b32 s10, exec_lo
; GFX1132-NEXT: s_mov_b32 s9, exec_lo
; GFX1132-NEXT: ; implicit-def: $vgpr1
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1132-NEXT: s_bcnt1_i32_b32 s10, s9
-; GFX1132-NEXT: s_mov_b32 s9, exec_lo
+; GFX1132-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1132-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1132-NEXT: s_cbranch_execz .LBB0_3
; GFX1132-NEXT: ; %bb.2:
+; GFX1132-NEXT: s_bcnt1_i32_b32 s10, s10
+; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1132-NEXT: s_mul_i32 s10, s10, 5
-; GFX1132-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1132-NEXT: v_mov_b32_e32 v1, s10
; GFX1132-NEXT: buffer_atomic_add_u32 v1, off, s[4:7], 0 glc
; GFX1132-NEXT: .LBB0_3:
@@ -260,18 +259,17 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1364-NEXT: s_cbranch_execz .LBB0_4
; GFX1364-NEXT: ; %bb.1:
; GFX1364-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1364-NEXT: s_mov_b64 s[12:13], exec
; GFX1364-NEXT: s_mov_b64 s[10:11], exec
; GFX1364-NEXT: ; implicit-def: $vgpr1
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1364-NEXT: s_bcnt1_i32_b64 s12, s[10:11]
-; GFX1364-NEXT: s_mov_b64 s[10:11], exec
+; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX1364-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX1364-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1364-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1364-NEXT: s_cbranch_execz .LBB0_3
; GFX1364-NEXT: ; %bb.2:
+; GFX1364-NEXT: s_bcnt1_i32_b64 s12, s[12:13]
+; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1364-NEXT: s_mul_i32 s12, s12, 5
-; GFX1364-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1364-NEXT: v_mov_b32_e32 v1, s12
; GFX1364-NEXT: buffer_atomic_add_u32 v1, off, s[4:7], null th:TH_ATOMIC_RETURN
; GFX1364-NEXT: .LBB0_3:
@@ -304,16 +302,16 @@ define amdgpu_ps void @add_i32_constant(ptr addrspace(8) inreg %out, ptr addrspa
; GFX1332-NEXT: s_cbranch_execz .LBB0_4
; GFX1332-NEXT: ; %bb.1:
; GFX1332-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX1332-NEXT: s_mov_b32 s10, exec_lo
; GFX1332-NEXT: s_mov_b32 s9, exec_lo
; GFX1332-NEXT: ; implicit-def: $vgpr1
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1332-NEXT: s_bcnt1_i32_b32 s10, s9
-; GFX1332-NEXT: s_mov_b32 s9, exec_lo
+; GFX1332-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX1332-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX1332-NEXT: s_cbranch_execz .LBB0_3
; GFX1332-NEXT: ; %bb.2:
+; GFX1332-NEXT: s_bcnt1_i32_b32 s10, s10
+; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1332-NEXT: s_mul_i32 s10, s10, 5
-; GFX1332-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1332-NEXT: v_mov_b32_e32 v1, s10
; GFX1332-NEXT: buffer_atomic_add_u32 v1, off, s[4:7], null th:TH_ATOMIC_RETURN
; GFX1332-NEXT: .LBB0_3:
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
index f09555fde5d32a..e5a5730005f88e 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
@@ -22,14 +22,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX6: ; %bb.0: ; %entry
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB0_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_mul_i32 s2, s2, 5
; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
@@ -50,14 +50,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB0_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_mul_i32 s2, s2, 5
; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
@@ -78,14 +78,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB0_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_mul_i32 s2, s2, 5
; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
@@ -104,15 +104,15 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W64-LABEL: add_i32_constant:
; GFX10W64: ; %bb.0: ; %entry
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB0_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: s_mul_i32 s2, s2, 5
; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
@@ -132,14 +132,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W32-LABEL: add_i32_constant:
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB0_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s1
; GFX10W32-NEXT: s_mul_i32 s1, s1, 5
; GFX10W32-NEXT: v_mov_b32_e32 v1, s1
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
@@ -159,19 +159,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-LABEL: add_i32_constant:
; GFX11W64: ; %bb.0: ; %entry
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB0_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], 0 glc
@@ -190,17 +189,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-LABEL: add_i32_constant:
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB0_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W32-NEXT: v_mov_b32_e32 v1, s1
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], 0 glc
@@ -219,19 +218,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-LABEL: add_i32_constant:
; GFX12W64: ; %bb.0: ; %entry
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB0_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: s_wait_kmcnt 0x0
; GFX12W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -251,17 +249,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-LABEL: add_i32_constant:
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB0_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W32-NEXT: v_mov_b32_e32 v1, s1
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -280,19 +278,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-LABEL: add_i32_constant:
; GFX13W64: ; %bb.0: ; %entry
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB0_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: s_wait_kmcnt 0x0
; GFX13W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -311,17 +308,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-LABEL: add_i32_constant:
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB0_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W32-NEXT: v_mov_b32_e32 v1, s1
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -345,26 +342,26 @@ entry:
define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(8) %inout, i32 %additive) {
; GFX6-LABEL: add_i32_uniform:
; GFX6: ; %bb.0: ; %entry
-; GFX6-NEXT: s_load_dword s2, s[4:5], 0x11
+; GFX6-NEXT: s_load_dword s6, s[4:5], 0x11
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB1_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_mul_i32 s3, s2, s3
-; GFX6-NEXT: v_mov_b32_e32 v1, s3
+; GFX6-NEXT: s_mul_i32 s2, s6, s2
+; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX6-NEXT: .LBB1_2:
; GFX6-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX6-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX6-NEXT: s_waitcnt vmcnt(0)
; GFX6-NEXT: v_readfirstlane_b32 s4, v1
; GFX6-NEXT: s_mov_b32 s3, 0xf000
@@ -375,26 +372,26 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX8-LABEL: add_i32_uniform:
; GFX8: ; %bb.0: ; %entry
-; GFX8-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX8-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB1_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_mul_i32 s3, s2, s3
-; GFX8-NEXT: v_mov_b32_e32 v1, s3
+; GFX8-NEXT: s_mul_i32 s2, s6, s2
+; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX8-NEXT: .LBB1_2:
; GFX8-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX8-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: v_readfirstlane_b32 s2, v1
; GFX8-NEXT: v_add_u32_e32 v2, vcc, s2, v0
@@ -405,26 +402,26 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX9-LABEL: add_i32_uniform:
; GFX9: ; %bb.0: ; %entry
-; GFX9-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX9-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB1_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s3, s2, s3
-; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_mul_i32 s2, s6, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX9-NEXT: .LBB1_2:
; GFX9-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX9-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX9-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: v_readfirstlane_b32 s2, v1
; GFX9-NEXT: v_mov_b32_e32 v2, 0
@@ -434,30 +431,29 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX10W64-LABEL: add_i32_uniform:
; GFX10W64: ; %bb.0: ; %entry
-; GFX10W64-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX10W64-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB1_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: s_mul_i32 s3, s2, s3
-; GFX10W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX10W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
; GFX10W64-NEXT: .LBB1_2:
; GFX10W64-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)
; GFX10W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX10W64-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX10W64-NEXT: s_waitcnt vmcnt(0)
-; GFX10W64-NEXT: s_mov_b32 null, 0
-; GFX10W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX10W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: v_mad_u64_u32 v[0:1], s[2:3], s2, v0, s[4:5]
+; GFX10W64-NEXT: v_mad_u64_u32 v[0:1], s[2:3], s6, v0, s[2:3]
; GFX10W64-NEXT: v_mov_b32_e32 v1, 0
; GFX10W64-NEXT: global_store_dword v1, v0, s[0:1]
; GFX10W64-NEXT: s_endpgm
@@ -466,14 +462,14 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: s_load_dword s0, s[4:5], 0x44
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s2, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s1
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s1, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB1_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX10W32-NEXT: s_mul_i32 s2, s0, s2
; GFX10W32-NEXT: v_mov_b32_e32 v1, s2
@@ -493,32 +489,31 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX11W64-LABEL: add_i32_uniform:
; GFX11W64: ; %bb.0: ; %entry
-; GFX11W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX11W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB1_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX11W64-NEXT: s_mul_i32 s2, s6, s2
; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], 0 glc
; GFX11W64-NEXT: .LBB1_2:
; GFX11W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX11W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11W64-NEXT: s_waitcnt vmcnt(0)
-; GFX11W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX11W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11W64-NEXT: v_mad_u64_u32 v[1:2], null, s2, v0, s[4:5]
+; GFX11W64-NEXT: v_mad_u64_u32 v[1:2], null, s6, v0, s[2:3]
; GFX11W64-NEXT: v_mov_b32_e32 v0, 0
; GFX11W64-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11W64-NEXT: s_endpgm
@@ -527,15 +522,15 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s2, exec_lo
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB1_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: s_mul_i32 s2, s0, s2
; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -555,32 +550,32 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX12W64-LABEL: add_i32_uniform:
; GFX12W64: ; %bb.0: ; %entry
-; GFX12W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX12W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB1_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX12W64-NEXT: s_wait_kmcnt 0x0
-; GFX12W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX12W64-NEXT: s_mul_i32 s2, s6, s2
; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
; GFX12W64-NEXT: .LBB1_2:
; GFX12W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX12W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX12W64-NEXT: s_wait_loadcnt 0x0
-; GFX12W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX12W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX12W64-NEXT: s_wait_kmcnt 0x0
+; GFX12W64-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s2, v0, s[4:5]
+; GFX12W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s6, v0, s[2:3]
; GFX12W64-NEXT: v_mov_b32_e32 v1, 0
; GFX12W64-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX12W64-NEXT: s_endpgm
@@ -589,15 +584,15 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s2, exec_lo
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB1_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: s_mul_i32 s2, s0, s2
; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -617,32 +612,31 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX13W64-LABEL: add_i32_uniform:
; GFX13W64: ; %bb.0: ; %entry
-; GFX13W64-NEXT: s_load_b32 s2, s[4:5], 0x44 nv
+; GFX13W64-NEXT: s_load_b32 s6, s[4:5], 0x44 nv
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB1_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX13W64-NEXT: s_wait_kmcnt 0x0
-; GFX13W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX13W64-NEXT: s_mul_i32 s2, s6, s2
; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX13W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
; GFX13W64-NEXT: .LBB1_2:
; GFX13W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX13W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13W64-NEXT: s_wait_loadcnt 0x0
-; GFX13W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX13W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX13W64-NEXT: s_wait_kmcnt 0x0
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX13W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s2, v0, s[4:5]
+; GFX13W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s6, v0, s[2:3]
; GFX13W64-NEXT: v_mov_b32_e32 v1, 0
; GFX13W64-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13W64-NEXT: s_endpgm
@@ -651,15 +645,15 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: s_load_b32 s0, s[4:5], 0x44 nv
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s2, exec_lo
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB1_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: s_mul_i32 s2, s0, s2
; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -1288,14 +1282,14 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX6: ; %bb.0: ; %entry
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB4_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_mul_i32 s2, s2, 5
; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
@@ -1317,14 +1311,14 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB4_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_mul_i32 s2, s2, 5
; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
@@ -1346,14 +1340,14 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB4_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_mul_i32 s2, s2, 5
; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
@@ -1373,15 +1367,15 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W64-LABEL: sub_i32_constant:
; GFX10W64: ; %bb.0: ; %entry
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB4_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: s_mul_i32 s2, s2, 5
; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
@@ -1402,14 +1396,14 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W32-LABEL: sub_i32_constant:
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB4_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s1
; GFX10W32-NEXT: s_mul_i32 s1, s1, 5
; GFX10W32-NEXT: v_mov_b32_e32 v1, s1
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
@@ -1430,19 +1424,18 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-LABEL: sub_i32_constant:
; GFX11W64: ; %bb.0: ; %entry
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB4_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], 0 glc
@@ -1462,17 +1455,17 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-LABEL: sub_i32_constant:
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB4_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W32-NEXT: v_mov_b32_e32 v1, s1
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], 0 glc
@@ -1492,19 +1485,18 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-LABEL: sub_i32_constant:
; GFX12W64: ; %bb.0: ; %entry
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB4_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: s_wait_kmcnt 0x0
; GFX12W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -1525,17 +1517,17 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-LABEL: sub_i32_constant:
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB4_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W32-NEXT: v_mov_b32_e32 v1, s1
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -1555,19 +1547,18 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-LABEL: sub_i32_constant:
; GFX13W64: ; %bb.0: ; %entry
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB4_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
+; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W64-NEXT: s_mul_i32 s2, s2, 5
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: s_wait_kmcnt 0x0
; GFX13W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -1587,17 +1578,17 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-LABEL: sub_i32_constant:
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB4_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W32-NEXT: v_mov_b32_e32 v1, s1
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
@@ -1621,26 +1612,26 @@ entry:
define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(8) %inout, i32 %subitive) {
; GFX6-LABEL: sub_i32_uniform:
; GFX6: ; %bb.0: ; %entry
-; GFX6-NEXT: s_load_dword s2, s[4:5], 0x11
+; GFX6-NEXT: s_load_dword s6, s[4:5], 0x11
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB5_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_mul_i32 s3, s2, s3
-; GFX6-NEXT: v_mov_b32_e32 v1, s3
+; GFX6-NEXT: s_mul_i32 s2, s6, s2
+; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: buffer_atomic_sub v1, off, s[8:11], 0 glc
; GFX6-NEXT: .LBB5_2:
; GFX6-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX6-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX6-NEXT: s_waitcnt vmcnt(0)
; GFX6-NEXT: v_readfirstlane_b32 s4, v1
; GFX6-NEXT: s_mov_b32 s3, 0xf000
@@ -1651,26 +1642,26 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX8-LABEL: sub_i32_uniform:
; GFX8: ; %bb.0: ; %entry
-; GFX8-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX8-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB5_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_mul_i32 s3, s2, s3
-; GFX8-NEXT: v_mov_b32_e32 v1, s3
+; GFX8-NEXT: s_mul_i32 s2, s6, s2
+; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: buffer_atomic_sub v1, off, s[8:11], 0 glc
; GFX8-NEXT: .LBB5_2:
; GFX8-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX8-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: v_readfirstlane_b32 s2, v1
; GFX8-NEXT: v_sub_u32_e32 v2, vcc, s2, v0
@@ -1681,26 +1672,26 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX9-LABEL: sub_i32_uniform:
; GFX9: ; %bb.0: ; %entry
-; GFX9-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX9-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB5_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s3, s2, s3
-; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_mul_i32 s2, s6, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: buffer_atomic_sub v1, off, s[8:11], 0 glc
; GFX9-NEXT: .LBB5_2:
; GFX9-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX9-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX9-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: v_readfirstlane_b32 s2, v1
; GFX9-NEXT: v_mov_b32_e32 v2, 0
@@ -1710,27 +1701,27 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX10W64-LABEL: sub_i32_uniform:
; GFX10W64: ; %bb.0: ; %entry
-; GFX10W64-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX10W64-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB5_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: s_mul_i32 s3, s2, s3
-; GFX10W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX10W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: buffer_atomic_sub v1, off, s[8:11], 0 glc
; GFX10W64-NEXT: .LBB5_2:
; GFX10W64-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)
; GFX10W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX10W64-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX10W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX10W64-NEXT: s_waitcnt vmcnt(0)
; GFX10W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX10W64-NEXT: v_mov_b32_e32 v1, 0
@@ -1742,14 +1733,14 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: s_load_dword s0, s[4:5], 0x44
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s2, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s1
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s1, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB5_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX10W32-NEXT: s_mul_i32 s2, s0, s2
; GFX10W32-NEXT: v_mov_b32_e32 v1, s2
@@ -1769,29 +1760,28 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX11W64-LABEL: sub_i32_uniform:
; GFX11W64: ; %bb.0: ; %entry
-; GFX11W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX11W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB5_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX11W64-NEXT: s_mul_i32 s2, s6, s2
; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], 0 glc
; GFX11W64-NEXT: .LBB5_2:
; GFX11W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX11W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX11W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX11W64-NEXT: s_waitcnt vmcnt(0)
; GFX11W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX11W64-NEXT: v_mov_b32_e32 v1, 0
@@ -1804,15 +1794,15 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s2, exec_lo
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB5_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: s_mul_i32 s2, s0, s2
; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -1833,29 +1823,28 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX12W64-LABEL: sub_i32_uniform:
; GFX12W64: ; %bb.0: ; %entry
-; GFX12W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX12W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB5_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX12W64-NEXT: s_wait_kmcnt 0x0
-; GFX12W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX12W64-NEXT: s_mul_i32 s2, s6, s2
; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
; GFX12W64-NEXT: .LBB5_2:
; GFX12W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX12W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX12W64-NEXT: s_wait_kmcnt 0x0
-; GFX12W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX12W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX12W64-NEXT: s_wait_loadcnt 0x0
; GFX12W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX12W64-NEXT: v_mov_b32_e32 v1, 0
@@ -1869,15 +1858,15 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s2, exec_lo
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB5_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: s_mul_i32 s2, s0, s2
; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -1899,29 +1888,28 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX13W64-LABEL: sub_i32_uniform:
; GFX13W64: ; %bb.0: ; %entry
-; GFX13W64-NEXT: s_load_b32 s2, s[4:5], 0x44 nv
+; GFX13W64-NEXT: s_load_b32 s6, s[4:5], 0x44 nv
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB5_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX13W64-NEXT: s_wait_kmcnt 0x0
-; GFX13W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX13W64-NEXT: s_mul_i32 s2, s6, s2
; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX13W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: buffer_atomic_sub_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
; GFX13W64-NEXT: .LBB5_2:
; GFX13W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX13W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13W64-NEXT: s_wait_kmcnt 0x0
-; GFX13W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX13W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX13W64-NEXT: s_wait_loadcnt 0x0
; GFX13W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX13W64-NEXT: v_mov_b32_e32 v1, 0
@@ -1934,15 +1922,15 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: s_load_b32 s0, s[4:5], 0x44 nv
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s2, exec_lo
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB5_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: s_mul_i32 s2, s0, s2
; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
index 83d8aa12d54aad..0334db94087d54 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
@@ -22,14 +22,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX6: ; %bb.0: ; %entry
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB0_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_mul_i32 s2, s2, 5
; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: v_mov_b32_e32 v2, 0
@@ -51,14 +51,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB0_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_mul_i32 s2, s2, 5
; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: v_mov_b32_e32 v2, 0
@@ -80,14 +80,14 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB0_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_mul_i32 s2, s2, 5
; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: v_mov_b32_e32 v2, 0
@@ -107,17 +107,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W64-LABEL: add_i32_constant:
; GFX10W64: ; %bb.0: ; %entry
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB0_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
-; GFX10W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: v_mov_b32_e32 v2, 0
+; GFX10W64-NEXT: s_mul_i32 s2, s2, 5
; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX10W64-NEXT: buffer_atomic_add v1, v2, s[8:11], 0 idxen glc
@@ -136,16 +136,16 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W32-LABEL: add_i32_constant:
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB0_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
-; GFX10W32-NEXT: s_mul_i32 s1, s1, 5
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s1
; GFX10W32-NEXT: v_mov_b32_e32 v2, 0
+; GFX10W32-NEXT: s_mul_i32 s1, s1, 5
; GFX10W32-NEXT: v_mov_b32_e32 v1, s1
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX10W32-NEXT: buffer_atomic_add v1, v2, s[8:11], 0 idxen glc
@@ -164,19 +164,19 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-LABEL: add_i32_constant:
; GFX11W64: ; %bb.0: ; %entry
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB0_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
-; GFX11W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX11W64-NEXT: v_mov_b32_e32 v2, 0
+; GFX11W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W64-NEXT: buffer_atomic_add_u32 v1, v2, s[8:11], 0 idxen glc
@@ -195,18 +195,19 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-LABEL: add_i32_constant:
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB0_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX11W32-NEXT: v_mov_b32_e32 v2, 0
; GFX11W32-NEXT: s_mul_i32 s1, s1, 5
; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11W32-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s1
+; GFX11W32-NEXT: v_mov_b32_e32 v1, s1
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: buffer_atomic_add_u32 v1, v2, s[8:11], 0 idxen glc
; GFX11W32-NEXT: .LBB0_2:
@@ -224,19 +225,19 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-LABEL: add_i32_constant:
; GFX12W64: ; %bb.0: ; %entry
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB0_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
-; GFX12W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX12W64-NEXT: v_mov_b32_e32 v2, 0
+; GFX12W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: s_wait_kmcnt 0x0
; GFX12W64-NEXT: buffer_atomic_add_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
@@ -256,17 +257,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-LABEL: add_i32_constant:
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB0_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W32-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s1
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: buffer_atomic_add_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
@@ -285,19 +286,19 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-LABEL: add_i32_constant:
; GFX13W64: ; %bb.0: ; %entry
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB0_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
-; GFX13W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX13W64-NEXT: v_mov_b32_e32 v2, 0
+; GFX13W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: s_wait_kmcnt 0x0
; GFX13W64-NEXT: buffer_atomic_add_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
@@ -316,17 +317,17 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-LABEL: add_i32_constant:
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB0_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W32-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s1
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: buffer_atomic_add_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
@@ -350,27 +351,27 @@ entry:
define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(8) %inout, i32 %additive) {
; GFX6-LABEL: add_i32_uniform:
; GFX6: ; %bb.0: ; %entry
-; GFX6-NEXT: s_load_dword s2, s[4:5], 0x11
+; GFX6-NEXT: s_load_dword s6, s[4:5], 0x11
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB1_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_mul_i32 s3, s2, s3
-; GFX6-NEXT: v_mov_b32_e32 v1, s3
+; GFX6-NEXT: s_mul_i32 s2, s6, s2
+; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: v_mov_b32_e32 v2, 0
; GFX6-NEXT: buffer_atomic_add v1, v2, s[8:11], 0 idxen glc
; GFX6-NEXT: .LBB1_2:
; GFX6-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX6-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX6-NEXT: s_waitcnt vmcnt(0)
; GFX6-NEXT: v_readfirstlane_b32 s4, v1
; GFX6-NEXT: s_mov_b32 s3, 0xf000
@@ -381,27 +382,27 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX8-LABEL: add_i32_uniform:
; GFX8: ; %bb.0: ; %entry
-; GFX8-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX8-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB1_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_mul_i32 s3, s2, s3
-; GFX8-NEXT: v_mov_b32_e32 v1, s3
+; GFX8-NEXT: s_mul_i32 s2, s6, s2
+; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: v_mov_b32_e32 v2, 0
; GFX8-NEXT: buffer_atomic_add v1, v2, s[8:11], 0 idxen glc
; GFX8-NEXT: .LBB1_2:
; GFX8-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX8-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: v_readfirstlane_b32 s2, v1
; GFX8-NEXT: v_add_u32_e32 v2, vcc, s2, v0
@@ -412,27 +413,27 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX9-LABEL: add_i32_uniform:
; GFX9: ; %bb.0: ; %entry
-; GFX9-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX9-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB1_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s3, s2, s3
-; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_mul_i32 s2, s6, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: v_mov_b32_e32 v2, 0
; GFX9-NEXT: buffer_atomic_add v1, v2, s[8:11], 0 idxen glc
; GFX9-NEXT: .LBB1_2:
; GFX9-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX9-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX9-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: v_readfirstlane_b32 s2, v1
; GFX9-NEXT: v_mov_b32_e32 v2, 0
@@ -442,31 +443,30 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX10W64-LABEL: add_i32_uniform:
; GFX10W64: ; %bb.0: ; %entry
-; GFX10W64-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX10W64-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB1_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
-; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: v_mov_b32_e32 v2, 0
-; GFX10W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: buffer_atomic_add v1, v2, s[8:11], 0 idxen glc
; GFX10W64-NEXT: .LBB1_2:
; GFX10W64-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)
; GFX10W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX10W64-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX10W64-NEXT: s_waitcnt vmcnt(0)
-; GFX10W64-NEXT: s_mov_b32 null, 0
-; GFX10W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX10W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: v_mad_u64_u32 v[0:1], s[2:3], s2, v0, s[4:5]
+; GFX10W64-NEXT: v_mad_u64_u32 v[0:1], s[2:3], s6, v0, s[2:3]
; GFX10W64-NEXT: v_mov_b32_e32 v1, 0
; GFX10W64-NEXT: global_store_dword v1, v0, s[0:1]
; GFX10W64-NEXT: s_endpgm
@@ -475,17 +475,17 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: s_load_dword s0, s[4:5], 0x44
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s2, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s1
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s1, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB1_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s2
+; GFX10W32-NEXT: v_mov_b32_e32 v2, 0
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX10W32-NEXT: s_mul_i32 s2, s0, s2
-; GFX10W32-NEXT: v_mov_b32_e32 v2, 0
; GFX10W32-NEXT: v_mov_b32_e32 v1, s2
; GFX10W32-NEXT: buffer_atomic_add v1, v2, s[8:11], 0 idxen glc
; GFX10W32-NEXT: .LBB1_2:
@@ -503,32 +503,32 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX11W64-LABEL: add_i32_uniform:
; GFX11W64: ; %bb.0: ; %entry
-; GFX11W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX11W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB1_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
-; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX11W64-NEXT: v_mov_b32_e32 v2, 0
-; GFX11W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: buffer_atomic_add_u32 v1, v2, s[8:11], 0 idxen glc
; GFX11W64-NEXT: .LBB1_2:
; GFX11W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX11W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11W64-NEXT: s_waitcnt vmcnt(0)
-; GFX11W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX11W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11W64-NEXT: v_mad_u64_u32 v[1:2], null, s2, v0, s[4:5]
+; GFX11W64-NEXT: v_mad_u64_u32 v[1:2], null, s6, v0, s[2:3]
; GFX11W64-NEXT: v_mov_b32_e32 v0, 0
; GFX11W64-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11W64-NEXT: s_endpgm
@@ -537,19 +537,20 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s2, exec_lo
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB1_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s2
+; GFX11W32-NEXT: v_mov_b32_e32 v2, 0
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: s_mul_i32 s2, s0, s2
; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11W32-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s2
+; GFX11W32-NEXT: v_mov_b32_e32 v1, s2
; GFX11W32-NEXT: buffer_atomic_add_u32 v1, v2, s[8:11], 0 idxen glc
; GFX11W32-NEXT: .LBB1_2:
; GFX11W32-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -565,32 +566,33 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX12W64-LABEL: add_i32_uniform:
; GFX12W64: ; %bb.0: ; %entry
-; GFX12W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX12W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB1_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
-; GFX12W64-NEXT: s_wait_kmcnt 0x0
-; GFX12W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX12W64-NEXT: v_mov_b32_e32 v2, 0
-; GFX12W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX12W64-NEXT: s_wait_kmcnt 0x0
+; GFX12W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: buffer_atomic_add_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
; GFX12W64-NEXT: .LBB1_2:
; GFX12W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX12W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX12W64-NEXT: s_wait_loadcnt 0x0
-; GFX12W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX12W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX12W64-NEXT: s_wait_kmcnt 0x0
+; GFX12W64-NEXT: s_wait_alu depctr_va_sdst(0)
; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s2, v0, s[4:5]
+; GFX12W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s6, v0, s[2:3]
; GFX12W64-NEXT: v_mov_b32_e32 v1, 0
; GFX12W64-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX12W64-NEXT: s_endpgm
@@ -599,15 +601,15 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s2, exec_lo
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB1_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: s_mul_i32 s2, s0, s2
; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -627,32 +629,32 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX13W64-LABEL: add_i32_uniform:
; GFX13W64: ; %bb.0: ; %entry
-; GFX13W64-NEXT: s_load_b32 s2, s[4:5], 0x44 nv
+; GFX13W64-NEXT: s_load_b32 s6, s[4:5], 0x44 nv
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB1_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
-; GFX13W64-NEXT: s_wait_kmcnt 0x0
-; GFX13W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX13W64-NEXT: v_mov_b32_e32 v2, 0
-; GFX13W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX13W64-NEXT: s_wait_kmcnt 0x0
+; GFX13W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: buffer_atomic_add_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
; GFX13W64-NEXT: .LBB1_2:
; GFX13W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX13W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13W64-NEXT: s_wait_loadcnt 0x0
-; GFX13W64-NEXT: v_readfirstlane_b32 s4, v1
+; GFX13W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX13W64-NEXT: s_wait_kmcnt 0x0
; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX13W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s2, v0, s[4:5]
+; GFX13W64-NEXT: v_mad_co_u64_u32 v[0:1], null, s6, v0, s[2:3]
; GFX13W64-NEXT: v_mov_b32_e32 v1, 0
; GFX13W64-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX13W64-NEXT: s_endpgm
@@ -661,15 +663,15 @@ define amdgpu_kernel void @add_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: s_load_b32 s0, s[4:5], 0x44 nv
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s2, exec_lo
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB1_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: s_mul_i32 s2, s0, s2
; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -1461,14 +1463,14 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX6: ; %bb.0: ; %entry
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB5_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_mul_i32 s2, s2, 5
; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: v_mov_b32_e32 v2, 0
@@ -1491,14 +1493,14 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX8: ; %bb.0: ; %entry
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB5_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_mul_i32 s2, s2, 5
; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: v_mov_b32_e32 v2, 0
@@ -1521,14 +1523,14 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB5_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_mul_i32 s2, s2, 5
; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: v_mov_b32_e32 v2, 0
@@ -1549,17 +1551,17 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W64-LABEL: sub_i32_constant:
; GFX10W64: ; %bb.0: ; %entry
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB5_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
-; GFX10W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: v_mov_b32_e32 v2, 0
+; GFX10W64-NEXT: s_mul_i32 s2, s2, 5
; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX10W64-NEXT: buffer_atomic_sub v1, v2, s[8:11], 0 idxen glc
@@ -1579,16 +1581,16 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10W32-LABEL: sub_i32_constant:
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB5_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
-; GFX10W32-NEXT: s_mul_i32 s1, s1, 5
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s1, s1
; GFX10W32-NEXT: v_mov_b32_e32 v2, 0
+; GFX10W32-NEXT: s_mul_i32 s1, s1, 5
; GFX10W32-NEXT: v_mov_b32_e32 v1, s1
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX10W32-NEXT: buffer_atomic_sub v1, v2, s[8:11], 0 idxen glc
@@ -1608,19 +1610,19 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W64-LABEL: sub_i32_constant:
; GFX11W64: ; %bb.0: ; %entry
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB5_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
-; GFX11W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX11W64-NEXT: v_mov_b32_e32 v2, 0
+; GFX11W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W64-NEXT: buffer_atomic_sub_u32 v1, v2, s[8:11], 0 idxen glc
@@ -1640,18 +1642,19 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11W32-LABEL: sub_i32_constant:
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX11W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB5_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX11W32-NEXT: v_mov_b32_e32 v2, 0
; GFX11W32-NEXT: s_mul_i32 s1, s1, 5
; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11W32-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s1
+; GFX11W32-NEXT: v_mov_b32_e32 v1, s1
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: buffer_atomic_sub_u32 v1, v2, s[8:11], 0 idxen glc
; GFX11W32-NEXT: .LBB5_2:
@@ -1670,19 +1673,19 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W64-LABEL: sub_i32_constant:
; GFX12W64: ; %bb.0: ; %entry
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB5_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
-; GFX12W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX12W64-NEXT: v_mov_b32_e32 v2, 0
+; GFX12W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: s_wait_kmcnt 0x0
; GFX12W64-NEXT: buffer_atomic_sub_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
@@ -1703,17 +1706,17 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12W32-LABEL: sub_i32_constant:
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX12W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB5_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12W32-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s1
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: buffer_atomic_sub_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
@@ -1733,19 +1736,19 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W64-LABEL: sub_i32_constant:
; GFX13W64: ; %bb.0: ; %entry
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB5_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
-; GFX13W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX13W64-NEXT: v_mov_b32_e32 v2, 0
+; GFX13W64-NEXT: s_mul_i32 s2, s2, 5
+; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: s_wait_kmcnt 0x0
; GFX13W64-NEXT: buffer_atomic_sub_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
@@ -1765,17 +1768,17 @@ define amdgpu_kernel void @sub_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX13W32-LABEL: sub_i32_constant:
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX13W32-NEXT: s_mov_b32 s0, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB5_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX13W32-NEXT: s_mul_i32 s1, s1, 5
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX13W32-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s1
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: buffer_atomic_sub_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
@@ -1799,27 +1802,27 @@ entry:
define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(8) %inout, i32 %subitive) {
; GFX6-LABEL: sub_i32_uniform:
; GFX6: ; %bb.0: ; %entry
-; GFX6-NEXT: s_load_dword s2, s[4:5], 0x11
+; GFX6-NEXT: s_load_dword s6, s[4:5], 0x11
; GFX6-NEXT: v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
; GFX6-NEXT: v_mbcnt_hi_u32_b32_e32 v0, exec_hi, v0
-; GFX6-NEXT: s_mov_b64 s[0:1], exec
-; GFX6-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX6-NEXT: s_mov_b64 s[2:3], exec
; GFX6-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX6-NEXT: ; implicit-def: $vgpr1
; GFX6-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX6-NEXT: s_cbranch_execz .LBB6_2
; GFX6-NEXT: ; %bb.1:
; GFX6-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0xd
+; GFX6-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_mul_i32 s3, s2, s3
-; GFX6-NEXT: v_mov_b32_e32 v1, s3
+; GFX6-NEXT: s_mul_i32 s2, s6, s2
+; GFX6-NEXT: v_mov_b32_e32 v1, s2
; GFX6-NEXT: v_mov_b32_e32 v2, 0
; GFX6-NEXT: buffer_atomic_sub v1, v2, s[8:11], 0 idxen glc
; GFX6-NEXT: .LBB6_2:
; GFX6-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX6-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX6-NEXT: s_waitcnt vmcnt(0)
; GFX6-NEXT: v_readfirstlane_b32 s4, v1
; GFX6-NEXT: s_mov_b32 s3, 0xf000
@@ -1830,27 +1833,27 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX8-LABEL: sub_i32_uniform:
; GFX8: ; %bb.0: ; %entry
-; GFX8-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX8-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX8-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX8-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX8-NEXT: s_mov_b64 s[0:1], exec
-; GFX8-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX8-NEXT: s_mov_b64 s[2:3], exec
; GFX8-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX8-NEXT: ; implicit-def: $vgpr1
; GFX8-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX8-NEXT: s_cbranch_execz .LBB6_2
; GFX8-NEXT: ; %bb.1:
; GFX8-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX8-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_mul_i32 s3, s2, s3
-; GFX8-NEXT: v_mov_b32_e32 v1, s3
+; GFX8-NEXT: s_mul_i32 s2, s6, s2
+; GFX8-NEXT: v_mov_b32_e32 v1, s2
; GFX8-NEXT: v_mov_b32_e32 v2, 0
; GFX8-NEXT: buffer_atomic_sub v1, v2, s[8:11], 0 idxen glc
; GFX8-NEXT: .LBB6_2:
; GFX8-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX8-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX8-NEXT: s_waitcnt vmcnt(0)
; GFX8-NEXT: v_readfirstlane_b32 s2, v1
; GFX8-NEXT: v_sub_u32_e32 v2, vcc, s2, v0
@@ -1861,27 +1864,27 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX9-LABEL: sub_i32_uniform:
; GFX9: ; %bb.0: ; %entry
-; GFX9-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX9-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB6_2
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: s_mul_i32 s3, s2, s3
-; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_mul_i32 s2, s6, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: v_mov_b32_e32 v2, 0
; GFX9-NEXT: buffer_atomic_sub v1, v2, s[8:11], 0 idxen glc
; GFX9-NEXT: .LBB6_2:
; GFX9-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX9-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX9-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: v_readfirstlane_b32 s2, v1
; GFX9-NEXT: v_mov_b32_e32 v2, 0
@@ -1891,28 +1894,28 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX10W64-LABEL: sub_i32_uniform:
; GFX10W64: ; %bb.0: ; %entry
-; GFX10W64-NEXT: s_load_dword s2, s[4:5], 0x44
+; GFX10W64-NEXT: s_load_dword s6, s[4:5], 0x44
; GFX10W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX10W64-NEXT: s_mov_b64 s[2:3], exec
; GFX10W64-NEXT: ; implicit-def: $vgpr1
-; GFX10W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
; GFX10W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX10W64-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX10W64-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX10W64-NEXT: s_cbranch_execz .LBB6_2
; GFX10W64-NEXT: ; %bb.1:
; GFX10W64-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
-; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX10W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX10W64-NEXT: v_mov_b32_e32 v2, 0
-; GFX10W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX10W64-NEXT: v_mov_b32_e32 v1, s2
; GFX10W64-NEXT: buffer_atomic_sub v1, v2, s[8:11], 0 idxen glc
; GFX10W64-NEXT: .LBB6_2:
; GFX10W64-NEXT: s_waitcnt_depctr depctr_vm_vsrc(0)
; GFX10W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX10W64-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX10W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX10W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX10W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX10W64-NEXT: s_waitcnt vmcnt(0)
; GFX10W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX10W64-NEXT: v_mov_b32_e32 v1, 0
@@ -1924,17 +1927,17 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX10W32: ; %bb.0: ; %entry
; GFX10W32-NEXT: s_load_dword s0, s[4:5], 0x44
; GFX10W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX10W32-NEXT: s_mov_b32 s2, exec_lo
; GFX10W32-NEXT: ; implicit-def: $vgpr1
-; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s1
; GFX10W32-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10W32-NEXT: s_and_saveexec_b32 s1, vcc_lo
; GFX10W32-NEXT: s_cbranch_execz .LBB6_2
; GFX10W32-NEXT: ; %bb.1:
; GFX10W32-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
+; GFX10W32-NEXT: s_bcnt1_i32_b32 s2, s2
+; GFX10W32-NEXT: v_mov_b32_e32 v2, 0
; GFX10W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX10W32-NEXT: s_mul_i32 s2, s0, s2
-; GFX10W32-NEXT: v_mov_b32_e32 v2, 0
; GFX10W32-NEXT: v_mov_b32_e32 v1, s2
; GFX10W32-NEXT: buffer_atomic_sub v1, v2, s[8:11], 0 idxen glc
; GFX10W32-NEXT: .LBB6_2:
@@ -1952,29 +1955,29 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX11W64-LABEL: sub_i32_uniform:
; GFX11W64: ; %bb.0: ; %entry
-; GFX11W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX11W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX11W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W64-NEXT: s_mov_b64 s[2:3], exec
; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
; GFX11W64-NEXT: ; implicit-def: $vgpr1
-; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX11W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX11W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W64-NEXT: s_cbranch_execz .LBB6_2
; GFX11W64-NEXT: ; %bb.1:
; GFX11W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
-; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX11W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX11W64-NEXT: v_mov_b32_e32 v2, 0
-; GFX11W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX11W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11W64-NEXT: v_mov_b32_e32 v1, s2
; GFX11W64-NEXT: buffer_atomic_sub_u32 v1, v2, s[8:11], 0 idxen glc
; GFX11W64-NEXT: .LBB6_2:
; GFX11W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX11W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11W64-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX11W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX11W64-NEXT: s_waitcnt vmcnt(0)
; GFX11W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX11W64-NEXT: v_mov_b32_e32 v1, 0
@@ -1987,19 +1990,20 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX11W32: ; %bb.0: ; %entry
; GFX11W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX11W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11W32-NEXT: s_mov_b32 s2, exec_lo
; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
; GFX11W32-NEXT: ; implicit-def: $vgpr1
-; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX11W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX11W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11W32-NEXT: s_cbranch_execz .LBB6_2
; GFX11W32-NEXT: ; %bb.1:
; GFX11W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX11W32-NEXT: s_bcnt1_i32_b32 s2, s2
+; GFX11W32-NEXT: v_mov_b32_e32 v2, 0
; GFX11W32-NEXT: s_waitcnt lgkmcnt(0)
; GFX11W32-NEXT: s_mul_i32 s2, s0, s2
; GFX11W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11W32-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s2
+; GFX11W32-NEXT: v_mov_b32_e32 v1, s2
; GFX11W32-NEXT: buffer_atomic_sub_u32 v1, v2, s[8:11], 0 idxen glc
; GFX11W32-NEXT: .LBB6_2:
; GFX11W32-NEXT: s_or_b32 exec_lo, exec_lo, s1
@@ -2016,29 +2020,29 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX12W64-LABEL: sub_i32_uniform:
; GFX12W64: ; %bb.0: ; %entry
-; GFX12W64-NEXT: s_load_b32 s2, s[4:5], 0x44
+; GFX12W64-NEXT: s_load_b32 s6, s[4:5], 0x44
; GFX12W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W64-NEXT: s_mov_b64 s[2:3], exec
; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
; GFX12W64-NEXT: ; implicit-def: $vgpr1
-; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX12W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX12W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W64-NEXT: s_cbranch_execz .LBB6_2
; GFX12W64-NEXT: ; %bb.1:
; GFX12W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
-; GFX12W64-NEXT: s_wait_kmcnt 0x0
-; GFX12W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX12W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX12W64-NEXT: v_mov_b32_e32 v2, 0
-; GFX12W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX12W64-NEXT: s_wait_kmcnt 0x0
+; GFX12W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX12W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12W64-NEXT: v_mov_b32_e32 v1, s2
; GFX12W64-NEXT: buffer_atomic_sub_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
; GFX12W64-NEXT: .LBB6_2:
; GFX12W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX12W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX12W64-NEXT: s_wait_kmcnt 0x0
-; GFX12W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX12W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX12W64-NEXT: s_wait_loadcnt 0x0
; GFX12W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX12W64-NEXT: v_mov_b32_e32 v1, 0
@@ -2052,15 +2056,15 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX12W32: ; %bb.0: ; %entry
; GFX12W32-NEXT: s_load_b32 s0, s[4:5], 0x44
; GFX12W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12W32-NEXT: s_mov_b32 s2, exec_lo
; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
; GFX12W32-NEXT: ; implicit-def: $vgpr1
-; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX12W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX12W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12W32-NEXT: s_cbranch_execz .LBB6_2
; GFX12W32-NEXT: ; %bb.1:
; GFX12W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
+; GFX12W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX12W32-NEXT: s_wait_kmcnt 0x0
; GFX12W32-NEXT: s_mul_i32 s2, s0, s2
; GFX12W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
@@ -2082,29 +2086,29 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
;
; GFX13W64-LABEL: sub_i32_uniform:
; GFX13W64: ; %bb.0: ; %entry
-; GFX13W64-NEXT: s_load_b32 s2, s[4:5], 0x44 nv
+; GFX13W64-NEXT: s_load_b32 s6, s[4:5], 0x44 nv
; GFX13W64-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W64-NEXT: s_mov_b64 s[2:3], exec
; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
; GFX13W64-NEXT: ; implicit-def: $vgpr1
-; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W64-NEXT: s_bcnt1_i32_b64 s3, s[0:1]
-; GFX13W64-NEXT: s_mov_b64 s[0:1], exec
+; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX13W64-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX13W64-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W64-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W64-NEXT: s_cbranch_execz .LBB6_2
; GFX13W64-NEXT: ; %bb.1:
; GFX13W64-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
-; GFX13W64-NEXT: s_wait_kmcnt 0x0
-; GFX13W64-NEXT: s_mul_i32 s3, s2, s3
+; GFX13W64-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX13W64-NEXT: v_mov_b32_e32 v2, 0
-; GFX13W64-NEXT: v_mov_b32_e32 v1, s3
+; GFX13W64-NEXT: s_wait_kmcnt 0x0
+; GFX13W64-NEXT: s_mul_i32 s2, s6, s2
+; GFX13W64-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13W64-NEXT: v_mov_b32_e32 v1, s2
; GFX13W64-NEXT: buffer_atomic_sub_u32 v1, v2, s[8:11], null idxen th:TH_ATOMIC_RETURN
; GFX13W64-NEXT: .LBB6_2:
; GFX13W64-NEXT: s_or_b64 exec, exec, s[0:1]
; GFX13W64-NEXT: s_load_b64 s[0:1], s[4:5], 0x24 nv
; GFX13W64-NEXT: s_wait_kmcnt 0x0
-; GFX13W64-NEXT: v_mul_lo_u32 v0, s2, v0
+; GFX13W64-NEXT: v_mul_lo_u32 v0, s6, v0
; GFX13W64-NEXT: s_wait_loadcnt 0x0
; GFX13W64-NEXT: v_readfirstlane_b32 s2, v1
; GFX13W64-NEXT: v_mov_b32_e32 v1, 0
@@ -2117,15 +2121,15 @@ define amdgpu_kernel void @sub_i32_uniform(ptr addrspace(1) %out, ptr addrspace(
; GFX13W32: ; %bb.0: ; %entry
; GFX13W32-NEXT: s_load_b32 s0, s[4:5], 0x44 nv
; GFX13W32-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX13W32-NEXT: s_mov_b32 s2, exec_lo
; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
; GFX13W32-NEXT: ; implicit-def: $vgpr1
-; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s1
-; GFX13W32-NEXT: s_mov_b32 s1, exec_lo
+; GFX13W32-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX13W32-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX13W32-NEXT: s_cbranch_execz .LBB6_2
; GFX13W32-NEXT: ; %bb.1:
; GFX13W32-NEXT: s_load_b128 s[8:11], s[4:5], 0x34 nv
+; GFX13W32-NEXT: s_bcnt1_i32_b32 s2, s2
; GFX13W32-NEXT: s_wait_kmcnt 0x0
; GFX13W32-NEXT: s_mul_i32 s2, s0, s2
; GFX13W32-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
diff --git a/llvm/test/CodeGen/AMDGPU/insert_waitcnt_for_precise_memory.ll b/llvm/test/CodeGen/AMDGPU/insert_waitcnt_for_precise_memory.ll
index dae9c8177bc9b0..6e428413a1f736 100644
--- a/llvm/test/CodeGen/AMDGPU/insert_waitcnt_for_precise_memory.ll
+++ b/llvm/test/CodeGen/AMDGPU/insert_waitcnt_for_precise_memory.ll
@@ -686,16 +686,16 @@ define amdgpu_kernel void @atomic_add_local(ptr addrspace(3) %local) {
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s0, s[0:1]
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX9-NEXT: s_cbranch_execz .LBB5_2
; GFX9-NEXT: ; %bb.1:
-; GFX9-NEXT: s_load_dword s1, s[4:5], 0x24
+; GFX9-NEXT: s_load_dword s2, s[4:5], 0x24
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: s_bcnt1_i32_b64 s0, s[0:1]
; GFX9-NEXT: s_mul_i32 s0, s0, 5
; GFX9-NEXT: v_mov_b32_e32 v1, s0
-; GFX9-NEXT: v_mov_b32_e32 v0, s1
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
; GFX9-NEXT: ds_add_u32 v0, v1
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-NEXT: .LBB5_2:
@@ -706,16 +706,16 @@ define amdgpu_kernel void @atomic_add_local(ptr addrspace(3) %local) {
; GFX90A-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX90A-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX90A-NEXT: s_mov_b64 s[0:1], exec
-; GFX90A-NEXT: s_bcnt1_i32_b64 s0, s[0:1]
; GFX90A-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX90A-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX90A-NEXT: s_cbranch_execz .LBB5_2
; GFX90A-NEXT: ; %bb.1:
-; GFX90A-NEXT: s_load_dword s1, s[4:5], 0x24
+; GFX90A-NEXT: s_load_dword s2, s[4:5], 0x24
; GFX90A-NEXT: s_waitcnt lgkmcnt(0)
+; GFX90A-NEXT: s_bcnt1_i32_b64 s0, s[0:1]
; GFX90A-NEXT: s_mul_i32 s0, s0, 5
; GFX90A-NEXT: v_mov_b32_e32 v1, s0
-; GFX90A-NEXT: v_mov_b32_e32 v0, s1
+; GFX90A-NEXT: v_mov_b32_e32 v0, s2
; GFX90A-NEXT: ds_add_u32 v0, v1
; GFX90A-NEXT: s_waitcnt lgkmcnt(0)
; GFX90A-NEXT: .LBB5_2:
@@ -725,13 +725,13 @@ define amdgpu_kernel void @atomic_add_local(ptr addrspace(3) %local) {
; GFX10: ; %bb.0:
; GFX10-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX10-NEXT: s_mov_b32 s0, exec_lo
-; GFX10-NEXT: s_bcnt1_i32_b32 s0, s0
; GFX10-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10-NEXT: s_and_saveexec_b32 s1, vcc_lo
; GFX10-NEXT: s_cbranch_execz .LBB5_2
; GFX10-NEXT: ; %bb.1:
; GFX10-NEXT: s_load_dword s1, s[4:5], 0x24
; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_bcnt1_i32_b32 s0, s0
; GFX10-NEXT: s_mul_i32 s0, s0, 5
; GFX10-NEXT: v_mov_b32_e32 v1, s0
; GFX10-NEXT: v_mov_b32_e32 v0, s1
@@ -746,16 +746,16 @@ define amdgpu_kernel void @atomic_add_local(ptr addrspace(3) %local) {
; GFX9-FLATSCR-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-FLATSCR-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
; GFX9-FLATSCR-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-FLATSCR-NEXT: s_bcnt1_i32_b64 s0, s[0:1]
; GFX9-FLATSCR-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-FLATSCR-NEXT: s_and_saveexec_b64 s[2:3], vcc
; GFX9-FLATSCR-NEXT: s_cbranch_execz .LBB5_2
; GFX9-FLATSCR-NEXT: ; %bb.1:
-; GFX9-FLATSCR-NEXT: s_load_dword s1, s[4:5], 0x24
+; GFX9-FLATSCR-NEXT: s_load_dword s2, s[4:5], 0x24
; GFX9-FLATSCR-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-FLATSCR-NEXT: s_bcnt1_i32_b64 s0, s[0:1]
; GFX9-FLATSCR-NEXT: s_mul_i32 s0, s0, 5
; GFX9-FLATSCR-NEXT: v_mov_b32_e32 v1, s0
-; GFX9-FLATSCR-NEXT: v_mov_b32_e32 v0, s1
+; GFX9-FLATSCR-NEXT: v_mov_b32_e32 v0, s2
; GFX9-FLATSCR-NEXT: ds_add_u32 v0, v1
; GFX9-FLATSCR-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-FLATSCR-NEXT: .LBB5_2:
@@ -766,15 +766,15 @@ define amdgpu_kernel void @atomic_add_local(ptr addrspace(3) %local) {
; GFX11-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: s_mov_b32 s1, exec_lo
-; GFX11-NEXT: s_bcnt1_i32_b32 s0, s0
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11-NEXT: s_cbranch_execz .LBB5_2
; GFX11-NEXT: ; %bb.1:
; GFX11-NEXT: s_load_b32 s1, s[4:5], 0x24
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: s_bcnt1_i32_b32 s0, s0
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_mul_i32 s0, s0, 5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v1, s0 :: v_dual_mov_b32 v0, s1
; GFX11-NEXT: ds_add_u32 v0, v1
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
@@ -787,15 +787,15 @@ define amdgpu_kernel void @atomic_add_local(ptr addrspace(3) %local) {
; GFX12-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: s_mov_b32 s1, exec_lo
-; GFX12-NEXT: s_bcnt1_i32_b32 s0, s0
; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12-NEXT: s_cbranch_execz .LBB5_2
; GFX12-NEXT: ; %bb.1:
; GFX12-NEXT: s_load_b32 s1, s[4:5], 0x24
; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: s_bcnt1_i32_b32 s0, s0
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_mul_i32 s0, s0, 5
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v1, s0 :: v_dual_mov_b32 v0, s1
; GFX12-NEXT: ds_add_u32 v0, v1
; GFX12-NEXT: s_wait_dscnt 0x0
@@ -884,18 +884,18 @@ define amdgpu_kernel void @atomic_add_ret_local(ptr addrspace(1) %out, ptr addrs
; GFX9: ; %bb.0:
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-NEXT: s_cbranch_execz .LBB7_2
; GFX9-NEXT: ; %bb.1:
-; GFX9-NEXT: s_load_dword s3, s[4:5], 0x2c
+; GFX9-NEXT: s_load_dword s6, s[4:5], 0x2c
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_mul_i32 s2, s2, 5
; GFX9-NEXT: v_mov_b32_e32 v2, s2
-; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: v_mov_b32_e32 v1, s6
; GFX9-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-NEXT: .LBB7_2:
@@ -913,18 +913,18 @@ define amdgpu_kernel void @atomic_add_ret_local(ptr addrspace(1) %out, ptr addrs
; GFX90A: ; %bb.0:
; GFX90A-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX90A-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX90A-NEXT: s_mov_b64 s[0:1], exec
-; GFX90A-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX90A-NEXT: s_mov_b64 s[2:3], exec
; GFX90A-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX90A-NEXT: ; implicit-def: $vgpr1
; GFX90A-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX90A-NEXT: s_cbranch_execz .LBB7_2
; GFX90A-NEXT: ; %bb.1:
-; GFX90A-NEXT: s_load_dword s3, s[4:5], 0x2c
+; GFX90A-NEXT: s_load_dword s6, s[4:5], 0x2c
; GFX90A-NEXT: s_waitcnt lgkmcnt(0)
+; GFX90A-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX90A-NEXT: s_mul_i32 s2, s2, 5
; GFX90A-NEXT: v_mov_b32_e32 v2, s2
-; GFX90A-NEXT: v_mov_b32_e32 v1, s3
+; GFX90A-NEXT: v_mov_b32_e32 v1, s6
; GFX90A-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX90A-NEXT: s_waitcnt lgkmcnt(0)
; GFX90A-NEXT: .LBB7_2:
@@ -941,15 +941,15 @@ define amdgpu_kernel void @atomic_add_ret_local(ptr addrspace(1) %out, ptr addrs
; GFX10-LABEL: atomic_add_ret_local:
; GFX10: ; %bb.0:
; GFX10-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10-NEXT: s_mov_b32 s0, exec_lo
+; GFX10-NEXT: s_mov_b32 s1, exec_lo
; GFX10-NEXT: ; implicit-def: $vgpr1
-; GFX10-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX10-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX10-NEXT: s_cbranch_execz .LBB7_2
; GFX10-NEXT: ; %bb.1:
; GFX10-NEXT: s_load_dword s2, s[4:5], 0x2c
; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_bcnt1_i32_b32 s1, s1
; GFX10-NEXT: s_mul_i32 s1, s1, 5
; GFX10-NEXT: v_mov_b32_e32 v2, s1
; GFX10-NEXT: v_mov_b32_e32 v1, s2
@@ -972,18 +972,18 @@ define amdgpu_kernel void @atomic_add_ret_local(ptr addrspace(1) %out, ptr addrs
; GFX9-FLATSCR: ; %bb.0:
; GFX9-FLATSCR-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-FLATSCR-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-FLATSCR-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-FLATSCR-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-FLATSCR-NEXT: s_mov_b64 s[2:3], exec
; GFX9-FLATSCR-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-FLATSCR-NEXT: ; implicit-def: $vgpr1
; GFX9-FLATSCR-NEXT: s_and_saveexec_b64 s[0:1], vcc
; GFX9-FLATSCR-NEXT: s_cbranch_execz .LBB7_2
; GFX9-FLATSCR-NEXT: ; %bb.1:
-; GFX9-FLATSCR-NEXT: s_load_dword s3, s[4:5], 0x2c
+; GFX9-FLATSCR-NEXT: s_load_dword s6, s[4:5], 0x2c
; GFX9-FLATSCR-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-FLATSCR-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-FLATSCR-NEXT: s_mul_i32 s2, s2, 5
; GFX9-FLATSCR-NEXT: v_mov_b32_e32 v2, s2
-; GFX9-FLATSCR-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-FLATSCR-NEXT: v_mov_b32_e32 v1, s6
; GFX9-FLATSCR-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX9-FLATSCR-NEXT: s_waitcnt lgkmcnt(0)
; GFX9-FLATSCR-NEXT: .LBB7_2:
@@ -1000,18 +1000,18 @@ define amdgpu_kernel void @atomic_add_ret_local(ptr addrspace(1) %out, ptr addrs
; GFX11-LABEL: atomic_add_ret_local:
; GFX11: ; %bb.0:
; GFX11-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11-NEXT: s_mov_b32 s1, exec_lo
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX11-NEXT: s_mov_b32 s0, exec_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11-NEXT: s_cbranch_execz .LBB7_2
; GFX11-NEXT: ; %bb.1:
; GFX11-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_mul_i32 s1, s1, 5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s2
; GFX11-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
@@ -1031,18 +1031,18 @@ define amdgpu_kernel void @atomic_add_ret_local(ptr addrspace(1) %out, ptr addrs
; GFX12-LABEL: atomic_add_ret_local:
; GFX12: ; %bb.0:
; GFX12-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12-NEXT: s_mov_b32 s1, exec_lo
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: ; implicit-def: $vgpr1
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX12-NEXT: s_mov_b32 s0, exec_lo
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12-NEXT: s_cbranch_execz .LBB7_2
; GFX12-NEXT: ; %bb.1:
; GFX12-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_mul_i32 s1, s1, 5
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_dual_mov_b32 v2, s1 :: v_dual_mov_b32 v1, s2
; GFX12-NEXT: ds_add_rtn_u32 v1, v1, v2
; GFX12-NEXT: s_wait_dscnt 0x0
@@ -1075,8 +1075,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9: ; %bb.0: ; %entry
; GFX9-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-NEXT: s_mov_b64 s[2:3], exec
; GFX9-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-NEXT: ; implicit-def: $vgpr1
; GFX9-NEXT: s_and_saveexec_b64 s[0:1], vcc
@@ -1084,6 +1083,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9-NEXT: ; %bb.1:
; GFX9-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-NEXT: s_mul_i32 s2, s2, 5
; GFX9-NEXT: v_mov_b32_e32 v1, s2
; GFX9-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
@@ -1103,8 +1103,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX90A: ; %bb.0: ; %entry
; GFX90A-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX90A-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX90A-NEXT: s_mov_b64 s[0:1], exec
-; GFX90A-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX90A-NEXT: s_mov_b64 s[2:3], exec
; GFX90A-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX90A-NEXT: ; implicit-def: $vgpr1
; GFX90A-NEXT: s_and_saveexec_b64 s[0:1], vcc
@@ -1112,6 +1111,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX90A-NEXT: ; %bb.1:
; GFX90A-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
; GFX90A-NEXT: s_waitcnt lgkmcnt(0)
+; GFX90A-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX90A-NEXT: s_mul_i32 s2, s2, 5
; GFX90A-NEXT: v_mov_b32_e32 v1, s2
; GFX90A-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
@@ -1130,15 +1130,15 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX10-LABEL: add_i32_constant:
; GFX10: ; %bb.0: ; %entry
; GFX10-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
-; GFX10-NEXT: s_mov_b32 s0, exec_lo
+; GFX10-NEXT: s_mov_b32 s1, exec_lo
; GFX10-NEXT: ; implicit-def: $vgpr1
-; GFX10-NEXT: s_bcnt1_i32_b32 s1, s0
; GFX10-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
; GFX10-NEXT: s_and_saveexec_b32 s0, vcc_lo
; GFX10-NEXT: s_cbranch_execz .LBB8_2
; GFX10-NEXT: ; %bb.1:
; GFX10-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_bcnt1_i32_b32 s1, s1
; GFX10-NEXT: s_mul_i32 s1, s1, 5
; GFX10-NEXT: v_mov_b32_e32 v1, s1
; GFX10-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
@@ -1159,8 +1159,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9-FLATSCR: ; %bb.0: ; %entry
; GFX9-FLATSCR-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
; GFX9-FLATSCR-NEXT: v_mbcnt_hi_u32_b32 v0, exec_hi, v0
-; GFX9-FLATSCR-NEXT: s_mov_b64 s[0:1], exec
-; GFX9-FLATSCR-NEXT: s_bcnt1_i32_b64 s2, s[0:1]
+; GFX9-FLATSCR-NEXT: s_mov_b64 s[2:3], exec
; GFX9-FLATSCR-NEXT: v_cmp_eq_u32_e32 vcc, 0, v0
; GFX9-FLATSCR-NEXT: ; implicit-def: $vgpr1
; GFX9-FLATSCR-NEXT: s_and_saveexec_b64 s[0:1], vcc
@@ -1168,6 +1167,7 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX9-FLATSCR-NEXT: ; %bb.1:
; GFX9-FLATSCR-NEXT: s_load_dwordx4 s[8:11], s[4:5], 0x34
; GFX9-FLATSCR-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-FLATSCR-NEXT: s_bcnt1_i32_b64 s2, s[2:3]
; GFX9-FLATSCR-NEXT: s_mul_i32 s2, s2, 5
; GFX9-FLATSCR-NEXT: v_mov_b32_e32 v1, s2
; GFX9-FLATSCR-NEXT: buffer_atomic_add v1, off, s[8:11], 0 glc
@@ -1186,18 +1186,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX11-LABEL: add_i32_constant:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX11-NEXT: s_mov_b32 s1, exec_lo
; GFX11-NEXT: s_mov_b32 s0, exec_lo
; GFX11-NEXT: ; implicit-def: $vgpr1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX11-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX11-NEXT: s_mov_b32 s0, exec_lo
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX11-NEXT: s_cbranch_execz .LBB8_2
; GFX11-NEXT: ; %bb.1:
; GFX11-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX11-NEXT: s_mul_i32 s1, s1, 5
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX11-NEXT: v_mov_b32_e32 v1, s1
; GFX11-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], 0 glc
; GFX11-NEXT: s_waitcnt vmcnt(0)
@@ -1216,18 +1216,18 @@ define amdgpu_kernel void @add_i32_constant(ptr addrspace(1) %out, ptr addrspace
; GFX12-LABEL: add_i32_constant:
; GFX12: ; %bb.0: ; %entry
; GFX12-NEXT: v_mbcnt_lo_u32_b32 v0, exec_lo, 0
+; GFX12-NEXT: s_mov_b32 s1, exec_lo
; GFX12-NEXT: s_mov_b32 s0, exec_lo
; GFX12-NEXT: ; implicit-def: $vgpr1
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX12-NEXT: s_bcnt1_i32_b32 s1, s0
-; GFX12-NEXT: s_mov_b32 s0, exec_lo
+; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX12-NEXT: v_cmpx_eq_u32_e32 0, v0
; GFX12-NEXT: s_cbranch_execz .LBB8_2
; GFX12-NEXT: ; %bb.1:
; GFX12-NEXT: s_load_b128 s[8:11], s[4:5], 0x34
; GFX12-NEXT: s_wait_kmcnt 0x0
+; GFX12-NEXT: s_bcnt1_i32_b32 s1, s1
+; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX12-NEXT: s_mul_i32 s1, s1, 5
-; GFX12-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX12-NEXT: v_mov_b32_e32 v1, s1
; GFX12-NEXT: buffer_atomic_add_u32 v1, off, s[8:11], null th:TH_ATOMIC_RETURN
; GFX12-NEXT: s_wait_loadcnt 0x0
More information about the llvm-commits
mailing list