[llvm] [AMDGPU] Fold away S_AND of a lane mask with EXEC (PR #227729)

Jay Foad via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 00:38:57 PDT 2026


https://github.com/jayfoad updated https://github.com/llvm/llvm-project/pull/227729

>From aba04059a90873f1bf35e871d6d140fc1d39fbb6 Mon Sep 17 00:00:00 2001
From: Jay Foad <jay.foad at amd.com>
Date: Fri, 4 Sep 2026 09:10:01 +0100
Subject: [PATCH] [AMDGPU] Fold away S_AND of a lane mask with EXEC

Instruction selection inserts an S_AND with EXEC on branch conditions
unconditionally, because it has not analysed what produced the lane
mask: see the AndExec handling in AMDGPUDAGToDAGISel::SelectBRCOND and
the S_CBRANCH_SCC* case of SIInstrInfo::moveToVALU. The result of a
V_CMP already has 0 in the bits of all inactive lanes, so the AND is
redundant.

SIPreEmitPeephole::optimizeVccBranch already removes some of these, but
it runs after register allocation and only matches a V_CMP in the VOPC
(_e32) encoding writing physical VCC. That requires SIShrinkInstructions
to have set a VCC register allocation hint, which it only does when
canShrink() succeeds, so any V_CMP that is VOP3-only (source modifiers,
or two SGPR sources) keeps its AND.

Fold the AND away in SIFoldOperands instead, which runs before the
register coalescer, so the V_CMP result can be coalesced into VCC.

Co-authored-by: Claude Opus 5 (1M context) <noreply at anthropic.com>
---
 llvm/lib/Target/AMDGPU/SIFoldOperands.cpp     |  59 +++
 ...-divergent-i1-phis-no-lane-mask-merging.ll |   3 +-
 ...vergence-divergent-i1-used-outside-loop.ll |  83 ++--
 .../GlobalISel/divergence-structurizer.ll     |  69 ++-
 .../divergence-temporal-divergent-i1.ll       |   6 +-
 .../GlobalISel/divergent-control-flow.ll      |   6 +-
 .../AMDGPU/GlobalISel/is-safe-to-sink-bug.ll  |   5 +-
 .../GlobalISel/llvm.amdgcn.ballot.i32.ll      |  12 +-
 .../GlobalISel/llvm.amdgcn.ballot.i64.ll      |   6 +-
 .../AMDGPU/GlobalISel/vni8-across-blocks.ll   |   3 +-
 llvm/test/CodeGen/AMDGPU/add.ll               |   3 +-
 llvm/test/CodeGen/AMDGPU/add_i1.ll            |  11 +-
 llvm/test/CodeGen/AMDGPU/andorbitset.ll       |   3 +-
 .../AMDGPU/atomic_optimizations_buffer.ll     |   9 +-
 .../atomic_optimizations_global_pointer.ll    |  12 +-
 .../atomic_optimizations_local_pointer.ll     |  54 +--
 .../AMDGPU/atomic_optimizations_raw_buffer.ll |   6 +-
 .../atomic_optimizations_struct_buffer.ll     |   6 +-
 llvm/test/CodeGen/AMDGPU/branch-relaxation.ll |   6 +-
 llvm/test/CodeGen/AMDGPU/dagcombine-select.ll |  48 +-
 .../CodeGen/AMDGPU/fneg-modifier-casting.ll   |   3 -
 .../CodeGen/AMDGPU/fold-and-exec-wave32.mir   |  43 ++
 llvm/test/CodeGen/AMDGPU/fold-and-exec.mir    | 281 ++++++++++++
 llvm/test/CodeGen/AMDGPU/frem.ll              | 430 +++++++-----------
 .../CodeGen/AMDGPU/llvm.amdgcn.ballot.i32.ll  |   6 +-
 .../CodeGen/AMDGPU/llvm.amdgcn.ballot.i64.ll  |   3 +-
 .../CodeGen/AMDGPU/llvm.amdgcn.div.fmas.ll    |   3 +-
 llvm/test/CodeGen/AMDGPU/llvm.amdgcn.kill.ll  |   9 +-
 .../CodeGen/AMDGPU/local-atomicrmw-fadd.ll    |  12 +-
 .../AMDGPU/long-branch-reserve-register.ll    |  11 +-
 llvm/test/CodeGen/AMDGPU/loop_break.ll        |  15 +-
 .../CodeGen/AMDGPU/memcpy-crash-issue63986.ll |   5 +-
 .../CodeGen/AMDGPU/memcpy_const_compare.ll    |   8 +-
 .../CodeGen/AMDGPU/memintrinsic-unroll.ll     | 364 +++++++--------
 .../AMDGPU/memset-param-combinations.ll       |  16 +-
 llvm/test/CodeGen/AMDGPU/memset-pattern.ll    |   7 +-
 llvm/test/CodeGen/AMDGPU/mul.ll               |   3 +-
 .../AMDGPU/multi-divergent-exit-region.ll     |  11 +-
 llvm/test/CodeGen/AMDGPU/multilevel-break.ll  |  16 +-
 .../CodeGen/AMDGPU/no-dup-inst-prefetch.ll    |   4 +-
 llvm/test/CodeGen/AMDGPU/select-opt.ll        |   2 +-
 llvm/test/CodeGen/AMDGPU/setcc.ll             |   2 +-
 llvm/test/CodeGen/AMDGPU/sgpr-control-flow.ll |   3 +-
 .../si-unify-exit-return-unreachable.ll       |   9 +-
 llvm/test/CodeGen/AMDGPU/skip-if-dead.ll      |  22 +-
 .../splitkit-getsubrangeformask-phi-extend.ll |   1 -
 llvm/test/CodeGen/AMDGPU/structurize-hoist.ll |   3 +-
 llvm/test/CodeGen/AMDGPU/sub_i1.ll            |  11 +-
 llvm/test/CodeGen/AMDGPU/uaddo.ll             |   6 +-
 llvm/test/CodeGen/AMDGPU/uniform-cfg.ll       |  24 +-
 .../AMDGPU/uniform-inside-divergent-cfg.ll    |   9 +-
 .../AMDGPU/unstructured-cfg-def-use-issue.ll  |   7 +-
 llvm/test/CodeGen/AMDGPU/usubo.ll             |   6 +-
 llvm/test/CodeGen/AMDGPU/valu-i1.ll           |   6 +-
 .../test/CodeGen/AMDGPU/vni8-across-blocks.ll |  22 +-
 llvm/test/CodeGen/AMDGPU/wave32.ll            |  58 +--
 llvm/test/CodeGen/AMDGPU/while-break.ll       |   9 +-
 llvm/test/CodeGen/AMDGPU/wqm.ll               |   4 +-
 llvm/test/CodeGen/AMDGPU/xor.ll               |   3 +-
 59 files changed, 991 insertions(+), 876 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/fold-and-exec-wave32.mir
 create mode 100644 llvm/test/CodeGen/AMDGPU/fold-and-exec.mir

diff --git a/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp b/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
index 21df0ec9376ae57..f214bedb684c615 100644
--- a/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFoldOperands.cpp
@@ -10,6 +10,7 @@
 
 #include "SIFoldOperands.h"
 #include "AMDGPU.h"
+#include "AMDGPULaneMaskUtils.h"
 #include "GCNSubtarget.h"
 #include "SIInstrInfo.h"
 #include "SIMachineFunctionInfo.h"
@@ -257,6 +258,7 @@ class SIFoldOperandsImpl {
   bool tryConstantFoldOp(MachineInstr *MI) const;
   bool tryFoldCndMask(MachineInstr &MI) const;
   bool tryFoldRedundantAND(MachineInstr &ChildMI) const;
+  bool tryFoldAndExec(MachineInstr &MI) const;
   bool foldInstOperand(MachineInstr &MI, const FoldableDef &OpToFold) const;
 
   bool foldCopyToAGPRRegSequence(MachineInstr *CopyMI) const;
@@ -1947,6 +1949,58 @@ bool SIFoldOperandsImpl::tryFoldRedundantAND(MachineInstr &ChildMI) const {
   return true;
 }
 
+/// Remove S_AND of a lane mask with EXEC, when the lane mask is already known
+/// to have 0 in the bits of all inactive lanes.
+///
+/// Instruction selection inserts these unconditionally because it has not
+/// analysed what produced the lane mask.
+bool SIFoldOperandsImpl::tryFoldAndExec(MachineInstr &MI) const {
+  const AMDGPU::LaneMaskConstants &LMC = AMDGPU::LaneMaskConstants::get(*ST);
+  if (MI.getOpcode() != LMC.AndOpc)
+    return false;
+
+  // The AND is going to be removed, so nothing may use the SCC it defines.
+  if (!MI.allImplicitDefsAreDead())
+    return false;
+
+  // Find the EXEC operand, and the lane mask it is being ANDed with.
+  unsigned ExecIdx = 0;
+  for (unsigned I : {1u, 2u}) {
+    const MachineOperand &MO = MI.getOperand(I);
+    if (MO.isReg() && MO.getReg() == LMC.ExecReg)
+      ExecIdx = I;
+  }
+  if (!ExecIdx)
+    return false;
+  MachineOperand &Src = MI.getOperand(3 - ExecIdx);
+  if (!Src.isReg() || !Src.getReg().isVirtual() || Src.getSubReg())
+    return false;
+
+  Register SrcReg = Src.getReg();
+  if (!TII->isMaskedByExec(SrcReg, MI, *MRI))
+    return false;
+
+  LLVM_DEBUG(dbgs() << "Folding redundant AND with EXEC: " << MI);
+
+  Register DstReg = MI.getOperand(0).getReg();
+  if (DstReg.isVirtual()) {
+    if (!MRI->constrainRegClass(SrcReg, MRI->getRegClass(DstReg)))
+      return false;
+    MRI->replaceRegWith(DstReg, SrcReg);
+  } else {
+    // A physical destination, e.g. the $vcc written by moveToVALU. Register
+    // allocation will usually make this copy an identity copy.
+    MachineBasicBlock *MBB = MI.getParent();
+    BuildMI(*MBB, MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), DstReg)
+        .addReg(SrcReg);
+  }
+
+  if (!Src.isKill())
+    MRI->clearKillFlags(SrcReg);
+  MI.eraseFromParent();
+  return true;
+}
+
 bool SIFoldOperandsImpl::foldInstOperand(MachineInstr &MI,
                                          const FoldableDef &OpToFold) const {
   // We need mutate the operands of new mov instructions to add implicit
@@ -3154,6 +3208,11 @@ bool SIFoldOperandsImpl::run(MachineFunction &MF, const MachineLoopInfo *MLI) {
         continue;
       }
 
+      if (tryFoldAndExec(MI)) {
+        Changed = true;
+        continue;
+      }
+
       if (MI.isRegSequence() && tryFoldRegSequence(MI)) {
         Changed = true;
         continue;
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-phis-no-lane-mask-merging.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-phis-no-lane-mask-merging.ll
index 2a4ce06feee4ba6..06f469045c16864 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-phis-no-lane-mask-merging.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-phis-no-lane-mask-merging.ll
@@ -71,8 +71,7 @@ define amdgpu_ps void @divergent_i1_phi_uniform_branch_simple(ptr addrspace(1) %
 ; GFX10-NEXT:  ; %bb.1: ; %B
 ; GFX10-NEXT:    v_cmp_gt_u32_e32 vcc_lo, 1, v2
 ; GFX10-NEXT:    s_andn2_b32 s0, s1, exec_lo
-; GFX10-NEXT:    s_and_b32 s1, exec_lo, vcc_lo
-; GFX10-NEXT:    s_or_b32 s1, s0, s1
+; GFX10-NEXT:    s_or_b32 s1, s0, vcc_lo
 ; GFX10-NEXT:  .LBB1_2: ; %exit
 ; GFX10-NEXT:    v_cndmask_b32_e64 v2, 0, -1, s1
 ; GFX10-NEXT:    v_add_nc_u32_e32 v2, 2, v2
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll
index f18efef146c7b72..31e8ef994a50413 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-divergent-i1-used-outside-loop.ll
@@ -60,9 +60,8 @@ define void @divergent_i1_phi_used_outside_loop_larger_loop_body(float %val, ptr
 ; GFX10:       ; %bb.0: ; %entry
 ; GFX10-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
 ; GFX10-NEXT:    s_andn2_b32 s5, s4, exec_lo
-; GFX10-NEXT:    s_and_b32 s6, exec_lo, exec_lo
 ; GFX10-NEXT:    s_mov_b32 s4, -1
-; GFX10-NEXT:    s_or_b32 s7, s5, s6
+; GFX10-NEXT:    s_or_b32 s7, s5, exec_lo
 ; GFX10-NEXT:    ; implicit-def: $sgpr5
 ; GFX10-NEXT:    s_branch .LBB1_2
 ; GFX10-NEXT:  .LBB1_1: ; %loop.cond
@@ -92,8 +91,7 @@ define void @divergent_i1_phi_used_outside_loop_larger_loop_body(float %val, ptr
 ; GFX10-NEXT:    s_andn2_b32 s5, s5, exec_lo
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 0, v0
-; GFX10-NEXT:    s_and_b32 s8, exec_lo, vcc_lo
-; GFX10-NEXT:    s_or_b32 s5, s5, s8
+; GFX10-NEXT:    s_or_b32 s5, s5, vcc_lo
 ; GFX10-NEXT:    s_branch .LBB1_1
 ; GFX10-NEXT:  .LBB1_4: ; %exit
 ; GFX10-NEXT:    v_cndmask_b32_e64 v0, 0, 1.0, s6
@@ -267,12 +265,10 @@ define void @divergent_i1_xor_used_outside_loop_larger_loop_body(i32 %num.elts,
 ; GFX10-NEXT:    s_andn2_b32 s5, s11, exec_lo
 ; GFX10-NEXT:    v_mov_b32_e32 v5, s12
 ; GFX10-NEXT:    v_mov_b32_e32 v6, s13
-; GFX10-NEXT:    s_and_b32 s11, exec_lo, exec_lo
-; GFX10-NEXT:    s_and_b32 s12, exec_lo, exec_lo
-; GFX10-NEXT:    s_or_b32 s11, s5, s11
+; GFX10-NEXT:    s_or_b32 s11, s5, exec_lo
+; GFX10-NEXT:    s_or_b32 s10, s10, exec_lo
 ; GFX10-NEXT:    v_add_co_u32 v5, vcc_lo, v1, v5
 ; GFX10-NEXT:    v_add_co_ci_u32_e32 v6, vcc_lo, v2, v6, vcc_lo
-; GFX10-NEXT:    s_or_b32 s10, s10, s12
 ; GFX10-NEXT:    global_load_dword v5, v[5:6], off
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v5
@@ -284,8 +280,7 @@ define void @divergent_i1_xor_used_outside_loop_larger_loop_body(i32 %num.elts,
 ; GFX10-NEXT:    s_andn2_b32 s10, s10, exec_lo
 ; GFX10-NEXT:    s_add_i32 s4, s4, 1
 ; GFX10-NEXT:    s_andn2_b32 s11, s11, exec_lo
-; GFX10-NEXT:    s_and_b32 s12, exec_lo, vcc_lo
-; GFX10-NEXT:    s_or_b32 s10, s10, s12
+; GFX10-NEXT:    s_or_b32 s10, s10, vcc_lo
 ; GFX10-NEXT:    s_branch .LBB4_2
 ; GFX10-NEXT:  .LBB4_5: ; %loop.exit.guard
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s8
@@ -342,58 +337,57 @@ define void @divergent_i1_icmp_used_outside_loop(i32 %v0, i32 %v1, ptr addrspace
 ; GFX10-LABEL: divergent_i1_icmp_used_outside_loop:
 ; GFX10:       ; %bb.0: ; %entry
 ; GFX10-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX10-NEXT:    s_mov_b32 s4, 0
 ; GFX10-NEXT:    s_mov_b32 s6, 0
-; GFX10-NEXT:    ; implicit-def: $sgpr5
+; GFX10-NEXT:    s_mov_b32 s5, 0
+; GFX10-NEXT:    ; implicit-def: $sgpr4
 ; GFX10-NEXT:    s_branch .LBB5_2
 ; GFX10-NEXT:  .LBB5_1: ; %Flow
 ; GFX10-NEXT:    ; in Loop: Header=BB5_2 Depth=1
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s9
-; GFX10-NEXT:    s_and_b32 s5, exec_lo, s5
-; GFX10-NEXT:    s_or_b32 s6, s5, s6
-; GFX10-NEXT:    s_or_b32 s5, s7, s8
-; GFX10-NEXT:    s_andn2_b32 exec_lo, exec_lo, s6
+; GFX10-NEXT:    s_and_b32 s4, exec_lo, s7
+; GFX10-NEXT:    s_or_b32 s5, s4, s5
+; GFX10-NEXT:    s_or_b32 s4, s8, vcc_lo
+; GFX10-NEXT:    s_andn2_b32 exec_lo, exec_lo, s5
 ; GFX10-NEXT:    s_cbranch_execz .LBB5_6
 ; GFX10-NEXT:  .LBB5_2: ; %cond.block.0
 ; GFX10-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX10-NEXT:    v_cmp_eq_u32_e32 vcc_lo, s4, v0
-; GFX10-NEXT:    v_mov_b32_e32 v4, s4
-; GFX10-NEXT:    s_andn2_b32 s7, s5, exec_lo
-; GFX10-NEXT:    s_and_b32 s8, exec_lo, vcc_lo
+; GFX10-NEXT:    v_mov_b32_e32 v4, s6
+; GFX10-NEXT:    v_cmp_eq_u32_e32 vcc_lo, s6, v0
+; GFX10-NEXT:    s_andn2_b32 s8, s4, exec_lo
 ; GFX10-NEXT:    s_and_saveexec_b32 s9, vcc_lo
 ; GFX10-NEXT:    s_cbranch_execz .LBB5_4
 ; GFX10-NEXT:  ; %bb.3: ; %if.block.0
 ; GFX10-NEXT:    ; in Loop: Header=BB5_2 Depth=1
-; GFX10-NEXT:    s_ashr_i32 s5, s4, 31
-; GFX10-NEXT:    v_mov_b32_e32 v5, s4
-; GFX10-NEXT:    s_lshl_b64 s[10:11], s[4:5], 2
+; GFX10-NEXT:    s_ashr_i32 s7, s6, 31
+; GFX10-NEXT:    v_mov_b32_e32 v5, s6
+; GFX10-NEXT:    s_lshl_b64 s[10:11], s[6:7], 2
 ; GFX10-NEXT:    v_mov_b32_e32 v8, s10
 ; GFX10-NEXT:    v_mov_b32_e32 v9, s11
-; GFX10-NEXT:    v_add_co_u32 v8, vcc_lo, v2, v8
-; GFX10-NEXT:    v_add_co_ci_u32_e32 v9, vcc_lo, v3, v9, vcc_lo
+; GFX10-NEXT:    v_add_co_u32 v8, s4, v2, v8
+; GFX10-NEXT:    v_add_co_ci_u32_e64 v9, s4, v3, v9, s4
 ; GFX10-NEXT:    global_store_dword v[8:9], v5, off
 ; GFX10-NEXT:  .LBB5_4: ; %loop.break.block
 ; GFX10-NEXT:    ; in Loop: Header=BB5_2 Depth=1
 ; GFX10-NEXT:    s_waitcnt_depctr depctr_vm_vsrc(0)
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s9
-; GFX10-NEXT:    v_cmp_ne_u32_e32 vcc_lo, s4, v1
-; GFX10-NEXT:    s_mov_b32 s5, exec_lo
-; GFX10-NEXT:    s_and_saveexec_b32 s9, vcc_lo
+; GFX10-NEXT:    v_cmp_ne_u32_e64 s4, s6, v1
+; GFX10-NEXT:    s_mov_b32 s7, exec_lo
+; GFX10-NEXT:    s_and_saveexec_b32 s9, s4
 ; GFX10-NEXT:    s_cbranch_execz .LBB5_1
 ; GFX10-NEXT:  ; %bb.5: ; %loop.cond
 ; GFX10-NEXT:    ; in Loop: Header=BB5_2 Depth=1
-; GFX10-NEXT:    s_add_i32 s4, s4, 1
-; GFX10-NEXT:    s_andn2_b32 s5, s5, exec_lo
+; GFX10-NEXT:    s_add_i32 s6, s6, 1
+; GFX10-NEXT:    s_andn2_b32 s7, s7, exec_lo
 ; GFX10-NEXT:    s_branch .LBB5_1
 ; GFX10-NEXT:  .LBB5_6: ; %cond.block.1
-; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s6
-; GFX10-NEXT:    s_and_saveexec_b32 s4, s5
+; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s5
+; GFX10-NEXT:    s_and_saveexec_b32 s5, s4
 ; GFX10-NEXT:    s_cbranch_execz .LBB5_8
 ; GFX10-NEXT:  ; %bb.7: ; %if.block.1
 ; GFX10-NEXT:    global_store_dword v[6:7], v4, off
 ; GFX10-NEXT:  .LBB5_8: ; %exit
 ; GFX10-NEXT:    s_waitcnt_depctr depctr_vm_vsrc(0)
-; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s4
+; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s5
 ; GFX10-NEXT:    s_setpc_b64 s[30:31]
 entry:
   br label %loop.start
@@ -489,8 +483,7 @@ define amdgpu_ps void @divergent_i1_freeze_used_outside_loop(i32 %n, ptr addrspa
 ; GFX10-NEXT:    global_load_dword v5, v[5:6], off
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 0, v5
-; GFX10-NEXT:    s_and_b32 s4, exec_lo, vcc_lo
-; GFX10-NEXT:    s_or_b32 s4, s1, s4
+; GFX10-NEXT:    s_or_b32 s4, s1, vcc_lo
 ; GFX10-NEXT:    ; implicit-def: $sgpr1
 ; GFX10-NEXT:    s_branch .LBB6_1
 ; GFX10-NEXT:  .LBB6_4: ; %exit
@@ -531,16 +524,16 @@ define amdgpu_cs void @loop_with_1break(ptr addrspace(1) %x, ptr addrspace(1) %a
 ; GFX10:       ; %bb.0: ; %entry
 ; GFX10-NEXT:    s_mov_b32 s1, -1
 ; GFX10-NEXT:    s_mov_b32 s0, 0
-; GFX10-NEXT:    ; implicit-def: $sgpr3
 ; GFX10-NEXT:    ; implicit-def: $sgpr4
+; GFX10-NEXT:    ; implicit-def: $sgpr3
 ; GFX10-NEXT:    ; implicit-def: $sgpr2
 ; GFX10-NEXT:    s_branch .LBB7_2
 ; GFX10-NEXT:  .LBB7_1: ; %Flow
 ; GFX10-NEXT:    ; in Loop: Header=BB7_2 Depth=1
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s5
-; GFX10-NEXT:    s_and_b32 s5, exec_lo, s3
+; GFX10-NEXT:    s_and_b32 s5, exec_lo, s4
 ; GFX10-NEXT:    s_andn2_b32 s2, s2, exec_lo
-; GFX10-NEXT:    s_and_b32 s6, exec_lo, s4
+; GFX10-NEXT:    s_and_b32 s6, exec_lo, s3
 ; GFX10-NEXT:    s_or_b32 s0, s5, s0
 ; GFX10-NEXT:    s_or_b32 s2, s2, s6
 ; GFX10-NEXT:    s_andn2_b32 exec_lo, exec_lo, s0
@@ -548,12 +541,10 @@ define amdgpu_cs void @loop_with_1break(ptr addrspace(1) %x, ptr addrspace(1) %a
 ; GFX10-NEXT:  .LBB7_2: ; %A
 ; GFX10-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX10-NEXT:    global_load_dword v6, v[2:3], off
-; GFX10-NEXT:    s_andn2_b32 s4, s4, exec_lo
-; GFX10-NEXT:    s_and_b32 s5, exec_lo, exec_lo
 ; GFX10-NEXT:    s_andn2_b32 s3, s3, exec_lo
-; GFX10-NEXT:    s_and_b32 s6, exec_lo, exec_lo
-; GFX10-NEXT:    s_or_b32 s4, s4, s5
-; GFX10-NEXT:    s_or_b32 s3, s3, s6
+; GFX10-NEXT:    s_andn2_b32 s4, s4, exec_lo
+; GFX10-NEXT:    s_or_b32 s3, s3, exec_lo
+; GFX10-NEXT:    s_or_b32 s4, s4, exec_lo
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v6
 ; GFX10-NEXT:    s_and_saveexec_b32 s5, vcc_lo
@@ -566,10 +557,10 @@ define amdgpu_cs void @loop_with_1break(ptr addrspace(1) %x, ptr addrspace(1) %a
 ; GFX10-NEXT:    s_cmpk_lt_u32 s1, 0x64
 ; GFX10-NEXT:    v_add_co_ci_u32_e32 v3, vcc_lo, 0, v3, vcc_lo
 ; GFX10-NEXT:    s_cselect_b32 s6, exec_lo, 0
-; GFX10-NEXT:    s_andn2_b32 s3, s3, exec_lo
-; GFX10-NEXT:    s_and_b32 s6, exec_lo, s6
 ; GFX10-NEXT:    s_andn2_b32 s4, s4, exec_lo
-; GFX10-NEXT:    s_or_b32 s3, s3, s6
+; GFX10-NEXT:    s_and_b32 s6, exec_lo, s6
+; GFX10-NEXT:    s_andn2_b32 s3, s3, exec_lo
+; GFX10-NEXT:    s_or_b32 s4, s4, s6
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_add_nc_u32_e32 v6, 1, v6
 ; GFX10-NEXT:    global_store_dword v[0:1], v6, off
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll
index 623e3a0e5369c48..fa7138473981bc7 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-structurizer.ll
@@ -13,8 +13,7 @@ define amdgpu_ps void @divergent_i1_phi_if_then(ptr addrspace(1) %out, i32 %tid,
 ; GFX10-NEXT:  ; %bb.1: ; %B
 ; GFX10-NEXT:    v_cmp_gt_u32_e32 vcc_lo, 1, v2
 ; GFX10-NEXT:    s_andn2_b32 s0, s0, exec_lo
-; GFX10-NEXT:    s_and_b32 s2, exec_lo, vcc_lo
-; GFX10-NEXT:    s_or_b32 s0, s0, s2
+; GFX10-NEXT:    s_or_b32 s0, s0, vcc_lo
 ; GFX10-NEXT:  ; %bb.2: ; %exit
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s1
 ; GFX10-NEXT:    v_cndmask_b32_e64 v2, 0, -1, s0
@@ -51,15 +50,13 @@ define amdgpu_ps void @divergent_i1_phi_if_else(ptr addrspace(1) %out, i32 %tid,
 ; GFX10-NEXT:    v_cmp_gt_u32_e32 vcc_lo, 2, v2
 ; GFX10-NEXT:    s_andn2_b32 s0, s0, exec_lo
 ; GFX10-NEXT:    ; implicit-def: $vgpr2
-; GFX10-NEXT:    s_and_b32 s2, exec_lo, vcc_lo
-; GFX10-NEXT:    s_or_b32 s0, s0, s2
+; GFX10-NEXT:    s_or_b32 s0, s0, vcc_lo
 ; GFX10-NEXT:  ; %bb.2: ; %Flow
 ; GFX10-NEXT:    s_andn2_saveexec_b32 s1, s1
 ; GFX10-NEXT:  ; %bb.3: ; %A
 ; GFX10-NEXT:    v_cmp_le_u32_e32 vcc_lo, 1, v2
 ; GFX10-NEXT:    s_andn2_b32 s0, s0, exec_lo
-; GFX10-NEXT:    s_and_b32 s2, exec_lo, vcc_lo
-; GFX10-NEXT:    s_or_b32 s0, s0, s2
+; GFX10-NEXT:    s_or_b32 s0, s0, vcc_lo
 ; GFX10-NEXT:  ; %bb.4: ; %exit
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s1
 ; GFX10-NEXT:    v_cndmask_b32_e64 v2, 0, -1, s0
@@ -121,8 +118,7 @@ define amdgpu_cs void @loop_with_1break(ptr addrspace(1) %x, ptr addrspace(1) %a
 ; GFX10-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX10-NEXT:    global_load_dword v4, v[2:3], off
 ; GFX10-NEXT:    s_andn2_b32 s2, s2, exec_lo
-; GFX10-NEXT:    s_and_b32 s3, exec_lo, exec_lo
-; GFX10-NEXT:    s_or_b32 s2, s2, s3
+; GFX10-NEXT:    s_or_b32 s2, s2, exec_lo
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v4
 ; GFX10-NEXT:    s_and_saveexec_b32 s3, vcc_lo
@@ -193,8 +189,7 @@ define amdgpu_cs void @loop_with_2breaks(ptr addrspace(1) %x, ptr addrspace(1) %
 ; GFX10-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX10-NEXT:    global_load_dword v6, v[2:3], off
 ; GFX10-NEXT:    s_andn2_b32 s2, s2, exec_lo
-; GFX10-NEXT:    s_and_b32 s3, exec_lo, exec_lo
-; GFX10-NEXT:    s_or_b32 s2, s2, s3
+; GFX10-NEXT:    s_or_b32 s2, s2, exec_lo
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v6
 ; GFX10-NEXT:    s_and_saveexec_b32 s3, vcc_lo
@@ -287,8 +282,7 @@ define amdgpu_cs void @loop_with_3breaks(ptr addrspace(1) %x, ptr addrspace(1) %
 ; GFX10-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX10-NEXT:    global_load_dword v8, v[2:3], off
 ; GFX10-NEXT:    s_andn2_b32 s2, s2, exec_lo
-; GFX10-NEXT:    s_and_b32 s3, exec_lo, exec_lo
-; GFX10-NEXT:    s_or_b32 s2, s2, s3
+; GFX10-NEXT:    s_or_b32 s2, s2, exec_lo
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v8
 ; GFX10-NEXT:    s_and_saveexec_b32 s3, vcc_lo
@@ -376,16 +370,16 @@ define amdgpu_cs void @loop_with_div_break_with_body(ptr addrspace(1) %x, ptr ad
 ; GFX10:       ; %bb.0: ; %entry
 ; GFX10-NEXT:    s_mov_b32 s1, -1
 ; GFX10-NEXT:    s_mov_b32 s0, 0
-; GFX10-NEXT:    ; implicit-def: $sgpr3
 ; GFX10-NEXT:    ; implicit-def: $sgpr4
+; GFX10-NEXT:    ; implicit-def: $sgpr3
 ; GFX10-NEXT:    ; implicit-def: $sgpr2
 ; GFX10-NEXT:    s_branch .LBB5_2
 ; GFX10-NEXT:  .LBB5_1: ; %Flow
 ; GFX10-NEXT:    ; in Loop: Header=BB5_2 Depth=1
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s5
-; GFX10-NEXT:    s_and_b32 s5, exec_lo, s3
+; GFX10-NEXT:    s_and_b32 s5, exec_lo, s4
 ; GFX10-NEXT:    s_andn2_b32 s2, s2, exec_lo
-; GFX10-NEXT:    s_and_b32 s6, exec_lo, s4
+; GFX10-NEXT:    s_and_b32 s6, exec_lo, s3
 ; GFX10-NEXT:    s_or_b32 s0, s5, s0
 ; GFX10-NEXT:    s_or_b32 s2, s2, s6
 ; GFX10-NEXT:    s_andn2_b32 exec_lo, exec_lo, s0
@@ -393,12 +387,10 @@ define amdgpu_cs void @loop_with_div_break_with_body(ptr addrspace(1) %x, ptr ad
 ; GFX10-NEXT:  .LBB5_2: ; %A
 ; GFX10-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX10-NEXT:    global_load_dword v6, v[2:3], off
-; GFX10-NEXT:    s_andn2_b32 s4, s4, exec_lo
-; GFX10-NEXT:    s_and_b32 s5, exec_lo, exec_lo
 ; GFX10-NEXT:    s_andn2_b32 s3, s3, exec_lo
-; GFX10-NEXT:    s_and_b32 s6, exec_lo, exec_lo
-; GFX10-NEXT:    s_or_b32 s4, s4, s5
-; GFX10-NEXT:    s_or_b32 s3, s3, s6
+; GFX10-NEXT:    s_andn2_b32 s4, s4, exec_lo
+; GFX10-NEXT:    s_or_b32 s3, s3, exec_lo
+; GFX10-NEXT:    s_or_b32 s4, s4, exec_lo
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v6
 ; GFX10-NEXT:    s_and_saveexec_b32 s5, vcc_lo
@@ -411,10 +403,10 @@ define amdgpu_cs void @loop_with_div_break_with_body(ptr addrspace(1) %x, ptr ad
 ; GFX10-NEXT:    s_cmpk_lt_u32 s1, 0x64
 ; GFX10-NEXT:    v_add_co_ci_u32_e32 v3, vcc_lo, 0, v3, vcc_lo
 ; GFX10-NEXT:    s_cselect_b32 s6, exec_lo, 0
-; GFX10-NEXT:    s_andn2_b32 s3, s3, exec_lo
-; GFX10-NEXT:    s_and_b32 s6, exec_lo, s6
 ; GFX10-NEXT:    s_andn2_b32 s4, s4, exec_lo
-; GFX10-NEXT:    s_or_b32 s3, s3, s6
+; GFX10-NEXT:    s_and_b32 s6, exec_lo, s6
+; GFX10-NEXT:    s_andn2_b32 s3, s3, exec_lo
+; GFX10-NEXT:    s_or_b32 s4, s4, s6
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_add_nc_u32_e32 v6, 1, v6
 ; GFX10-NEXT:    global_store_dword v[0:1], v6, off
@@ -496,51 +488,50 @@ define amdgpu_ps i32 @irreducible_cfg(i32 %x, i32 %y, i32 %a0, i32 %a1, i32 %a2,
 ; GFX10-NEXT:    s_and_b32 s4, exec_lo, s4
 ; GFX10-NEXT:    s_mov_b32 s0, exec_lo
 ; GFX10-NEXT:    s_or_b32 s1, s4, s1
-; GFX10-NEXT:    s_or_b32 s3, s3, s6
+; GFX10-NEXT:    s_or_b32 s3, s3, s5
 ; GFX10-NEXT:    s_andn2_b32 exec_lo, exec_lo, s1
 ; GFX10-NEXT:    s_cbranch_execz .LBB6_8
 ; GFX10-NEXT:  .LBB6_2: ; %irr.guard
 ; GFX10-NEXT:    ; =>This Loop Header: Depth=1
 ; GFX10-NEXT:    ; Child Loop BB6_6 Depth 2
-; GFX10-NEXT:    s_mov_b32 s5, exec_lo
+; GFX10-NEXT:    s_mov_b32 s6, exec_lo
 ; GFX10-NEXT:    s_and_saveexec_b32 s4, s0
 ; GFX10-NEXT:    s_xor_b32 s4, exec_lo, s4
 ; GFX10-NEXT:  ; %bb.3: ; %.loopexit
 ; GFX10-NEXT:    ; in Loop: Header=BB6_2 Depth=1
 ; GFX10-NEXT:    v_cmp_gt_i32_e64 s0, v5, v0
-; GFX10-NEXT:    s_mov_b32 s6, exec_lo
+; GFX10-NEXT:    s_mov_b32 s5, exec_lo
 ; GFX10-NEXT:    s_mov_b32 s7, exec_lo
-; GFX10-NEXT:    s_xor_b32 s6, vcc_lo, s6
+; GFX10-NEXT:    s_xor_b32 s5, vcc_lo, s5
 ; GFX10-NEXT:    s_andn2_b32 s2, s2, exec_lo
-; GFX10-NEXT:    s_or_b32 s6, s0, s6
-; GFX10-NEXT:    s_and_b32 s0, exec_lo, s0
-; GFX10-NEXT:    s_xor_b32 s6, s6, s7
-; GFX10-NEXT:    s_andn2_b32 s5, s5, exec_lo
-; GFX10-NEXT:    s_and_b32 s6, exec_lo, s6
+; GFX10-NEXT:    s_or_b32 s5, s0, s5
+; GFX10-NEXT:    s_andn2_b32 s6, s6, exec_lo
+; GFX10-NEXT:    s_xor_b32 s5, s5, s7
 ; GFX10-NEXT:    s_or_b32 s2, s2, s0
-; GFX10-NEXT:    s_or_b32 s5, s5, s6
+; GFX10-NEXT:    s_and_b32 s5, exec_lo, s5
+; GFX10-NEXT:    s_or_b32 s6, s6, s5
 ; GFX10-NEXT:  ; %bb.4: ; %Flow1
 ; GFX10-NEXT:    ; in Loop: Header=BB6_2 Depth=1
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s4
 ; GFX10-NEXT:    s_mov_b32 s4, exec_lo
 ; GFX10-NEXT:    s_andn2_b32 s3, s3, exec_lo
-; GFX10-NEXT:    s_and_b32 s6, exec_lo, s2
-; GFX10-NEXT:    s_and_saveexec_b32 s7, s5
+; GFX10-NEXT:    s_and_b32 s5, exec_lo, s2
+; GFX10-NEXT:    s_and_saveexec_b32 s7, s6
 ; GFX10-NEXT:    s_cbranch_execz .LBB6_1
 ; GFX10-NEXT:  ; %bb.5: ; %.preheader
 ; GFX10-NEXT:    ; in Loop: Header=BB6_2 Depth=1
 ; GFX10-NEXT:    v_cmp_le_i32_e64 s0, v4, v0
-; GFX10-NEXT:    s_mov_b32 s5, 0
+; GFX10-NEXT:    s_mov_b32 s6, 0
 ; GFX10-NEXT:  .LBB6_6: ; %.inner_loop
 ; GFX10-NEXT:    ; Parent Loop BB6_2 Depth=1
 ; GFX10-NEXT:    ; => This Inner Loop Header: Depth=2
 ; GFX10-NEXT:    s_and_b32 s8, exec_lo, s0
-; GFX10-NEXT:    s_or_b32 s5, s8, s5
-; GFX10-NEXT:    s_andn2_b32 exec_lo, exec_lo, s5
+; GFX10-NEXT:    s_or_b32 s6, s8, s6
+; GFX10-NEXT:    s_andn2_b32 exec_lo, exec_lo, s6
 ; GFX10-NEXT:    s_cbranch_execnz .LBB6_6
 ; GFX10-NEXT:  ; %bb.7: ; %Flow
 ; GFX10-NEXT:    ; in Loop: Header=BB6_2 Depth=1
-; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s5
+; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s6
 ; GFX10-NEXT:    s_andn2_b32 s4, s4, exec_lo
 ; GFX10-NEXT:    s_branch .LBB6_1
 ; GFX10-NEXT:  .LBB6_8: ; %.exit
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-temporal-divergent-i1.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-temporal-divergent-i1.ll
index 31745ec365e50db..2ad5affd97cd729 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-temporal-divergent-i1.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergence-temporal-divergent-i1.ll
@@ -117,8 +117,7 @@ define amdgpu_cs void @loop_with_1break(ptr addrspace(1) %x, i32 %x.size, ptr ad
 ; GFX10-NEXT:    v_cmp_lt_u32_e32 vcc_lo, s4, v2
 ; GFX10-NEXT:    s_add_i32 s4, s4, 1
 ; GFX10-NEXT:    global_load_dword v6, v[4:5], off
-; GFX10-NEXT:    s_and_b32 s7, exec_lo, vcc_lo
-; GFX10-NEXT:    s_or_b32 s10, s6, s7
+; GFX10-NEXT:    s_or_b32 s10, s6, vcc_lo
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_add_nc_u32_e32 v6, 1, v6
 ; GFX10-NEXT:    global_store_dword v[4:5], v6, off
@@ -146,8 +145,7 @@ define amdgpu_cs void @loop_with_1break(ptr addrspace(1) %x, i32 %x.size, ptr ad
 ; GFX10-NEXT:    s_cmp_lg_u32 s5, 0
 ; GFX10-NEXT:    s_cselect_b32 s5, 1, 0
 ; GFX10-NEXT:    s_andn2_b32 s10, s10, exec_lo
-; GFX10-NEXT:    s_and_b32 s11, exec_lo, exec_lo
-; GFX10-NEXT:    s_or_b32 s10, s10, s11
+; GFX10-NEXT:    s_or_b32 s10, s10, exec_lo
 ; GFX10-NEXT:    s_cmp_lg_u32 s5, 0
 ; GFX10-NEXT:    s_cbranch_scc1 .LBB2_1
 ; GFX10-NEXT:  ; %bb.4: ; in Loop: Header=BB2_3 Depth=1
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergent-control-flow.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergent-control-flow.ll
index e32f09a38c83b04..d19aa391443e042 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/divergent-control-flow.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/divergent-control-flow.ll
@@ -227,8 +227,7 @@ define amdgpu_kernel void @break_loop(i32 %arg) {
 ; CHECK-NEXT:    s_cmp_ge_i32 s4, 0
 ; CHECK-NEXT:    s_cselect_b32 s5, 1, 0
 ; CHECK-NEXT:    s_andn2_b64 s[2:3], s[2:3], exec
-; CHECK-NEXT:    s_and_b64 s[6:7], exec, exec
-; CHECK-NEXT:    s_or_b64 s[2:3], s[2:3], s[6:7]
+; CHECK-NEXT:    s_or_b64 s[2:3], s[2:3], exec
 ; CHECK-NEXT:    s_cmp_lg_u32 s5, 0
 ; CHECK-NEXT:    s_cbranch_scc1 .LBB5_1
 ; CHECK-NEXT:  ; %bb.3: ; %bb4
@@ -237,8 +236,7 @@ define amdgpu_kernel void @break_loop(i32 %arg) {
 ; CHECK-NEXT:    s_waitcnt vmcnt(0)
 ; CHECK-NEXT:    s_andn2_b64 s[2:3], s[2:3], exec
 ; CHECK-NEXT:    v_cmp_ge_i32_e32 vcc, v0, v1
-; CHECK-NEXT:    s_and_b64 s[6:7], exec, vcc
-; CHECK-NEXT:    s_or_b64 s[2:3], s[2:3], s[6:7]
+; CHECK-NEXT:    s_or_b64 s[2:3], s[2:3], vcc
 ; CHECK-NEXT:    s_branch .LBB5_1
 ; CHECK-NEXT:  .LBB5_4: ; %bb9
 ; CHECK-NEXT:    s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/is-safe-to-sink-bug.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/is-safe-to-sink-bug.ll
index d039334ab63dce7..b1af15ac1613a67 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/is-safe-to-sink-bug.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/is-safe-to-sink-bug.ll
@@ -47,11 +47,10 @@ define amdgpu_ps void @_amdgpu_ps_main(i1 %arg) {
 ; CHECK-NEXT:    s_or_b32 s4, s3, 1
 ; CHECK-NEXT:    s_and_b32 s0, s0, s2
 ; CHECK-NEXT:    s_cmp_lt_i32 s3, 0
+; CHECK-NEXT:    s_mov_b32 s3, s4
 ; CHECK-NEXT:    s_cselect_b32 s5, 1, 0
 ; CHECK-NEXT:    s_andn2_b32 s2, s2, exec_lo
-; CHECK-NEXT:    s_and_b32 s3, exec_lo, s0
-; CHECK-NEXT:    s_or_b32 s2, s2, s3
-; CHECK-NEXT:    s_mov_b32 s3, s4
+; CHECK-NEXT:    s_or_b32 s2, s2, s0
 ; CHECK-NEXT:    s_cmp_lg_u32 s5, 0
 ; CHECK-NEXT:    s_cbranch_scc0 .LBB0_1
 ; CHECK-NEXT:  .LBB0_4: ; %bb2
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.ballot.i32.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.ballot.i32.ll
index c1742bff4b06f8d..680d4876d744c3c 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.ballot.i32.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.ballot.i32.ll
@@ -418,15 +418,13 @@ define amdgpu_ps void @non_cst_non_compare_input(ptr addrspace(1) %out, i32 %tid
 ; GFX10-NEXT:    v_cmp_gt_u32_e32 vcc_lo, 2, v2
 ; GFX10-NEXT:    s_andn2_b32 s0, s0, exec_lo
 ; GFX10-NEXT:    ; implicit-def: $vgpr2
-; GFX10-NEXT:    s_and_b32 s2, exec_lo, vcc_lo
-; GFX10-NEXT:    s_or_b32 s0, s0, s2
+; GFX10-NEXT:    s_or_b32 s0, s0, vcc_lo
 ; GFX10-NEXT:  ; %bb.2: ; %Flow
 ; GFX10-NEXT:    s_andn2_saveexec_b32 s1, s1
 ; GFX10-NEXT:  ; %bb.3: ; %A
 ; GFX10-NEXT:    v_cmp_le_u32_e32 vcc_lo, 1, v2
 ; GFX10-NEXT:    s_andn2_b32 s0, s0, exec_lo
-; GFX10-NEXT:    s_and_b32 s2, exec_lo, vcc_lo
-; GFX10-NEXT:    s_or_b32 s0, s0, s2
+; GFX10-NEXT:    s_or_b32 s0, s0, vcc_lo
 ; GFX10-NEXT:  ; %bb.4: ; %exit
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s1
 ; GFX10-NEXT:    s_and_b32 s0, s0, exec_lo
@@ -446,15 +444,13 @@ define amdgpu_ps void @non_cst_non_compare_input(ptr addrspace(1) %out, i32 %tid
 ; GFX11-NEXT:    v_cmp_gt_u32_e32 vcc_lo, 2, v2
 ; GFX11-NEXT:    s_and_not1_b32 s0, s0, exec_lo
 ; GFX11-NEXT:    ; implicit-def: $vgpr2
-; GFX11-NEXT:    s_and_b32 s2, exec_lo, vcc_lo
-; GFX11-NEXT:    s_or_b32 s0, s0, s2
+; GFX11-NEXT:    s_or_b32 s0, s0, vcc_lo
 ; GFX11-NEXT:  ; %bb.2: ; %Flow
 ; GFX11-NEXT:    s_and_not1_saveexec_b32 s1, s1
 ; GFX11-NEXT:  ; %bb.3: ; %A
 ; GFX11-NEXT:    v_cmp_le_u32_e32 vcc_lo, 1, v2
 ; GFX11-NEXT:    s_and_not1_b32 s0, s0, exec_lo
-; GFX11-NEXT:    s_and_b32 s2, exec_lo, vcc_lo
-; GFX11-NEXT:    s_or_b32 s0, s0, s2
+; GFX11-NEXT:    s_or_b32 s0, s0, vcc_lo
 ; GFX11-NEXT:  ; %bb.4: ; %exit
 ; GFX11-NEXT:    s_or_b32 exec_lo, exec_lo, s1
 ; GFX11-NEXT:    s_and_b32 s0, s0, exec_lo
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.ballot.i64.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.ballot.i64.ll
index 7b726ac5f3a8e6d..702d8969732153e 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.ballot.i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.ballot.i64.ll
@@ -419,16 +419,14 @@ define amdgpu_ps void @non_cst_non_compare_input(ptr addrspace(1) %out, i32 %tid
 ; CHECK-NEXT:  ; %bb.1: ; %B
 ; CHECK-NEXT:    v_cmp_gt_u32_e32 vcc, 2, v2
 ; CHECK-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec
-; CHECK-NEXT:    s_and_b64 s[4:5], exec, vcc
-; CHECK-NEXT:    s_or_b64 s[0:1], s[0:1], s[4:5]
+; CHECK-NEXT:    s_or_b64 s[0:1], s[0:1], vcc
 ; CHECK-NEXT:    ; implicit-def: $vgpr2
 ; CHECK-NEXT:  ; %bb.2: ; %Flow
 ; CHECK-NEXT:    s_andn2_saveexec_b64 s[2:3], s[2:3]
 ; CHECK-NEXT:  ; %bb.3: ; %A
 ; CHECK-NEXT:    v_cmp_le_u32_e32 vcc, 1, v2
 ; CHECK-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec
-; CHECK-NEXT:    s_and_b64 s[4:5], exec, vcc
-; CHECK-NEXT:    s_or_b64 s[0:1], s[0:1], s[4:5]
+; CHECK-NEXT:    s_or_b64 s[0:1], s[0:1], vcc
 ; CHECK-NEXT:  ; %bb.4: ; %exit
 ; CHECK-NEXT:    s_or_b64 exec, exec, s[2:3]
 ; CHECK-NEXT:    s_and_b64 s[0:1], s[0:1], exec
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll
index ef059e0ab3042b4..c2c309ad4bedc65 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll
@@ -489,8 +489,7 @@ define amdgpu_kernel void @v8i8_phi_chain(ptr addrspace(1) %src1, ptr addrspace(
 ; GFX906-NEXT:    global_load_dwordx2 v[1:2], v3, s[10:11]
 ; GFX906-NEXT:    v_cmp_gt_u32_e32 vcc, 7, v0
 ; GFX906-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec
-; GFX906-NEXT:    s_and_b64 s[4:5], exec, vcc
-; GFX906-NEXT:    s_or_b64 s[0:1], s[0:1], s[4:5]
+; GFX906-NEXT:    s_or_b64 s[0:1], s[0:1], vcc
 ; GFX906-NEXT:  .LBB8_2: ; %Flow
 ; GFX906-NEXT:    s_or_b64 exec, exec, s[2:3]
 ; GFX906-NEXT:    s_and_saveexec_b64 s[2:3], s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/add.ll b/llvm/test/CodeGen/AMDGPU/add.ll
index 3a8d9a0a30e1e14..447182a850895a0 100644
--- a/llvm/test/CodeGen/AMDGPU/add.ll
+++ b/llvm/test/CodeGen/AMDGPU/add.ll
@@ -1159,8 +1159,7 @@ define amdgpu_kernel void @add64_in_branch(ptr addrspace(1) %out, ptr addrspace(
 ; GFX6-NEXT:    s_load_dwordx8 s[0:7], s[4:5], 0x9
 ; GFX6-NEXT:    s_mov_b64 s[8:9], 0
 ; GFX6-NEXT:    s_waitcnt lgkmcnt(0)
-; GFX6-NEXT:    v_cmp_ne_u64_e64 s[10:11], s[4:5], 0
-; GFX6-NEXT:    s_and_b64 vcc, exec, s[10:11]
+; GFX6-NEXT:    v_cmp_ne_u64_e64 vcc, s[4:5], 0
 ; GFX6-NEXT:    s_cbranch_vccz .LBB9_2
 ; GFX6-NEXT:  ; %bb.1: ; %else
 ; GFX6-NEXT:    s_add_u32 s4, s4, s6
diff --git a/llvm/test/CodeGen/AMDGPU/add_i1.ll b/llvm/test/CodeGen/AMDGPU/add_i1.ll
index bb200468b763d33..30a2c5519e127d8 100644
--- a/llvm/test/CodeGen/AMDGPU/add_i1.ll
+++ b/llvm/test/CodeGen/AMDGPU/add_i1.ll
@@ -142,8 +142,7 @@ define amdgpu_kernel void @add_i1_cf(ptr addrspace(1) %out, ptr addrspace(1) %a,
 ; GFX9-NEXT:    s_andn2_b64 s[2:3], s[4:5], exec
 ; GFX9-NEXT:    v_and_b32_e32 v0, 1, v0
 ; GFX9-NEXT:    v_cmp_eq_u32_e32 vcc, 1, v0
-; GFX9-NEXT:    s_and_b64 s[4:5], vcc, exec
-; GFX9-NEXT:    s_or_b64 s[4:5], s[2:3], s[4:5]
+; GFX9-NEXT:    s_or_b64 s[4:5], s[2:3], vcc
 ; GFX9-NEXT:  .LBB2_4: ; %endif
 ; GFX9-NEXT:    s_or_b64 exec, exec, s[6:7]
 ; GFX9-NEXT:    s_waitcnt lgkmcnt(0)
@@ -182,8 +181,7 @@ define amdgpu_kernel void @add_i1_cf(ptr addrspace(1) %out, ptr addrspace(1) %a,
 ; GFX10-NEXT:    s_andn2_b32 s2, s4, exec_lo
 ; GFX10-NEXT:    v_and_b32_e32 v0, 1, v0
 ; GFX10-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 1, v0
-; GFX10-NEXT:    s_and_b32 s3, vcc_lo, exec_lo
-; GFX10-NEXT:    s_or_b32 s4, s2, s3
+; GFX10-NEXT:    s_or_b32 s4, s2, vcc_lo
 ; GFX10-NEXT:  .LBB2_4: ; %endif
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s5
 ; GFX10-NEXT:    s_waitcnt lgkmcnt(0)
@@ -223,10 +221,9 @@ define amdgpu_kernel void @add_i1_cf(ptr addrspace(1) %out, ptr addrspace(1) %a,
 ; GFX11-NEXT:    s_waitcnt vmcnt(0)
 ; GFX11-NEXT:    s_and_not1_b32 s2, s6, exec_lo
 ; GFX11-NEXT:    v_and_b32_e32 v0, 1, v0
-; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX11-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 1, v0
-; GFX11-NEXT:    s_and_b32 s3, vcc_lo, exec_lo
-; GFX11-NEXT:    s_or_b32 s6, s2, s3
+; GFX11-NEXT:    s_or_b32 s6, s2, vcc_lo
 ; GFX11-NEXT:  .LBB2_4: ; %endif
 ; GFX11-NEXT:    s_or_b32 exec_lo, exec_lo, s4
 ; GFX11-NEXT:    s_xor_b32 s2, s6, -1
diff --git a/llvm/test/CodeGen/AMDGPU/andorbitset.ll b/llvm/test/CodeGen/AMDGPU/andorbitset.ll
index 459e578437a60dc..03499b184d565dd 100644
--- a/llvm/test/CodeGen/AMDGPU/andorbitset.ll
+++ b/llvm/test/CodeGen/AMDGPU/andorbitset.ll
@@ -122,8 +122,7 @@ define void @bitset_verifier_error_freeze_poison() local_unnamed_addr #0 {
 ; SI-NEXT:    buffer_store_dword v0, off, s[4:7], 0
 ; SI-NEXT:    s_waitcnt expcnt(0)
 ; SI-NEXT:    v_mov_b32_e32 v0, 0x3f7fbe77
-; SI-NEXT:    v_cmp_ge_f32_e64 s[4:5], |s4|, v0
-; SI-NEXT:    s_and_b64 vcc, exec, s[4:5]
+; SI-NEXT:    v_cmp_ge_f32_e64 vcc, |s4|, v0
 ; SI-NEXT:    s_cbranch_vccnz .LBB6_2
 ; SI-NEXT:  ; %bb.1: ; %bb5
 ; SI-NEXT:  .LBB6_2: ; %bb6
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
index a912ceb4b16363d..0cc9f3413453266 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
@@ -688,12 +688,11 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
 ; GFX6-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX6-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX6-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX6-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX6-NEXT:    s_mov_b32 m0, s3
 ; GFX6-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX6-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX6-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX6-NEXT:    s_add_i32 s2, s2, s8
-; GFX6-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX6-NEXT:    s_cbranch_vccnz .LBB2_1
 ; GFX6-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX6-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -1149,12 +1148,11 @@ define amdgpu_kernel void @struct_add_i32_varying_vdata(ptr addrspace(1) %out, p
 ; GFX6-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX6-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX6-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX6-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX6-NEXT:    s_mov_b32 m0, s3
 ; GFX6-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX6-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX6-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX6-NEXT:    s_add_i32 s2, s2, s8
-; GFX6-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX6-NEXT:    s_cbranch_vccnz .LBB3_1
 ; GFX6-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX6-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -2453,12 +2451,11 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
 ; GFX6-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX6-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX6-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX6-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX6-NEXT:    s_mov_b32 m0, s3
 ; GFX6-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX6-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX6-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX6-NEXT:    s_add_i32 s2, s2, s8
-; GFX6-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX6-NEXT:    s_cbranch_vccnz .LBB7_1
 ; GFX6-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX6-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll
index b6125580321bdc5..6f0911bc3cd7f1a 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_global_pointer.ll
@@ -819,10 +819,9 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out, ptr addrspace(
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s7, v0, s2
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[2:3], 1, s2
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[2:3]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[2:3], s[0:1], 0
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s6, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_add_i32 s6, s6, s7
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[2:3]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB2_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x9
@@ -2787,8 +2786,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out, ptr addrspace(
 ; GFX7LESS_ITERATIVE-NEXT:    s_addc_u32 s7, s7, s3
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[2:3], 1, s2
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[2:3]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[2:3], s[0:1], 0
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB5_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x9
@@ -5246,10 +5244,9 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out, ptr addrspace(
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s6, v0, s2
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[2:3], 1, s2
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[2:3]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[2:3], s[0:1], 0
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v2, s12, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_add_i32 s12, s12, s6
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[2:3]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB8_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x9
@@ -7748,8 +7745,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out, ptr addrspace(
 ; GFX7LESS_ITERATIVE-NEXT:    s_addc_u32 s9, s9, s3
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[2:3], 1, s2
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[2:3]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[2:3], s[0:1], 0
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB11_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x9
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll
index 3988eab1a5d7158..c52c68167a56f72 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_local_pointer.ll
@@ -581,12 +581,11 @@ define amdgpu_kernel void @add_i32_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_mov_b32 m0, s3
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_add_i32 s2, s2, s8
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB2_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -2445,8 +2444,7 @@ define amdgpu_kernel void @add_i64_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_addc_u32 s1, s1, s7
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s6
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[2:3], 0
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[2:3], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB6_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -4827,12 +4825,11 @@ define amdgpu_kernel void @sub_i32_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_mov_b32 m0, s3
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_add_i32 s2, s2, s8
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB10_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -6722,8 +6719,7 @@ define amdgpu_kernel void @sub_i64_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_addc_u32 s1, s1, s7
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s6
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[2:3], 0
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[2:3], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB14_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -7770,12 +7766,11 @@ define amdgpu_kernel void @and_i32_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_mov_b32 m0, s3
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_and_b32 s2, s2, s8
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB15_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -8509,11 +8504,10 @@ define amdgpu_kernel void @and_i64_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s6, v0, s8
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[8:9], 1, s8
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[8:9]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[8:9], s[2:3], 0
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[2:3], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v2, s1, m0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s0, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[8:9]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB16_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -9408,12 +9402,11 @@ define amdgpu_kernel void @or_i32_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_mov_b32 m0, s3
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_or_b32 s2, s2, s8
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB17_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -10119,11 +10112,10 @@ define amdgpu_kernel void @or_i64_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s6, v0, s8
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[8:9], 1, s8
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[8:9]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[8:9], s[2:3], 0
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[2:3], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v2, s1, m0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s0, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_or_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[8:9]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB18_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -11018,12 +11010,11 @@ define amdgpu_kernel void @xor_i32_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_mov_b32 m0, s3
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_xor_b32 s2, s2, s8
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB19_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -11729,11 +11720,10 @@ define amdgpu_kernel void @xor_i64_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s6, v0, s8
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[8:9], 1, s8
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[8:9]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[8:9], s[2:3], 0
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[2:3], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v2, s1, m0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s0, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_xor_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[8:9]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB20_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -12628,12 +12618,11 @@ define amdgpu_kernel void @max_i32_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_mov_b32 m0, s3
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_max_i32 s2, s2, s8
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB21_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -13649,8 +13638,7 @@ define amdgpu_kernel void @max_i64_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_cselect_b32 s0, s0, s10
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s8
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[2:3], 0
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[2:3], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB23_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -14833,12 +14821,11 @@ define amdgpu_kernel void @min_i32_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_mov_b32 m0, s3
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_min_i32 s2, s2, s8
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB24_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -15854,8 +15841,7 @@ define amdgpu_kernel void @min_i64_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_cselect_b32 s0, s0, s10
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s8
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[2:3], 0
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[2:3], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB26_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -17038,12 +17024,11 @@ define amdgpu_kernel void @umax_i32_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_mov_b32 m0, s3
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_max_u32 s2, s2, s8
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB27_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -18053,8 +18038,7 @@ define amdgpu_kernel void @umax_i64_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_cselect_b32 s0, s0, s10
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s8
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[2:3], 0
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[2:3], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB29_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -19225,12 +19209,11 @@ define amdgpu_kernel void @umin_i32_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_mov_b32 m0, s3
 ; GFX7LESS_ITERATIVE-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7LESS_ITERATIVE-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX7LESS_ITERATIVE-NEXT:    s_min_u32 s2, s2, s8
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB30_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -20241,8 +20224,7 @@ define amdgpu_kernel void @umin_i64_varying(ptr addrspace(1) %out) {
 ; GFX7LESS_ITERATIVE-NEXT:    s_cselect_b32 s0, s0, s10
 ; GFX7LESS_ITERATIVE-NEXT:    s_lshl_b64 s[6:7], 1, s8
 ; GFX7LESS_ITERATIVE-NEXT:    s_andn2_b64 s[2:3], s[2:3], s[6:7]
-; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[2:3], 0
-; GFX7LESS_ITERATIVE-NEXT:    s_and_b64 vcc, exec, s[6:7]
+; GFX7LESS_ITERATIVE-NEXT:    v_cmp_ne_u64_e64 vcc, s[2:3], 0
 ; GFX7LESS_ITERATIVE-NEXT:    s_cbranch_vccnz .LBB32_1
 ; GFX7LESS_ITERATIVE-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX7LESS_ITERATIVE-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
index e5a5730005f88e7..0d46931f681556b 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
@@ -687,12 +687,11 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
 ; GFX6-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX6-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX6-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX6-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX6-NEXT:    s_mov_b32 m0, s3
 ; GFX6-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX6-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX6-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX6-NEXT:    s_add_i32 s2, s2, s8
-; GFX6-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX6-NEXT:    s_cbranch_vccnz .LBB2_1
 ; GFX6-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX6-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -1964,12 +1963,11 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
 ; GFX6-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX6-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX6-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX6-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX6-NEXT:    s_mov_b32 m0, s3
 ; GFX6-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX6-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX6-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX6-NEXT:    s_add_i32 s2, s2, s8
-; GFX6-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX6-NEXT:    s_cbranch_vccnz .LBB6_1
 ; GFX6-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX6-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
index 0334db94087d54c..c005e46d73ece7f 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
@@ -705,12 +705,11 @@ define amdgpu_kernel void @add_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
 ; GFX6-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX6-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX6-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX6-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX6-NEXT:    s_mov_b32 m0, s3
 ; GFX6-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX6-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX6-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX6-NEXT:    s_add_i32 s2, s2, s8
-; GFX6-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX6-NEXT:    s_cbranch_vccnz .LBB2_1
 ; GFX6-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX6-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
@@ -2163,12 +2162,11 @@ define amdgpu_kernel void @sub_i32_varying_vdata(ptr addrspace(1) %out, ptr addr
 ; GFX6-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX6-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX6-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX6-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX6-NEXT:    s_mov_b32 m0, s3
 ; GFX6-NEXT:    v_readlane_b32 s8, v0, s3
+; GFX6-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX6-NEXT:    v_writelane_b32 v1, s2, m0
 ; GFX6-NEXT:    s_add_i32 s2, s2, s8
-; GFX6-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX6-NEXT:    s_cbranch_vccnz .LBB7_1
 ; GFX6-NEXT:  ; %bb.2: ; %ComputeEnd
 ; GFX6-NEXT:    v_mbcnt_lo_u32_b32_e64 v0, exec_lo, 0
diff --git a/llvm/test/CodeGen/AMDGPU/branch-relaxation.ll b/llvm/test/CodeGen/AMDGPU/branch-relaxation.ll
index b40636776d84fec..71895a2b3797a5a 100644
--- a/llvm/test/CodeGen/AMDGPU/branch-relaxation.ll
+++ b/llvm/test/CodeGen/AMDGPU/branch-relaxation.ll
@@ -232,8 +232,7 @@ define amdgpu_kernel void @uniform_conditional_min_long_forward_vcnd_branch(ptr
 ; GCN:       ; %bb.0: ; %bb0
 ; GCN-NEXT:    s_load_dword s0, s[4:5], 0xb
 ; GCN-NEXT:    s_waitcnt lgkmcnt(0)
-; GCN-NEXT:    v_cmp_eq_f32_e64 s[2:3], s0, 0
-; GCN-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GCN-NEXT:    v_cmp_eq_f32_e64 vcc, s0, 0
 ; GCN-NEXT:    s_cbranch_vccz .LBB2_1
 ; GCN-NEXT:  ; %bb.3: ; %bb0
 ; GCN-NEXT:    s_getpc_b64 s[2:3]
@@ -263,8 +262,7 @@ define amdgpu_kernel void @uniform_conditional_min_long_forward_vcnd_branch(ptr
 ; GFX11:       ; %bb.0: ; %bb0
 ; GFX11-NEXT:    s_load_b32 s0, s[4:5], 0x2c
 ; GFX11-NEXT:    s_waitcnt lgkmcnt(0)
-; GFX11-NEXT:    v_cmp_eq_f32_e64 s[2:3], s0, 0
-; GFX11-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX11-NEXT:    v_cmp_eq_f32_e64 vcc, s0, 0
 ; GFX11-NEXT:    s_cbranch_vccz .LBB2_1
 ; GFX11-NEXT:  ; %bb.3: ; %bb0
 ; GFX11-NEXT:    s_getpc_b64 s[2:3]
diff --git a/llvm/test/CodeGen/AMDGPU/dagcombine-select.ll b/llvm/test/CodeGen/AMDGPU/dagcombine-select.ll
index 29efadfeea1279f..c3b15615e5f2ed2 100644
--- a/llvm/test/CodeGen/AMDGPU/dagcombine-select.ll
+++ b/llvm/test/CodeGen/AMDGPU/dagcombine-select.ll
@@ -591,8 +591,7 @@ define amdgpu_kernel void @frem_constant_sel_constants(ptr addrspace(1) %p, i1 %
 ; GFX9-NEXT:    s_cselect_b64 s[0:1], -1, 0
 ; GFX9-NEXT:    v_cndmask_b32_e64 v1, v0, -4.0, s[0:1]
 ; GFX9-NEXT:    s_mov_b32 s0, 0x40a00000
-; GFX9-NEXT:    v_cmp_ge_f32_e64 s[2:3], |v1|, s0
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ge_f32_e64 vcc, |v1|, s0
 ; GFX9-NEXT:    s_cbranch_vccz .LBB26_2
 ; GFX9-NEXT:  ; %bb.1: ; %frem.else
 ; GFX9-NEXT:    v_mov_b32_e32 v0, 0x40a00000
@@ -613,6 +612,7 @@ define amdgpu_kernel void @frem_constant_sel_constants(ptr addrspace(1) %p, i1 %
 ; GFX9-NEXT:    v_ldexp_f32 v0, v0, 1
 ; GFX9-NEXT:    v_div_scale_f32 v2, s[0:1], v0, v0, 1.0
 ; GFX9-NEXT:    v_div_scale_f32 v3, vcc, 1.0, v0, 1.0
+; GFX9-NEXT:    v_frexp_exp_i32_f32_e32 v1, v1
 ; GFX9-NEXT:    v_rcp_f32_e32 v4, v2
 ; GFX9-NEXT:    v_fma_f32 v5, -v2, v4, 1.0
 ; GFX9-NEXT:    v_fma_f32 v4, v5, v4, v4
@@ -620,40 +620,39 @@ define amdgpu_kernel void @frem_constant_sel_constants(ptr addrspace(1) %p, i1 %
 ; GFX9-NEXT:    v_fma_f32 v6, -v2, v5, v3
 ; GFX9-NEXT:    v_fma_f32 v5, v6, v4, v5
 ; GFX9-NEXT:    v_fma_f32 v2, -v2, v5, v3
-; GFX9-NEXT:    v_div_fmas_f32 v3, v2, v4, v5
-; GFX9-NEXT:    v_frexp_exp_i32_f32_e32 v2, v1
-; GFX9-NEXT:    v_sub_u32_e32 v1, 3, v2
-; GFX9-NEXT:    v_cmp_gt_i32_e32 vcc, 13, v1
+; GFX9-NEXT:    v_div_fmas_f32 v2, v2, v4, v5
+; GFX9-NEXT:    v_sub_u32_e32 v3, 3, v1
 ; GFX9-NEXT:    v_mov_b32_e32 v4, 0x45200000
-; GFX9-NEXT:    v_div_fixup_f32 v3, v3, v0, 1.0
+; GFX9-NEXT:    v_cmp_gt_i32_e32 vcc, 13, v3
+; GFX9-NEXT:    v_div_fixup_f32 v2, v2, v0, 1.0
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB26_7
 ; GFX9-NEXT:  ; %bb.5: ; %frem.loop_body.preheader
-; GFX9-NEXT:    v_sub_u32_e32 v1, 15, v2
+; GFX9-NEXT:    v_sub_u32_e32 v3, 15, v1
 ; GFX9-NEXT:    v_mov_b32_e32 v5, 0x45200000
 ; GFX9-NEXT:  .LBB26_6: ; %frem.loop_body
 ; GFX9-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX9-NEXT:    v_mov_b32_e32 v4, v5
-; GFX9-NEXT:    v_mul_f32_e32 v5, v4, v3
+; GFX9-NEXT:    v_mul_f32_e32 v5, v4, v2
 ; GFX9-NEXT:    v_rndne_f32_e32 v5, v5
 ; GFX9-NEXT:    v_fma_f32 v5, -v5, v0, v4
 ; GFX9-NEXT:    v_add_f32_e32 v6, v5, v0
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v5
-; GFX9-NEXT:    v_add_u32_e32 v1, -12, v1
 ; GFX9-NEXT:    v_cndmask_b32_e32 v5, v5, v6, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v1
+; GFX9-NEXT:    v_add_u32_e32 v3, -12, v3
 ; GFX9-NEXT:    v_ldexp_f32 v5, v5, 12
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v3
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB26_6
 ; GFX9-NEXT:  .LBB26_7: ; %Flow12
-; GFX9-NEXT:    v_add_u32_e32 v1, -11, v1
-; GFX9-NEXT:    v_ldexp_f32 v1, v4, v1
-; GFX9-NEXT:    v_mul_f32_e32 v3, v1, v3
-; GFX9-NEXT:    v_rndne_f32_e32 v3, v3
-; GFX9-NEXT:    v_fma_f32 v1, -v3, v0, v1
-; GFX9-NEXT:    v_add_f32_e32 v0, v1, v0
-; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v1
-; GFX9-NEXT:    v_add_u32_e32 v2, -1, v2
-; GFX9-NEXT:    v_cndmask_b32_e32 v0, v1, v0, vcc
-; GFX9-NEXT:    v_ldexp_f32 v0, v0, v2
+; GFX9-NEXT:    v_add_u32_e32 v3, -11, v3
+; GFX9-NEXT:    v_ldexp_f32 v3, v4, v3
+; GFX9-NEXT:    v_mul_f32_e32 v2, v3, v2
+; GFX9-NEXT:    v_rndne_f32_e32 v2, v2
+; GFX9-NEXT:    v_fma_f32 v2, -v2, v0, v3
+; GFX9-NEXT:    v_add_f32_e32 v0, v2, v0
+; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v2
+; GFX9-NEXT:    v_add_u32_e32 v1, -1, v1
+; GFX9-NEXT:    v_cndmask_b32_e32 v0, v2, v0, vcc
+; GFX9-NEXT:    v_ldexp_f32 v0, v0, v1
 ; GFX9-NEXT:    v_and_b32_e32 v0, 0x7fffffff, v0
 ; GFX9-NEXT:  .LBB26_8: ; %Flow14
 ; GFX9-NEXT:    s_load_dwordx2 s[0:1], s[4:5], 0x24
@@ -671,8 +670,7 @@ define amdgpu_kernel void @frem_constant_sel_constants(ptr addrspace(1) %p, i1 %
 ; GFX942-NEXT:    s_cselect_b64 s[0:1], -1, 0
 ; GFX942-NEXT:    v_cndmask_b32_e64 v1, v0, -4.0, s[0:1]
 ; GFX942-NEXT:    s_mov_b32 s0, 0x40a00000
-; GFX942-NEXT:    v_cmp_ge_f32_e64 s[2:3], |v1|, s0
-; GFX942-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX942-NEXT:    v_cmp_ge_f32_e64 vcc, |v1|, s0
 ; GFX942-NEXT:    s_cbranch_vccz .LBB26_2
 ; GFX942-NEXT:  ; %bb.1: ; %frem.else
 ; GFX942-NEXT:    v_mov_b32_e32 v0, 0x40a00000
@@ -704,8 +702,8 @@ define amdgpu_kernel void @frem_constant_sel_constants(ptr addrspace(1) %p, i1 %
 ; GFX942-NEXT:    v_fmac_f32_e32 v6, v7, v4
 ; GFX942-NEXT:    v_fma_f32 v3, -v3, v6, v5
 ; GFX942-NEXT:    v_div_fmas_f32 v3, v3, v4, v6
-; GFX942-NEXT:    v_cmp_gt_i32_e32 vcc, 13, v1
 ; GFX942-NEXT:    v_div_fixup_f32 v3, v3, v0, 1.0
+; GFX942-NEXT:    v_cmp_gt_i32_e32 vcc, 13, v1
 ; GFX942-NEXT:    v_mov_b32_e32 v4, 0x45200000
 ; GFX942-NEXT:    s_cbranch_vccnz .LBB26_7
 ; GFX942-NEXT:  ; %bb.5: ; %frem.loop_body.preheader
@@ -722,8 +720,8 @@ define amdgpu_kernel void @frem_constant_sel_constants(ptr addrspace(1) %p, i1 %
 ; GFX942-NEXT:    v_add_u32_e32 v1, -12, v1
 ; GFX942-NEXT:    s_nop 0
 ; GFX942-NEXT:    v_cndmask_b32_e32 v5, v5, v6, vcc
-; GFX942-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v1
 ; GFX942-NEXT:    v_ldexp_f32 v5, v5, 12
+; GFX942-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v1
 ; GFX942-NEXT:    s_cbranch_vccnz .LBB26_6
 ; GFX942-NEXT:  .LBB26_7: ; %Flow12
 ; GFX942-NEXT:    v_add_u32_e32 v1, -11, v1
diff --git a/llvm/test/CodeGen/AMDGPU/fneg-modifier-casting.ll b/llvm/test/CodeGen/AMDGPU/fneg-modifier-casting.ll
index 3dc817eb1c884fe..4892b9ab9b5275b 100644
--- a/llvm/test/CodeGen/AMDGPU/fneg-modifier-casting.ll
+++ b/llvm/test/CodeGen/AMDGPU/fneg-modifier-casting.ll
@@ -1688,7 +1688,6 @@ define amdgpu_kernel void @fnge_select_f32_multi_use_regression(float %.i2369) {
 ; GFX7-NEXT:    v_mov_b32_e32 v0, s0
 ; GFX7-NEXT:    v_mul_f32_e64 v0, -s2, v0
 ; GFX7-NEXT:    v_cmp_le_f32_e32 vcc, 0, v0
-; GFX7-NEXT:    s_and_b64 vcc, exec, vcc
 ; GFX7-NEXT:    s_endpgm
 ;
 ; GFX9-LABEL: fnge_select_f32_multi_use_regression:
@@ -1704,7 +1703,6 @@ define amdgpu_kernel void @fnge_select_f32_multi_use_regression(float %.i2369) {
 ; GFX9-NEXT:    v_mov_b32_e32 v0, s0
 ; GFX9-NEXT:    v_mul_f32_e64 v0, -s2, v0
 ; GFX9-NEXT:    v_cmp_le_f32_e32 vcc, 0, v0
-; GFX9-NEXT:    s_and_b64 vcc, exec, vcc
 ; GFX9-NEXT:    s_endpgm
 ;
 ; GFX11-LABEL: fnge_select_f32_multi_use_regression:
@@ -1721,7 +1719,6 @@ define amdgpu_kernel void @fnge_select_f32_multi_use_regression(float %.i2369) {
 ; GFX11-NEXT:    v_mul_f32_e64 v0, -s0, s1
 ; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX11-NEXT:    v_cmp_le_f32_e32 vcc_lo, 0, v0
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, vcc_lo
 ; GFX11-NEXT:    s_endpgm
 .entry:
   %i = fcmp uge float %.i2369, 0.000000e+00
diff --git a/llvm/test/CodeGen/AMDGPU/fold-and-exec-wave32.mir b/llvm/test/CodeGen/AMDGPU/fold-and-exec-wave32.mir
new file mode 100644
index 000000000000000..c7ba9d0e8ace020
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/fold-and-exec-wave32.mir
@@ -0,0 +1,43 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6
+# RUN: llc -mtriple=amdgpu11.00 -run-pass=si-fold-operands -o - %s | FileCheck %s
+
+# Wave32 version of fold-and-exec.mir: a V_CMP already writes 0 to the bits of
+# inactive lanes, so ANDing its result with EXEC_LO is a no-op.
+
+---
+name: vcmp_e64
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: vcmp_e64
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_32_xm0_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: $vcc_lo = COPY [[V_CMP_LT_I32_e64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc_lo
+    %0:sreg_32_xm0_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    %1:sreg_32_xm0_xexec = S_AND_B32 $exec_lo, %0, implicit-def dead $scc
+    $vcc_lo = COPY %1
+    S_ENDPGM 0, implicit $vcc_lo
+...
+
+# Negative test: nothing is known about the lane mask.
+---
+name: unknown_mask
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $sgpr0
+    ; CHECK-LABEL: name: unknown_mask
+    ; CHECK: liveins: $sgpr0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[COPY:%[0-9]+]]:sreg_32_xm0_xexec = COPY $sgpr0
+    ; CHECK-NEXT: [[S_AND_B32_:%[0-9]+]]:sreg_32_xm0_xexec = S_AND_B32 $exec_lo, [[COPY]], implicit-def dead $scc
+    ; CHECK-NEXT: $vcc_lo = COPY [[S_AND_B32_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc_lo
+    %0:sreg_32_xm0_xexec = COPY $sgpr0
+    %1:sreg_32_xm0_xexec = S_AND_B32 $exec_lo, %0, implicit-def dead $scc
+    $vcc_lo = COPY %1
+    S_ENDPGM 0, implicit $vcc_lo
+...
diff --git a/llvm/test/CodeGen/AMDGPU/fold-and-exec.mir b/llvm/test/CodeGen/AMDGPU/fold-and-exec.mir
new file mode 100644
index 000000000000000..fe97720ad9abf48
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/fold-and-exec.mir
@@ -0,0 +1,281 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6
+# RUN: llc -mtriple=amdgpu9.00 -run-pass=si-fold-operands -o - %s | FileCheck %s
+
+# A V_CMP already writes 0 to the bits of inactive lanes, so ANDing its result
+# with EXEC is a no-op.
+
+---
+name: vcmp_e64
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: vcmp_e64
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: $vcc = COPY [[V_CMP_LT_I32_e64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    %1:sreg_64_xexec = S_AND_B64 $exec, %0, implicit-def dead $scc
+    $vcc = COPY %1
+    S_ENDPGM 0, implicit $vcc
+...
+
+# The AND is commutative, so EXEC may be either operand.
+---
+name: vcmp_e64_commuted
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: vcmp_e64_commuted
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: $vcc = COPY [[V_CMP_LT_I32_e64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    %1:sreg_64_xexec = S_AND_B64 %0, $exec, implicit-def dead $scc
+    $vcc = COPY %1
+    S_ENDPGM 0, implicit $vcc
+...
+
+# V_CMP_CLASS is not marked isCompare, but is still a VOPC.
+---
+name: vcmp_class
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0
+    ; CHECK-LABEL: name: vcmp_class
+    ; CHECK: liveins: $vgpr0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_CLASS_F32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_CLASS_F32_e64 0, $vgpr0, 3, implicit $exec
+    ; CHECK-NEXT: $vcc = COPY [[V_CMP_CLASS_F32_e64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc
+    %0:sreg_64_xexec = V_CMP_CLASS_F32_e64 0, $vgpr0, 3, implicit $exec
+    %1:sreg_64_xexec = S_AND_B64 $exec, %0, implicit-def dead $scc
+    $vcc = COPY %1
+    S_ENDPGM 0, implicit $vcc
+...
+
+# Bitwise combinations of lane masks are still masked by EXEC.
+---
+name: and_or_xor_tree
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1, $vgpr2
+    ; CHECK-LABEL: name: and_or_xor_tree
+    ; CHECK: liveins: $vgpr0, $vgpr1, $vgpr2
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_1:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr1, $vgpr2, implicit $exec
+    ; CHECK-NEXT: [[S_OR_B64_:%[0-9]+]]:sreg_64_xexec = S_OR_B64 [[V_CMP_LT_I32_e64_]], [[V_CMP_LT_I32_e64_1]], implicit-def dead $scc
+    ; CHECK-NEXT: [[S_XOR_B64_:%[0-9]+]]:sreg_64_xexec = S_XOR_B64 [[S_OR_B64_]], [[V_CMP_LT_I32_e64_]], implicit-def dead $scc
+    ; CHECK-NEXT: $vcc = COPY [[S_XOR_B64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    %1:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr1, $vgpr2, implicit $exec
+    %2:sreg_64_xexec = S_OR_B64 %0, %1, implicit-def dead $scc
+    %3:sreg_64_xexec = S_XOR_B64 %2, %0, implicit-def dead $scc
+    %4:sreg_64_xexec = S_AND_B64 $exec, %3, implicit-def dead $scc
+    $vcc = COPY %4
+    S_ENDPGM 0, implicit $vcc
+...
+
+# S_AND only needs one of its operands to be masked by EXEC.
+---
+name: and_one_masked_operand
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1, $sgpr0_sgpr1
+    ; CHECK-LABEL: name: and_one_masked_operand
+    ; CHECK: liveins: $vgpr0, $vgpr1, $sgpr0_sgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: [[S_AND_B64_:%[0-9]+]]:sreg_64_xexec = S_AND_B64 [[V_CMP_LT_I32_e64_]], $sgpr0_sgpr1, implicit-def dead $scc
+    ; CHECK-NEXT: $vcc = COPY [[S_AND_B64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    %1:sreg_64_xexec = S_AND_B64 %0, $sgpr0_sgpr1, implicit-def dead $scc
+    %2:sreg_64_xexec = S_AND_B64 $exec, %1, implicit-def dead $scc
+    $vcc = COPY %2
+    S_ENDPGM 0, implicit $vcc
+...
+
+# The AND may write a physical register, as inserted by moveToVALU.
+---
+name: physreg_dst
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: physreg_dst
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: $vcc = COPY [[V_CMP_LT_I32_e64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    $vcc = S_AND_B64 $exec, %0, implicit-def dead $scc
+    S_ENDPGM 0, implicit $vcc
+...
+
+# Negative test: S_OR needs both operands to be masked by EXEC.
+---
+name: or_one_masked_operand
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1, $sgpr0_sgpr1
+    ; CHECK-LABEL: name: or_one_masked_operand
+    ; CHECK: liveins: $vgpr0, $vgpr1, $sgpr0_sgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: [[S_OR_B64_:%[0-9]+]]:sreg_64_xexec = S_OR_B64 [[V_CMP_LT_I32_e64_]], $sgpr0_sgpr1, implicit-def dead $scc
+    ; CHECK-NEXT: [[S_AND_B64_:%[0-9]+]]:sreg_64_xexec = S_AND_B64 $exec, [[S_OR_B64_]], implicit-def dead $scc
+    ; CHECK-NEXT: $vcc = COPY [[S_AND_B64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    %1:sreg_64_xexec = S_OR_B64 %0, $sgpr0_sgpr1, implicit-def dead $scc
+    %2:sreg_64_xexec = S_AND_B64 $exec, %1, implicit-def dead $scc
+    $vcc = COPY %2
+    S_ENDPGM 0, implicit $vcc
+...
+
+# Negative test: the V_CMP is in a different basic block, so we cannot tell
+# which EXEC mask it ran under.
+---
+name: vcmp_in_predecessor
+tracksRegLiveness: true
+body: |
+  ; CHECK-LABEL: name: vcmp_in_predecessor
+  ; CHECK: bb.0:
+  ; CHECK-NEXT:   successors: %bb.1(0x80000000)
+  ; CHECK-NEXT:   liveins: $vgpr0, $vgpr1
+  ; CHECK-NEXT: {{  $}}
+  ; CHECK-NEXT:   [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+  ; CHECK-NEXT: {{  $}}
+  ; CHECK-NEXT: bb.1:
+  ; CHECK-NEXT:   [[S_AND_B64_:%[0-9]+]]:sreg_64_xexec = S_AND_B64 $exec, [[V_CMP_LT_I32_e64_]], implicit-def dead $scc
+  ; CHECK-NEXT:   $vcc = COPY [[S_AND_B64_]]
+  ; CHECK-NEXT:   S_ENDPGM 0, implicit $vcc
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+
+  bb.1:
+    %1:sreg_64_xexec = S_AND_B64 $exec, %0, implicit-def dead $scc
+    $vcc = COPY %1
+    S_ENDPGM 0, implicit $vcc
+...
+
+# Negative test: the SCC defined by the AND is used.
+---
+name: scc_not_dead
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: scc_not_dead
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: [[S_AND_B64_:%[0-9]+]]:sreg_64_xexec = S_AND_B64 $exec, [[V_CMP_LT_I32_e64_]], implicit-def $scc
+    ; CHECK-NEXT: [[S_CSELECT_B32_:%[0-9]+]]:sreg_32 = S_CSELECT_B32 1, 0, implicit $scc
+    ; CHECK-NEXT: $vcc = COPY [[S_AND_B64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc, implicit [[S_CSELECT_B32_]]
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    %1:sreg_64_xexec = S_AND_B64 $exec, %0, implicit-def $scc
+    %2:sreg_32 = S_CSELECT_B32 1, 0, implicit $scc
+    $vcc = COPY %1
+    S_ENDPGM 0, implicit $vcc, implicit %2
+...
+
+# S_ANDN2 is masked by EXEC if its first operand is.
+---
+name: andn2_first_operand_masked
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1, $sgpr0_sgpr1
+    ; CHECK-LABEL: name: andn2_first_operand_masked
+    ; CHECK: liveins: $vgpr0, $vgpr1, $sgpr0_sgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: [[S_ANDN2_B64_:%[0-9]+]]:sreg_64_xexec = S_ANDN2_B64 [[V_CMP_LT_I32_e64_]], $sgpr0_sgpr1, implicit-def dead $scc
+    ; CHECK-NEXT: $vcc = COPY [[S_ANDN2_B64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    %1:sreg_64_xexec = S_ANDN2_B64 %0, $sgpr0_sgpr1, implicit-def dead $scc
+    %2:sreg_64_xexec = S_AND_B64 $exec, %1, implicit-def dead $scc
+    $vcc = COPY %2
+    S_ENDPGM 0, implicit $vcc
+...
+
+# Negative test: the second operand of S_ANDN2 is inverted, so it does not help.
+---
+name: andn2_second_operand_masked
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1, $sgpr0_sgpr1
+    ; CHECK-LABEL: name: andn2_second_operand_masked
+    ; CHECK: liveins: $vgpr0, $vgpr1, $sgpr0_sgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: [[S_ANDN2_B64_:%[0-9]+]]:sreg_64_xexec = S_ANDN2_B64 $sgpr0_sgpr1, [[V_CMP_LT_I32_e64_]], implicit-def dead $scc
+    ; CHECK-NEXT: [[S_AND_B64_:%[0-9]+]]:sreg_64_xexec = S_AND_B64 $exec, [[S_ANDN2_B64_]], implicit-def dead $scc
+    ; CHECK-NEXT: $vcc = COPY [[S_AND_B64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    %1:sreg_64_xexec = S_ANDN2_B64 $sgpr0_sgpr1, %0, implicit-def dead $scc
+    %2:sreg_64_xexec = S_AND_B64 $exec, %1, implicit-def dead $scc
+    $vcc = COPY %2
+    S_ENDPGM 0, implicit $vcc
+...
+
+# Look through a COPY of the lane mask.
+---
+name: copy_of_vcmp
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: copy_of_vcmp
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: $vcc = COPY [[V_CMP_LT_I32_e64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    %1:sreg_64 = COPY %0
+    %2:sreg_64_xexec = S_AND_B64 $exec, %1, implicit-def dead $scc
+    $vcc = COPY %2
+    S_ENDPGM 0, implicit $vcc
+...
+
+# After the fold the result of the AND is replaced by %0, which is then used
+# after the S_OR, so the kill flag on the S_OR must be cleared.
+---
+name: src_used_after_kill
+tracksRegLiveness: true
+body: |
+  bb.0:
+    liveins: $vgpr0, $vgpr1, $sgpr0_sgpr1
+    ; CHECK-LABEL: name: src_used_after_kill
+    ; CHECK: liveins: $vgpr0, $vgpr1, $sgpr0_sgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[V_CMP_LT_I32_e64_:%[0-9]+]]:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: [[S_OR_B64_:%[0-9]+]]:sreg_64_xexec = S_OR_B64 [[V_CMP_LT_I32_e64_]], $sgpr0_sgpr1, implicit-def dead $scc
+    ; CHECK-NEXT: $vcc = COPY [[V_CMP_LT_I32_e64_]]
+    ; CHECK-NEXT: S_ENDPGM 0, implicit $vcc, implicit [[S_OR_B64_]]
+    %0:sreg_64_xexec = V_CMP_LT_I32_e64 $vgpr0, $vgpr1, implicit $exec
+    %1:sreg_64_xexec = S_AND_B64 $exec, %0, implicit-def dead $scc
+    %2:sreg_64_xexec = S_OR_B64 killed %0, $sgpr0_sgpr1, implicit-def dead $scc
+    $vcc = COPY %1
+    S_ENDPGM 0, implicit $vcc, implicit %2
+...
diff --git a/llvm/test/CodeGen/AMDGPU/frem.ll b/llvm/test/CodeGen/AMDGPU/frem.ll
index 3aa5c4cec8f1728..57ea88d0c5c5431 100644
--- a/llvm/test/CodeGen/AMDGPU/frem.ll
+++ b/llvm/test/CodeGen/AMDGPU/frem.ll
@@ -198,8 +198,8 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; CI-NEXT:    v_add_f32_e32 v8, v5, v3
 ; CI-NEXT:    v_cndmask_b32_e32 v5, v5, v8, vcc
 ; CI-NEXT:    v_add_i32_e32 v4, vcc, -11, v4
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; CI-NEXT:    v_ldexp_f32_e64 v5, v5, 11
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; CI-NEXT:    s_cbranch_vccnz .LBB0_6
 ; CI-NEXT:  ; %bb.7: ; %Flow
 ; CI-NEXT:    v_mov_b32_e32 v5, v7
@@ -302,8 +302,8 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; VI-NEXT:    v_add_f32_e32 v8, v5, v3
 ; VI-NEXT:    v_cndmask_b32_e32 v5, v5, v8, vcc
 ; VI-NEXT:    v_add_u32_e32 v4, vcc, -11, v4
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; VI-NEXT:    v_ldexp_f32 v5, v5, 11
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; VI-NEXT:    s_cbranch_vccnz .LBB0_6
 ; VI-NEXT:  ; %bb.7: ; %Flow
 ; VI-NEXT:    v_mov_b32_e32 v5, v7
@@ -396,10 +396,10 @@ define amdgpu_kernel void @frem_f16(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; GFX9-NEXT:    v_fma_f32 v5, -v5, v3, v7
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v5
 ; GFX9-NEXT:    v_add_f32_e32 v8, v5, v3
-; GFX9-NEXT:    v_add_u32_e32 v4, -11, v4
 ; GFX9-NEXT:    v_cndmask_b32_e32 v5, v5, v8, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
+; GFX9-NEXT:    v_add_u32_e32 v4, -11, v4
 ; GFX9-NEXT:    v_ldexp_f32 v5, v5, 11
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB0_6
 ; GFX9-NEXT:  ; %bb.7: ; %Flow
 ; GFX9-NEXT:    v_mov_b32_e32 v5, v7
@@ -2180,8 +2180,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; SI-NEXT:    buffer_load_dword v0, off, s[8:11], 0
 ; SI-NEXT:    buffer_load_dword v1, off, s[4:7], 0 offset:16
 ; SI-NEXT:    s_waitcnt vmcnt(0)
-; SI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v1|
-; SI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; SI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v1|
 ; SI-NEXT:    s_cbranch_vccz .LBB3_2
 ; SI-NEXT:  ; %bb.1: ; %frem.else
 ; SI-NEXT:    v_and_b32_e32 v2, 0x80000000, v0
@@ -2289,8 +2288,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; CI-NEXT:    buffer_load_dword v0, off, s[8:11], 0
 ; CI-NEXT:    buffer_load_dword v1, off, s[4:7], 0 offset:16
 ; CI-NEXT:    s_waitcnt vmcnt(0)
-; CI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v1|
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v1|
 ; CI-NEXT:    s_cbranch_vccz .LBB3_2
 ; CI-NEXT:  ; %bb.1: ; %frem.else
 ; CI-NEXT:    v_and_b32_e32 v2, 0x80000000, v0
@@ -2344,8 +2342,8 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; CI-NEXT:    v_add_f32_e32 v8, v5, v3
 ; CI-NEXT:    v_cndmask_b32_e32 v5, v5, v8, vcc
 ; CI-NEXT:    v_add_i32_e32 v4, vcc, -12, v4
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v4
 ; CI-NEXT:    v_ldexp_f32_e64 v5, v5, 12
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v4
 ; CI-NEXT:    s_cbranch_vccnz .LBB3_6
 ; CI-NEXT:  ; %bb.7: ; %Flow
 ; CI-NEXT:    v_mov_b32_e32 v5, v7
@@ -2387,8 +2385,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; VI-NEXT:    v_mov_b32_e32 v2, s3
 ; VI-NEXT:    flat_load_dword v1, v[1:2]
 ; VI-NEXT:    s_waitcnt vmcnt(0)
-; VI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v1|
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v1|
 ; VI-NEXT:    s_cbranch_vccz .LBB3_2
 ; VI-NEXT:  ; %bb.1: ; %frem.else
 ; VI-NEXT:    v_and_b32_e32 v2, 0x80000000, v0
@@ -2442,8 +2439,8 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; VI-NEXT:    v_add_f32_e32 v8, v5, v3
 ; VI-NEXT:    v_cndmask_b32_e32 v5, v5, v8, vcc
 ; VI-NEXT:    v_add_u32_e32 v4, vcc, -12, v4
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v4
 ; VI-NEXT:    v_ldexp_f32 v5, v5, 12
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v4
 ; VI-NEXT:    s_cbranch_vccnz .LBB3_6
 ; VI-NEXT:  ; %bb.7: ; %Flow
 ; VI-NEXT:    v_mov_b32_e32 v5, v7
@@ -2480,8 +2477,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; GFX9-NEXT:    global_load_dword v0, v2, s[2:3]
 ; GFX9-NEXT:    global_load_dword v1, v2, s[6:7] offset:16
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
-; GFX9-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v1|
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v1|
 ; GFX9-NEXT:    s_cbranch_vccz .LBB3_2
 ; GFX9-NEXT:  ; %bb.1: ; %frem.else
 ; GFX9-NEXT:    v_and_b32_e32 v2, 0x80000000, v0
@@ -2533,10 +2529,10 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; GFX9-NEXT:    v_fma_f32 v5, -v5, v3, v7
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v5
 ; GFX9-NEXT:    v_add_f32_e32 v8, v5, v3
-; GFX9-NEXT:    v_add_u32_e32 v4, -12, v4
 ; GFX9-NEXT:    v_cndmask_b32_e32 v5, v5, v8, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v4
+; GFX9-NEXT:    v_add_u32_e32 v4, -12, v4
 ; GFX9-NEXT:    v_ldexp_f32 v5, v5, 12
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v4
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB3_6
 ; GFX9-NEXT:  ; %bb.7: ; %Flow
 ; GFX9-NEXT:    v_mov_b32_e32 v5, v7
@@ -2574,8 +2570,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; GFX10-NEXT:    global_load_dword v0, v2, s[2:3]
 ; GFX10-NEXT:    global_load_dword v1, v2, s[6:7] offset:16
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
-; GFX10-NEXT:    v_cmp_ngt_f32_e64 s2, |v0|, |v1|
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v0|, |v1|
 ; GFX10-NEXT:    s_cbranch_vccz .LBB3_2
 ; GFX10-NEXT:  ; %bb.1: ; %frem.else
 ; GFX10-NEXT:    v_and_b32_e32 v2, 0x80000000, v0
@@ -2668,8 +2663,7 @@ define amdgpu_kernel void @frem_f32(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; GFX11-NEXT:    global_load_b32 v0, v1, s[2:3]
 ; GFX11-NEXT:    global_load_b32 v1, v1, s[4:5] offset:16
 ; GFX11-NEXT:    s_waitcnt vmcnt(0)
-; GFX11-NEXT:    v_cmp_ngt_f32_e64 s2, |v0|, |v1|
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v0|, |v1|
 ; GFX11-NEXT:    s_cbranch_vccz .LBB3_2
 ; GFX11-NEXT:  ; %bb.1: ; %frem.else
 ; GFX11-NEXT:    v_and_b32_e32 v2, 0x80000000, v0
@@ -3595,11 +3589,10 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; SI-NEXT:    s_waitcnt vmcnt(1)
 ; SI-NEXT:    v_readfirstlane_b32 s2, v0
 ; SI-NEXT:    s_waitcnt vmcnt(0)
-; SI-NEXT:    v_cmp_ngt_f64_e64 s[0:1], |v[0:1]|, |v[2:3]|
+; SI-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[0:1]|, |v[2:3]|
 ; SI-NEXT:    v_readfirstlane_b32 s3, v1
 ; SI-NEXT:    v_readfirstlane_b32 s4, v2
 ; SI-NEXT:    v_readfirstlane_b32 s5, v3
-; SI-NEXT:    s_and_b64 vcc, exec, s[0:1]
 ; SI-NEXT:    s_cbranch_vccz .LBB6_2
 ; SI-NEXT:  ; %bb.1: ; %frem.else
 ; SI-NEXT:    v_mov_b32_e32 v0, s4
@@ -3750,8 +3743,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; CI-NEXT:    buffer_load_dwordx2 v[0:1], off, s[8:11], 0
 ; CI-NEXT:    buffer_load_dwordx2 v[2:3], off, s[4:7], 0
 ; CI-NEXT:    s_waitcnt vmcnt(0)
-; CI-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[0:1]|, |v[2:3]|
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[0:1]|, |v[2:3]|
 ; CI-NEXT:    s_cbranch_vccz .LBB6_2
 ; CI-NEXT:  ; %bb.1: ; %frem.else
 ; CI-NEXT:    v_cmp_eq_f64_e64 vcc, |v[0:1]|, |v[2:3]|
@@ -3852,8 +3844,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; VI-NEXT:    flat_load_dwordx2 v[0:1], v[0:1]
 ; VI-NEXT:    flat_load_dwordx2 v[2:3], v[2:3]
 ; VI-NEXT:    s_waitcnt vmcnt(0)
-; VI-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[0:1]|, |v[2:3]|
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[0:1]|, |v[2:3]|
 ; VI-NEXT:    s_cbranch_vccz .LBB6_2
 ; VI-NEXT:  ; %bb.1: ; %frem.else
 ; VI-NEXT:    v_cmp_eq_f64_e64 vcc, |v[0:1]|, |v[2:3]|
@@ -3951,8 +3942,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; GFX9-NEXT:    global_load_dwordx2 v[0:1], v4, s[2:3]
 ; GFX9-NEXT:    global_load_dwordx2 v[2:3], v4, s[6:7]
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
-; GFX9-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[0:1]|, |v[2:3]|
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[0:1]|, |v[2:3]|
 ; GFX9-NEXT:    s_cbranch_vccz .LBB6_2
 ; GFX9-NEXT:  ; %bb.1: ; %frem.else
 ; GFX9-NEXT:    v_cmp_eq_f64_e64 vcc, |v[0:1]|, |v[2:3]|
@@ -4051,8 +4041,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; GFX10-NEXT:    global_load_dwordx2 v[0:1], v4, s[2:3]
 ; GFX10-NEXT:    global_load_dwordx2 v[2:3], v4, s[6:7]
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
-; GFX10-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[2:3]|
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, |v[2:3]|
 ; GFX10-NEXT:    s_cbranch_vccz .LBB6_2
 ; GFX10-NEXT:  ; %bb.1: ; %frem.else
 ; GFX10-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[2:3]|
@@ -4149,8 +4138,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; GFX11-NEXT:    global_load_b64 v[0:1], v2, s[2:3]
 ; GFX11-NEXT:    global_load_b64 v[2:3], v2, s[4:5]
 ; GFX11-NEXT:    s_waitcnt vmcnt(0)
-; GFX11-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[2:3]|
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, |v[2:3]|
 ; GFX11-NEXT:    s_cbranch_vccz .LBB6_2
 ; GFX11-NEXT:  ; %bb.1: ; %frem.else
 ; GFX11-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[2:3]|
@@ -4263,8 +4251,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; GFX1150-NEXT:    global_load_b64 v[0:1], v2, s[2:3]
 ; GFX1150-NEXT:    global_load_b64 v[2:3], v2, s[4:5]
 ; GFX1150-NEXT:    s_waitcnt vmcnt(0)
-; GFX1150-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[2:3]|
-; GFX1150-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX1150-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, |v[2:3]|
 ; GFX1150-NEXT:    s_cbranch_vccz .LBB6_2
 ; GFX1150-NEXT:  ; %bb.1: ; %frem.else
 ; GFX1150-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[2:3]|
@@ -4376,8 +4363,7 @@ define amdgpu_kernel void @frem_f64(ptr addrspace(1) %out, ptr addrspace(1) %in1
 ; GFX1200-NEXT:    global_load_b64 v[0:1], v2, s[2:3]
 ; GFX1200-NEXT:    global_load_b64 v[2:3], v2, s[4:5]
 ; GFX1200-NEXT:    s_wait_loadcnt 0x0
-; GFX1200-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[2:3]|
-; GFX1200-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX1200-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, |v[2:3]|
 ; GFX1200-NEXT:    s_cbranch_vccz .LBB6_2
 ; GFX1200-NEXT:  ; %bb.1: ; %frem.else
 ; GFX1200-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[2:3]|
@@ -5327,8 +5313,8 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v8, v5, v3
 ; CI-NEXT:    v_cndmask_b32_e32 v5, v5, v8, vcc
 ; CI-NEXT:    v_add_i32_e32 v4, vcc, -11, v4
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; CI-NEXT:    v_ldexp_f32_e64 v5, v5, 11
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; CI-NEXT:    s_cbranch_vccnz .LBB9_6
 ; CI-NEXT:  ; %bb.7: ; %Flow55
 ; CI-NEXT:    v_mov_b32_e32 v5, v7
@@ -5405,8 +5391,8 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v11, v8, v6
 ; CI-NEXT:    v_cndmask_b32_e32 v8, v8, v11, vcc
 ; CI-NEXT:    v_add_i32_e32 v7, vcc, -11, v7
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v7
 ; CI-NEXT:    v_ldexp_f32_e64 v8, v8, 11
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v7
 ; CI-NEXT:    s_cbranch_vccnz .LBB9_15
 ; CI-NEXT:  ; %bb.16: ; %Flow
 ; CI-NEXT:    v_mov_b32_e32 v8, v10
@@ -5518,8 +5504,8 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v8, v5, v3
 ; VI-NEXT:    v_cndmask_b32_e32 v5, v5, v8, vcc
 ; VI-NEXT:    v_add_u32_e32 v4, vcc, -11, v4
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; VI-NEXT:    v_ldexp_f32 v5, v5, 11
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; VI-NEXT:    s_cbranch_vccnz .LBB9_6
 ; VI-NEXT:  ; %bb.7: ; %Flow55
 ; VI-NEXT:    v_mov_b32_e32 v5, v7
@@ -5595,8 +5581,8 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v11, v8, v6
 ; VI-NEXT:    v_cndmask_b32_e32 v8, v8, v11, vcc
 ; VI-NEXT:    v_add_u32_e32 v7, vcc, -11, v7
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v7
 ; VI-NEXT:    v_ldexp_f32 v8, v8, 11
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v7
 ; VI-NEXT:    s_cbranch_vccnz .LBB9_15
 ; VI-NEXT:  ; %bb.16: ; %Flow
 ; VI-NEXT:    v_mov_b32_e32 v8, v10
@@ -5694,10 +5680,10 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v5, -v5, v3, v7
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v5
 ; GFX9-NEXT:    v_add_f32_e32 v8, v5, v3
-; GFX9-NEXT:    v_add_u32_e32 v4, -11, v4
 ; GFX9-NEXT:    v_cndmask_b32_e32 v5, v5, v8, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
+; GFX9-NEXT:    v_add_u32_e32 v4, -11, v4
 ; GFX9-NEXT:    v_ldexp_f32 v5, v5, 11
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB9_6
 ; GFX9-NEXT:  ; %bb.7: ; %Flow55
 ; GFX9-NEXT:    v_mov_b32_e32 v5, v7
@@ -5770,10 +5756,10 @@ define amdgpu_kernel void @frem_v2f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v7, -v7, v5, v9
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v7
 ; GFX9-NEXT:    v_add_f32_e32 v10, v7, v5
-; GFX9-NEXT:    v_add_u32_e32 v6, -11, v6
 ; GFX9-NEXT:    v_cndmask_b32_e32 v7, v7, v10, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v6
+; GFX9-NEXT:    v_add_u32_e32 v6, -11, v6
 ; GFX9-NEXT:    v_ldexp_f32 v7, v7, 11
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v6
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB9_15
 ; GFX9-NEXT:  ; %bb.16: ; %Flow
 ; GFX9-NEXT:    v_mov_b32_e32 v7, v9
@@ -7829,8 +7815,8 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v8, v5, v3
 ; CI-NEXT:    v_cndmask_b32_e32 v5, v5, v8, vcc
 ; CI-NEXT:    v_add_i32_e32 v4, vcc, -11, v4
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; CI-NEXT:    v_ldexp_f32_e64 v5, v5, 11
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v4
 ; CI-NEXT:    s_cbranch_vccnz .LBB10_6
 ; CI-NEXT:  ; %bb.7: ; %Flow133
 ; CI-NEXT:    v_mov_b32_e32 v5, v7
@@ -7914,8 +7900,8 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v11, v8, v6
 ; CI-NEXT:    v_cndmask_b32_e32 v8, v8, v11, vcc
 ; CI-NEXT:    v_add_i32_e32 v7, vcc, -11, v7
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v7
 ; CI-NEXT:    v_ldexp_f32_e64 v8, v8, 11
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v7
 ; CI-NEXT:    s_cbranch_vccnz .LBB10_17
 ; CI-NEXT:  ; %bb.18: ; %Flow129
 ; CI-NEXT:    v_mov_b32_e32 v8, v10
@@ -7994,8 +7980,8 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v12, v9, v7
 ; CI-NEXT:    v_cndmask_b32_e32 v9, v9, v12, vcc
 ; CI-NEXT:    v_add_i32_e32 v8, vcc, -11, v8
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v8
 ; CI-NEXT:    v_ldexp_f32_e64 v9, v9, 11
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v8
 ; CI-NEXT:    s_cbranch_vccnz .LBB10_27
 ; CI-NEXT:  ; %bb.28: ; %Flow125
 ; CI-NEXT:    v_mov_b32_e32 v9, v11
@@ -8074,8 +8060,8 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v13, v10, v8
 ; CI-NEXT:    v_cndmask_b32_e32 v10, v10, v13, vcc
 ; CI-NEXT:    v_add_i32_e32 v9, vcc, -11, v9
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v9
 ; CI-NEXT:    v_ldexp_f32_e64 v10, v10, 11
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v9
 ; CI-NEXT:    s_cbranch_vccnz .LBB10_37
 ; CI-NEXT:  ; %bb.38: ; %Flow
 ; CI-NEXT:    v_mov_b32_e32 v10, v12
@@ -8205,8 +8191,8 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v10, v7, v5
 ; VI-NEXT:    v_cndmask_b32_e32 v7, v7, v10, vcc
 ; VI-NEXT:    v_add_u32_e32 v6, vcc, -11, v6
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v6
 ; VI-NEXT:    v_ldexp_f32 v7, v7, 11
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v6
 ; VI-NEXT:    s_cbranch_vccnz .LBB10_6
 ; VI-NEXT:  ; %bb.7: ; %Flow133
 ; VI-NEXT:    v_mov_b32_e32 v7, v9
@@ -8282,8 +8268,8 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v13, v10, v8
 ; VI-NEXT:    v_cndmask_b32_e32 v10, v10, v13, vcc
 ; VI-NEXT:    v_add_u32_e32 v9, vcc, -11, v9
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v9
 ; VI-NEXT:    v_ldexp_f32 v10, v10, 11
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v9
 ; VI-NEXT:    s_cbranch_vccnz .LBB10_15
 ; VI-NEXT:  ; %bb.16: ; %Flow129
 ; VI-NEXT:    v_mov_b32_e32 v10, v12
@@ -8357,8 +8343,8 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v14, v11, v9
 ; VI-NEXT:    v_cndmask_b32_e32 v11, v11, v14, vcc
 ; VI-NEXT:    v_add_u32_e32 v10, vcc, -11, v10
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v10
 ; VI-NEXT:    v_ldexp_f32 v11, v11, 11
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v10
 ; VI-NEXT:    s_cbranch_vccnz .LBB10_24
 ; VI-NEXT:  ; %bb.25: ; %Flow125
 ; VI-NEXT:    v_mov_b32_e32 v11, v13
@@ -8434,8 +8420,8 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v17, v14, v12
 ; VI-NEXT:    v_cndmask_b32_e32 v14, v14, v17, vcc
 ; VI-NEXT:    v_add_u32_e32 v13, vcc, -11, v13
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v13
 ; VI-NEXT:    v_ldexp_f32 v14, v14, 11
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v13
 ; VI-NEXT:    s_cbranch_vccnz .LBB10_33
 ; VI-NEXT:  ; %bb.34: ; %Flow
 ; VI-NEXT:    v_mov_b32_e32 v14, v16
@@ -8542,10 +8528,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v7, -v7, v5, v9
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v7
 ; GFX9-NEXT:    v_add_f32_e32 v10, v7, v5
-; GFX9-NEXT:    v_add_u32_e32 v6, -11, v6
 ; GFX9-NEXT:    v_cndmask_b32_e32 v7, v7, v10, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v6
+; GFX9-NEXT:    v_add_u32_e32 v6, -11, v6
 ; GFX9-NEXT:    v_ldexp_f32 v7, v7, 11
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v6
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB10_6
 ; GFX9-NEXT:  ; %bb.7: ; %Flow133
 ; GFX9-NEXT:    v_mov_b32_e32 v7, v9
@@ -8618,10 +8604,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v9, -v9, v7, v11
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v9
 ; GFX9-NEXT:    v_add_f32_e32 v12, v9, v7
-; GFX9-NEXT:    v_add_u32_e32 v8, -11, v8
 ; GFX9-NEXT:    v_cndmask_b32_e32 v9, v9, v12, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v8
+; GFX9-NEXT:    v_add_u32_e32 v8, -11, v8
 ; GFX9-NEXT:    v_ldexp_f32 v9, v9, 11
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v8
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB10_15
 ; GFX9-NEXT:  ; %bb.16: ; %Flow129
 ; GFX9-NEXT:    v_mov_b32_e32 v9, v11
@@ -8693,10 +8679,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v10, -v10, v8, v12
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v10
 ; GFX9-NEXT:    v_add_f32_e32 v13, v10, v8
-; GFX9-NEXT:    v_add_u32_e32 v9, -11, v9
 ; GFX9-NEXT:    v_cndmask_b32_e32 v10, v10, v13, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v9
+; GFX9-NEXT:    v_add_u32_e32 v9, -11, v9
 ; GFX9-NEXT:    v_ldexp_f32 v10, v10, 11
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v9
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB10_24
 ; GFX9-NEXT:  ; %bb.25: ; %Flow125
 ; GFX9-NEXT:    v_mov_b32_e32 v10, v12
@@ -8769,10 +8755,10 @@ define amdgpu_kernel void @frem_v4f16(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v12, -v12, v10, v14
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v12
 ; GFX9-NEXT:    v_add_f32_e32 v15, v12, v10
-; GFX9-NEXT:    v_add_u32_e32 v11, -11, v11
 ; GFX9-NEXT:    v_cndmask_b32_e32 v12, v12, v15, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v11
+; GFX9-NEXT:    v_add_u32_e32 v11, -11, v11
 ; GFX9-NEXT:    v_ldexp_f32 v12, v12, 11
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 11, v11
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB10_33
 ; GFX9-NEXT:  ; %bb.34: ; %Flow
 ; GFX9-NEXT:    v_mov_b32_e32 v12, v14
@@ -11764,8 +11750,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; SI-NEXT:    buffer_load_dwordx2 v[0:1], off, s[8:11], 0
 ; SI-NEXT:    buffer_load_dwordx2 v[2:3], off, s[4:7], 0 offset:32
 ; SI-NEXT:    s_waitcnt vmcnt(0)
-; SI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v2|
-; SI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; SI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v2|
 ; SI-NEXT:    s_cbranch_vccz .LBB11_2
 ; SI-NEXT:  ; %bb.1: ; %frem.else16
 ; SI-NEXT:    v_and_b32_e32 v4, 0x80000000, v0
@@ -11848,8 +11833,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; SI-NEXT:    s_brev_b32 s2, -2
 ; SI-NEXT:    v_bfi_b32 v4, s2, v4, v0
 ; SI-NEXT:  .LBB11_9:
-; SI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v1|, |v3|
-; SI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; SI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v1|, |v3|
 ; SI-NEXT:    s_cbranch_vccz .LBB11_11
 ; SI-NEXT:  ; %bb.10: ; %frem.else
 ; SI-NEXT:    v_and_b32_e32 v5, 0x80000000, v1
@@ -11961,8 +11945,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    buffer_load_dwordx2 v[0:1], off, s[8:11], 0
 ; CI-NEXT:    buffer_load_dwordx2 v[2:3], off, s[4:7], 0 offset:32
 ; CI-NEXT:    s_waitcnt vmcnt(0)
-; CI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v2|
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v2|
 ; CI-NEXT:    s_cbranch_vccz .LBB11_2
 ; CI-NEXT:  ; %bb.1: ; %frem.else16
 ; CI-NEXT:    v_and_b32_e32 v4, 0x80000000, v0
@@ -12016,8 +11999,8 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v10, v7, v5
 ; CI-NEXT:    v_cndmask_b32_e32 v7, v7, v10, vcc
 ; CI-NEXT:    v_add_i32_e32 v6, vcc, -12, v6
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v6
 ; CI-NEXT:    v_ldexp_f32_e64 v7, v7, 12
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v6
 ; CI-NEXT:    s_cbranch_vccnz .LBB11_6
 ; CI-NEXT:  ; %bb.7: ; %Flow51
 ; CI-NEXT:    v_mov_b32_e32 v7, v9
@@ -12034,8 +12017,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    s_brev_b32 s2, -2
 ; CI-NEXT:    v_bfi_b32 v4, s2, v4, v0
 ; CI-NEXT:  .LBB11_9:
-; CI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v1|, |v3|
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v1|, |v3|
 ; CI-NEXT:    s_cbranch_vccz .LBB11_11
 ; CI-NEXT:  ; %bb.10: ; %frem.else
 ; CI-NEXT:    v_and_b32_e32 v5, 0x80000000, v1
@@ -12089,8 +12071,8 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v11, v8, v6
 ; CI-NEXT:    v_cndmask_b32_e32 v8, v8, v11, vcc
 ; CI-NEXT:    v_add_i32_e32 v7, vcc, -12, v7
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v7
 ; CI-NEXT:    v_ldexp_f32_e64 v8, v8, 12
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v7
 ; CI-NEXT:    s_cbranch_vccnz .LBB11_15
 ; CI-NEXT:  ; %bb.16: ; %Flow
 ; CI-NEXT:    v_mov_b32_e32 v8, v10
@@ -12136,8 +12118,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    flat_load_dwordx2 v[0:1], v[0:1]
 ; VI-NEXT:    flat_load_dwordx2 v[2:3], v[2:3]
 ; VI-NEXT:    s_waitcnt vmcnt(0)
-; VI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v2|
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v2|
 ; VI-NEXT:    s_cbranch_vccz .LBB11_2
 ; VI-NEXT:  ; %bb.1: ; %frem.else16
 ; VI-NEXT:    v_and_b32_e32 v4, 0x80000000, v0
@@ -12191,8 +12172,8 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v10, v7, v5
 ; VI-NEXT:    v_cndmask_b32_e32 v7, v7, v10, vcc
 ; VI-NEXT:    v_add_u32_e32 v6, vcc, -12, v6
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v6
 ; VI-NEXT:    v_ldexp_f32 v7, v7, 12
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v6
 ; VI-NEXT:    s_cbranch_vccnz .LBB11_6
 ; VI-NEXT:  ; %bb.7: ; %Flow51
 ; VI-NEXT:    v_mov_b32_e32 v7, v9
@@ -12209,8 +12190,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    s_brev_b32 s2, -2
 ; VI-NEXT:    v_bfi_b32 v4, s2, v4, v0
 ; VI-NEXT:  .LBB11_9:
-; VI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v1|, |v3|
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v1|, |v3|
 ; VI-NEXT:    s_cbranch_vccz .LBB11_11
 ; VI-NEXT:  ; %bb.10: ; %frem.else
 ; VI-NEXT:    v_and_b32_e32 v5, 0x80000000, v1
@@ -12264,8 +12244,8 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v11, v8, v6
 ; VI-NEXT:    v_cndmask_b32_e32 v8, v8, v11, vcc
 ; VI-NEXT:    v_add_u32_e32 v7, vcc, -12, v7
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v7
 ; VI-NEXT:    v_ldexp_f32 v8, v8, 12
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v7
 ; VI-NEXT:    s_cbranch_vccnz .LBB11_15
 ; VI-NEXT:  ; %bb.16: ; %Flow
 ; VI-NEXT:    v_mov_b32_e32 v8, v10
@@ -12306,8 +12286,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    global_load_dwordx2 v[0:1], v4, s[2:3]
 ; GFX9-NEXT:    global_load_dwordx2 v[2:3], v4, s[6:7] offset:32
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
-; GFX9-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v2|
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v2|
 ; GFX9-NEXT:    s_cbranch_vccz .LBB11_2
 ; GFX9-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX9-NEXT:    v_and_b32_e32 v4, 0x80000000, v0
@@ -12359,10 +12338,10 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v7, -v7, v5, v9
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v7
 ; GFX9-NEXT:    v_add_f32_e32 v10, v7, v5
-; GFX9-NEXT:    v_add_u32_e32 v6, -12, v6
 ; GFX9-NEXT:    v_cndmask_b32_e32 v7, v7, v10, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v6
+; GFX9-NEXT:    v_add_u32_e32 v6, -12, v6
 ; GFX9-NEXT:    v_ldexp_f32 v7, v7, 12
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v6
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB11_6
 ; GFX9-NEXT:  ; %bb.7: ; %Flow51
 ; GFX9-NEXT:    v_mov_b32_e32 v7, v9
@@ -12379,8 +12358,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    s_brev_b32 s2, -2
 ; GFX9-NEXT:    v_bfi_b32 v4, s2, v4, v0
 ; GFX9-NEXT:  .LBB11_9:
-; GFX9-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v1|, |v3|
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f32_e64 vcc, |v1|, |v3|
 ; GFX9-NEXT:    s_cbranch_vccz .LBB11_11
 ; GFX9-NEXT:  ; %bb.10: ; %frem.else
 ; GFX9-NEXT:    v_and_b32_e32 v5, 0x80000000, v1
@@ -12432,10 +12410,10 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v8, -v8, v6, v10
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v8
 ; GFX9-NEXT:    v_add_f32_e32 v11, v8, v6
-; GFX9-NEXT:    v_add_u32_e32 v7, -12, v7
 ; GFX9-NEXT:    v_cndmask_b32_e32 v8, v8, v11, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v7
+; GFX9-NEXT:    v_add_u32_e32 v7, -12, v7
 ; GFX9-NEXT:    v_ldexp_f32 v8, v8, 12
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v7
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB11_15
 ; GFX9-NEXT:  ; %bb.16: ; %Flow
 ; GFX9-NEXT:    v_mov_b32_e32 v8, v10
@@ -12477,8 +12455,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX10-NEXT:    global_load_dwordx2 v[0:1], v4, s[2:3]
 ; GFX10-NEXT:    global_load_dwordx2 v[2:3], v4, s[6:7] offset:32
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
-; GFX10-NEXT:    v_cmp_ngt_f32_e64 s2, |v0|, |v2|
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v0|, |v2|
 ; GFX10-NEXT:    s_cbranch_vccz .LBB11_2
 ; GFX10-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX10-NEXT:    v_and_b32_e32 v4, 0x80000000, v0
@@ -12552,8 +12529,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX10-NEXT:    v_ldexp_f32 v4, v5, v4
 ; GFX10-NEXT:    v_bfi_b32 v4, 0x7fffffff, v4, v0
 ; GFX10-NEXT:  .LBB11_9:
-; GFX10-NEXT:    v_cmp_ngt_f32_e64 s2, |v1|, |v3|
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v1|, |v3|
 ; GFX10-NEXT:    s_cbranch_vccz .LBB11_11
 ; GFX10-NEXT:  ; %bb.10: ; %frem.else
 ; GFX10-NEXT:    v_and_b32_e32 v5, 0x80000000, v1
@@ -12650,8 +12626,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX11-NEXT:    global_load_b64 v[0:1], v2, s[2:3]
 ; GFX11-NEXT:    global_load_b64 v[2:3], v2, s[4:5] offset:32
 ; GFX11-NEXT:    s_waitcnt vmcnt(0)
-; GFX11-NEXT:    v_cmp_ngt_f32_e64 s2, |v0|, |v2|
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v0|, |v2|
 ; GFX11-NEXT:    s_cbranch_vccz .LBB11_2
 ; GFX11-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX11-NEXT:    v_and_b32_e32 v4, 0x80000000, v0
@@ -12745,8 +12720,7 @@ define amdgpu_kernel void @frem_v2f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX11-NEXT:    v_bfi_b32 v4, 0x7fffffff, v4, v0
 ; GFX11-NEXT:  .LBB11_9:
-; GFX11-NEXT:    v_cmp_ngt_f32_e64 s2, |v1|, |v3|
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v1|, |v3|
 ; GFX11-NEXT:    s_cbranch_vccz .LBB11_11
 ; GFX11-NEXT:  ; %bb.10: ; %frem.else
 ; GFX11-NEXT:    v_and_b32_e32 v5, 0x80000000, v1
@@ -13360,8 +13334,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; SI-NEXT:    buffer_load_dwordx4 v[0:3], off, s[8:11], 0
 ; SI-NEXT:    buffer_load_dwordx4 v[4:7], off, s[4:7], 0 offset:64
 ; SI-NEXT:    s_waitcnt vmcnt(0)
-; SI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v4|
-; SI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; SI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v4|
 ; SI-NEXT:    s_cbranch_vccz .LBB12_2
 ; SI-NEXT:  ; %bb.1: ; %frem.else78
 ; SI-NEXT:    v_and_b32_e32 v8, 0x80000000, v0
@@ -13444,8 +13417,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; SI-NEXT:    s_brev_b32 s2, -2
 ; SI-NEXT:    v_bfi_b32 v8, s2, v8, v0
 ; SI-NEXT:  .LBB12_9:
-; SI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v1|, |v5|
-; SI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; SI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v1|, |v5|
 ; SI-NEXT:    s_cbranch_vccz .LBB12_11
 ; SI-NEXT:  ; %bb.10: ; %frem.else47
 ; SI-NEXT:    v_and_b32_e32 v9, 0x80000000, v1
@@ -13528,8 +13500,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; SI-NEXT:    s_brev_b32 s2, -2
 ; SI-NEXT:    v_bfi_b32 v9, s2, v9, v1
 ; SI-NEXT:  .LBB12_18:
-; SI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v2|, |v6|
-; SI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; SI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v2|, |v6|
 ; SI-NEXT:    s_cbranch_vccz .LBB12_20
 ; SI-NEXT:  ; %bb.19: ; %frem.else16
 ; SI-NEXT:    v_and_b32_e32 v10, 0x80000000, v2
@@ -13612,8 +13583,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; SI-NEXT:    s_brev_b32 s2, -2
 ; SI-NEXT:    v_bfi_b32 v10, s2, v10, v2
 ; SI-NEXT:  .LBB12_27:
-; SI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v3|, |v7|
-; SI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; SI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v3|, |v7|
 ; SI-NEXT:    s_cbranch_vccz .LBB12_29
 ; SI-NEXT:  ; %bb.28: ; %frem.else
 ; SI-NEXT:    v_and_b32_e32 v11, 0x80000000, v3
@@ -13733,8 +13703,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    buffer_load_dwordx4 v[0:3], off, s[8:11], 0
 ; CI-NEXT:    buffer_load_dwordx4 v[4:7], off, s[4:7], 0 offset:64
 ; CI-NEXT:    s_waitcnt vmcnt(0)
-; CI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v4|
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v4|
 ; CI-NEXT:    s_cbranch_vccz .LBB12_2
 ; CI-NEXT:  ; %bb.1: ; %frem.else78
 ; CI-NEXT:    v_and_b32_e32 v8, 0x80000000, v0
@@ -13788,8 +13757,8 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v14, v11, v9
 ; CI-NEXT:    v_cndmask_b32_e32 v11, v11, v14, vcc
 ; CI-NEXT:    v_add_i32_e32 v10, vcc, -12, v10
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v10
 ; CI-NEXT:    v_ldexp_f32_e64 v11, v11, 12
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v10
 ; CI-NEXT:    s_cbranch_vccnz .LBB12_6
 ; CI-NEXT:  ; %bb.7: ; %Flow125
 ; CI-NEXT:    v_mov_b32_e32 v11, v13
@@ -13806,8 +13775,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    s_brev_b32 s2, -2
 ; CI-NEXT:    v_bfi_b32 v8, s2, v8, v0
 ; CI-NEXT:  .LBB12_9:
-; CI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v1|, |v5|
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v1|, |v5|
 ; CI-NEXT:    s_cbranch_vccz .LBB12_11
 ; CI-NEXT:  ; %bb.10: ; %frem.else47
 ; CI-NEXT:    v_and_b32_e32 v9, 0x80000000, v1
@@ -13861,8 +13829,8 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v15, v12, v10
 ; CI-NEXT:    v_cndmask_b32_e32 v12, v12, v15, vcc
 ; CI-NEXT:    v_add_i32_e32 v11, vcc, -12, v11
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v11
 ; CI-NEXT:    v_ldexp_f32_e64 v12, v12, 12
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v11
 ; CI-NEXT:    s_cbranch_vccnz .LBB12_15
 ; CI-NEXT:  ; %bb.16: ; %Flow121
 ; CI-NEXT:    v_mov_b32_e32 v12, v14
@@ -13879,8 +13847,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    s_brev_b32 s2, -2
 ; CI-NEXT:    v_bfi_b32 v9, s2, v9, v1
 ; CI-NEXT:  .LBB12_18:
-; CI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v2|, |v6|
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v2|, |v6|
 ; CI-NEXT:    s_cbranch_vccz .LBB12_20
 ; CI-NEXT:  ; %bb.19: ; %frem.else16
 ; CI-NEXT:    v_and_b32_e32 v10, 0x80000000, v2
@@ -13934,8 +13901,8 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v16, v13, v11
 ; CI-NEXT:    v_cndmask_b32_e32 v13, v13, v16, vcc
 ; CI-NEXT:    v_add_i32_e32 v12, vcc, -12, v12
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v12
 ; CI-NEXT:    v_ldexp_f32_e64 v13, v13, 12
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v12
 ; CI-NEXT:    s_cbranch_vccnz .LBB12_24
 ; CI-NEXT:  ; %bb.25: ; %Flow117
 ; CI-NEXT:    v_mov_b32_e32 v13, v15
@@ -13952,8 +13919,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    s_brev_b32 s2, -2
 ; CI-NEXT:    v_bfi_b32 v10, s2, v10, v2
 ; CI-NEXT:  .LBB12_27:
-; CI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v3|, |v7|
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v3|, |v7|
 ; CI-NEXT:    s_cbranch_vccz .LBB12_29
 ; CI-NEXT:  ; %bb.28: ; %frem.else
 ; CI-NEXT:    v_and_b32_e32 v11, 0x80000000, v3
@@ -14007,8 +13973,8 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_add_f32_e32 v17, v14, v12
 ; CI-NEXT:    v_cndmask_b32_e32 v14, v14, v17, vcc
 ; CI-NEXT:    v_add_i32_e32 v13, vcc, -12, v13
-; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v13
 ; CI-NEXT:    v_ldexp_f32_e64 v14, v14, 12
+; CI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v13
 ; CI-NEXT:    s_cbranch_vccnz .LBB12_33
 ; CI-NEXT:  ; %bb.34: ; %Flow
 ; CI-NEXT:    v_mov_b32_e32 v14, v16
@@ -14062,8 +14028,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    flat_load_dwordx4 v[0:3], v[0:1]
 ; VI-NEXT:    flat_load_dwordx4 v[4:7], v[4:5]
 ; VI-NEXT:    s_waitcnt vmcnt(0)
-; VI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v4|
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v4|
 ; VI-NEXT:    s_cbranch_vccz .LBB12_2
 ; VI-NEXT:  ; %bb.1: ; %frem.else78
 ; VI-NEXT:    v_and_b32_e32 v8, 0x80000000, v0
@@ -14117,8 +14082,8 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v14, v11, v9
 ; VI-NEXT:    v_cndmask_b32_e32 v11, v11, v14, vcc
 ; VI-NEXT:    v_add_u32_e32 v10, vcc, -12, v10
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v10
 ; VI-NEXT:    v_ldexp_f32 v11, v11, 12
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v10
 ; VI-NEXT:    s_cbranch_vccnz .LBB12_6
 ; VI-NEXT:  ; %bb.7: ; %Flow125
 ; VI-NEXT:    v_mov_b32_e32 v11, v13
@@ -14135,8 +14100,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    s_brev_b32 s2, -2
 ; VI-NEXT:    v_bfi_b32 v8, s2, v8, v0
 ; VI-NEXT:  .LBB12_9:
-; VI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v1|, |v5|
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v1|, |v5|
 ; VI-NEXT:    s_cbranch_vccz .LBB12_11
 ; VI-NEXT:  ; %bb.10: ; %frem.else47
 ; VI-NEXT:    v_and_b32_e32 v9, 0x80000000, v1
@@ -14190,8 +14154,8 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v15, v12, v10
 ; VI-NEXT:    v_cndmask_b32_e32 v12, v12, v15, vcc
 ; VI-NEXT:    v_add_u32_e32 v11, vcc, -12, v11
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v11
 ; VI-NEXT:    v_ldexp_f32 v12, v12, 12
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v11
 ; VI-NEXT:    s_cbranch_vccnz .LBB12_15
 ; VI-NEXT:  ; %bb.16: ; %Flow121
 ; VI-NEXT:    v_mov_b32_e32 v12, v14
@@ -14208,8 +14172,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    s_brev_b32 s2, -2
 ; VI-NEXT:    v_bfi_b32 v9, s2, v9, v1
 ; VI-NEXT:  .LBB12_18:
-; VI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v2|, |v6|
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v2|, |v6|
 ; VI-NEXT:    s_cbranch_vccz .LBB12_20
 ; VI-NEXT:  ; %bb.19: ; %frem.else16
 ; VI-NEXT:    v_and_b32_e32 v10, 0x80000000, v2
@@ -14263,8 +14226,8 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v16, v13, v11
 ; VI-NEXT:    v_cndmask_b32_e32 v13, v13, v16, vcc
 ; VI-NEXT:    v_add_u32_e32 v12, vcc, -12, v12
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v12
 ; VI-NEXT:    v_ldexp_f32 v13, v13, 12
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v12
 ; VI-NEXT:    s_cbranch_vccnz .LBB12_24
 ; VI-NEXT:  ; %bb.25: ; %Flow117
 ; VI-NEXT:    v_mov_b32_e32 v13, v15
@@ -14281,8 +14244,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    s_brev_b32 s2, -2
 ; VI-NEXT:    v_bfi_b32 v10, s2, v10, v2
 ; VI-NEXT:  .LBB12_27:
-; VI-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v3|, |v7|
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f32_e64 vcc, |v3|, |v7|
 ; VI-NEXT:    s_cbranch_vccz .LBB12_29
 ; VI-NEXT:  ; %bb.28: ; %frem.else
 ; VI-NEXT:    v_and_b32_e32 v11, 0x80000000, v3
@@ -14336,8 +14298,8 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_add_f32_e32 v17, v14, v12
 ; VI-NEXT:    v_cndmask_b32_e32 v14, v14, v17, vcc
 ; VI-NEXT:    v_add_u32_e32 v13, vcc, -12, v13
-; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v13
 ; VI-NEXT:    v_ldexp_f32 v14, v14, 12
+; VI-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v13
 ; VI-NEXT:    s_cbranch_vccnz .LBB12_33
 ; VI-NEXT:  ; %bb.34: ; %Flow
 ; VI-NEXT:    v_mov_b32_e32 v14, v16
@@ -14386,8 +14348,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    global_load_dwordx4 v[0:3], v8, s[2:3]
 ; GFX9-NEXT:    global_load_dwordx4 v[4:7], v8, s[6:7] offset:64
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
-; GFX9-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v0|, |v4|
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f32_e64 vcc, |v0|, |v4|
 ; GFX9-NEXT:    s_cbranch_vccz .LBB12_2
 ; GFX9-NEXT:  ; %bb.1: ; %frem.else78
 ; GFX9-NEXT:    v_and_b32_e32 v8, 0x80000000, v0
@@ -14439,10 +14400,10 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v11, -v11, v9, v13
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v11
 ; GFX9-NEXT:    v_add_f32_e32 v14, v11, v9
-; GFX9-NEXT:    v_add_u32_e32 v10, -12, v10
 ; GFX9-NEXT:    v_cndmask_b32_e32 v11, v11, v14, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v10
+; GFX9-NEXT:    v_add_u32_e32 v10, -12, v10
 ; GFX9-NEXT:    v_ldexp_f32 v11, v11, 12
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v10
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB12_6
 ; GFX9-NEXT:  ; %bb.7: ; %Flow125
 ; GFX9-NEXT:    v_mov_b32_e32 v11, v13
@@ -14459,8 +14420,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    s_brev_b32 s2, -2
 ; GFX9-NEXT:    v_bfi_b32 v8, s2, v8, v0
 ; GFX9-NEXT:  .LBB12_9:
-; GFX9-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v1|, |v5|
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f32_e64 vcc, |v1|, |v5|
 ; GFX9-NEXT:    s_cbranch_vccz .LBB12_11
 ; GFX9-NEXT:  ; %bb.10: ; %frem.else47
 ; GFX9-NEXT:    v_and_b32_e32 v9, 0x80000000, v1
@@ -14512,10 +14472,10 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v12, -v12, v10, v14
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v12
 ; GFX9-NEXT:    v_add_f32_e32 v15, v12, v10
-; GFX9-NEXT:    v_add_u32_e32 v11, -12, v11
 ; GFX9-NEXT:    v_cndmask_b32_e32 v12, v12, v15, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v11
+; GFX9-NEXT:    v_add_u32_e32 v11, -12, v11
 ; GFX9-NEXT:    v_ldexp_f32 v12, v12, 12
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v11
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB12_15
 ; GFX9-NEXT:  ; %bb.16: ; %Flow121
 ; GFX9-NEXT:    v_mov_b32_e32 v12, v14
@@ -14532,8 +14492,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    s_brev_b32 s2, -2
 ; GFX9-NEXT:    v_bfi_b32 v9, s2, v9, v1
 ; GFX9-NEXT:  .LBB12_18:
-; GFX9-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v2|, |v6|
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f32_e64 vcc, |v2|, |v6|
 ; GFX9-NEXT:    s_cbranch_vccz .LBB12_20
 ; GFX9-NEXT:  ; %bb.19: ; %frem.else16
 ; GFX9-NEXT:    v_and_b32_e32 v10, 0x80000000, v2
@@ -14585,10 +14544,10 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v13, -v13, v11, v15
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v13
 ; GFX9-NEXT:    v_add_f32_e32 v16, v13, v11
-; GFX9-NEXT:    v_add_u32_e32 v12, -12, v12
 ; GFX9-NEXT:    v_cndmask_b32_e32 v13, v13, v16, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v12
+; GFX9-NEXT:    v_add_u32_e32 v12, -12, v12
 ; GFX9-NEXT:    v_ldexp_f32 v13, v13, 12
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v12
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB12_24
 ; GFX9-NEXT:  ; %bb.25: ; %Flow117
 ; GFX9-NEXT:    v_mov_b32_e32 v13, v15
@@ -14605,8 +14564,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    s_brev_b32 s2, -2
 ; GFX9-NEXT:    v_bfi_b32 v10, s2, v10, v2
 ; GFX9-NEXT:  .LBB12_27:
-; GFX9-NEXT:    v_cmp_ngt_f32_e64 s[2:3], |v3|, |v7|
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f32_e64 vcc, |v3|, |v7|
 ; GFX9-NEXT:    s_cbranch_vccz .LBB12_29
 ; GFX9-NEXT:  ; %bb.28: ; %frem.else
 ; GFX9-NEXT:    v_and_b32_e32 v11, 0x80000000, v3
@@ -14658,10 +14616,10 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_fma_f32 v14, -v14, v12, v16
 ; GFX9-NEXT:    v_cmp_gt_f32_e32 vcc, 0, v14
 ; GFX9-NEXT:    v_add_f32_e32 v17, v14, v12
-; GFX9-NEXT:    v_add_u32_e32 v13, -12, v13
 ; GFX9-NEXT:    v_cndmask_b32_e32 v14, v14, v17, vcc
-; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v13
+; GFX9-NEXT:    v_add_u32_e32 v13, -12, v13
 ; GFX9-NEXT:    v_ldexp_f32 v14, v14, 12
+; GFX9-NEXT:    v_cmp_lt_i32_e32 vcc, 12, v13
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB12_33
 ; GFX9-NEXT:  ; %bb.34: ; %Flow
 ; GFX9-NEXT:    v_mov_b32_e32 v14, v16
@@ -14711,8 +14669,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX10-NEXT:    global_load_dwordx4 v[0:3], v8, s[2:3]
 ; GFX10-NEXT:    global_load_dwordx4 v[4:7], v8, s[6:7] offset:64
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
-; GFX10-NEXT:    v_cmp_ngt_f32_e64 s2, |v0|, |v4|
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v0|, |v4|
 ; GFX10-NEXT:    s_cbranch_vccz .LBB12_2
 ; GFX10-NEXT:  ; %bb.1: ; %frem.else78
 ; GFX10-NEXT:    v_and_b32_e32 v8, 0x80000000, v0
@@ -14786,8 +14743,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX10-NEXT:    v_ldexp_f32 v8, v9, v8
 ; GFX10-NEXT:    v_bfi_b32 v8, 0x7fffffff, v8, v0
 ; GFX10-NEXT:  .LBB12_9:
-; GFX10-NEXT:    v_cmp_ngt_f32_e64 s2, |v1|, |v5|
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v1|, |v5|
 ; GFX10-NEXT:    s_cbranch_vccz .LBB12_11
 ; GFX10-NEXT:  ; %bb.10: ; %frem.else47
 ; GFX10-NEXT:    v_and_b32_e32 v9, 0x80000000, v1
@@ -14861,8 +14817,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX10-NEXT:    v_ldexp_f32 v9, v10, v9
 ; GFX10-NEXT:    v_bfi_b32 v9, 0x7fffffff, v9, v1
 ; GFX10-NEXT:  .LBB12_18:
-; GFX10-NEXT:    v_cmp_ngt_f32_e64 s2, |v2|, |v6|
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v2|, |v6|
 ; GFX10-NEXT:    s_cbranch_vccz .LBB12_20
 ; GFX10-NEXT:  ; %bb.19: ; %frem.else16
 ; GFX10-NEXT:    v_and_b32_e32 v10, 0x80000000, v2
@@ -14936,8 +14891,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX10-NEXT:    v_ldexp_f32 v10, v11, v10
 ; GFX10-NEXT:    v_bfi_b32 v10, 0x7fffffff, v10, v2
 ; GFX10-NEXT:  .LBB12_27:
-; GFX10-NEXT:    v_cmp_ngt_f32_e64 s2, |v3|, |v7|
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v3|, |v7|
 ; GFX10-NEXT:    s_cbranch_vccz .LBB12_29
 ; GFX10-NEXT:  ; %bb.28: ; %frem.else
 ; GFX10-NEXT:    v_and_b32_e32 v11, 0x80000000, v3
@@ -15042,8 +14996,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX11-NEXT:    global_load_b128 v[0:3], v4, s[2:3]
 ; GFX11-NEXT:    global_load_b128 v[4:7], v4, s[4:5] offset:64
 ; GFX11-NEXT:    s_waitcnt vmcnt(0)
-; GFX11-NEXT:    v_cmp_ngt_f32_e64 s2, |v0|, |v4|
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v0|, |v4|
 ; GFX11-NEXT:    s_cbranch_vccz .LBB12_2
 ; GFX11-NEXT:  ; %bb.1: ; %frem.else78
 ; GFX11-NEXT:    v_and_b32_e32 v8, 0x80000000, v0
@@ -15137,8 +15090,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX11-NEXT:    v_bfi_b32 v8, 0x7fffffff, v8, v0
 ; GFX11-NEXT:  .LBB12_9:
-; GFX11-NEXT:    v_cmp_ngt_f32_e64 s2, |v1|, |v5|
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v1|, |v5|
 ; GFX11-NEXT:    s_cbranch_vccz .LBB12_11
 ; GFX11-NEXT:  ; %bb.10: ; %frem.else47
 ; GFX11-NEXT:    v_and_b32_e32 v9, 0x80000000, v1
@@ -15232,8 +15184,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX11-NEXT:    v_bfi_b32 v9, 0x7fffffff, v9, v1
 ; GFX11-NEXT:  .LBB12_18:
-; GFX11-NEXT:    v_cmp_ngt_f32_e64 s2, |v2|, |v6|
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v2|, |v6|
 ; GFX11-NEXT:    s_cbranch_vccz .LBB12_20
 ; GFX11-NEXT:  ; %bb.19: ; %frem.else16
 ; GFX11-NEXT:    v_and_b32_e32 v10, 0x80000000, v2
@@ -15327,8 +15278,7 @@ define amdgpu_kernel void @frem_v4f32(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX11-NEXT:    v_bfi_b32 v10, 0x7fffffff, v10, v2
 ; GFX11-NEXT:  .LBB12_27:
-; GFX11-NEXT:    v_cmp_ngt_f32_e64 s2, |v3|, |v7|
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f32_e64 vcc_lo, |v3|, |v7|
 ; GFX11-NEXT:    s_cbranch_vccz .LBB12_29
 ; GFX11-NEXT:  ; %bb.28: ; %frem.else
 ; GFX11-NEXT:    v_and_b32_e32 v11, 0x80000000, v3
@@ -16416,10 +16366,9 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; SI-NEXT:    v_readfirstlane_b32 s10, v2
 ; SI-NEXT:    v_mov_b32_e32 v0, s10
 ; SI-NEXT:    v_mov_b32_e32 v1, s11
-; SI-NEXT:    v_cmp_ngt_f64_e64 s[0:1], |s[6:7]|, |v[0:1]|
+; SI-NEXT:    v_cmp_ngt_f64_e64 vcc, |s[6:7]|, |v[0:1]|
 ; SI-NEXT:    v_readfirstlane_b32 s4, v4
 ; SI-NEXT:    v_readfirstlane_b32 s5, v5
-; SI-NEXT:    s_and_b64 vcc, exec, s[0:1]
 ; SI-NEXT:    s_cbranch_vccz .LBB13_2
 ; SI-NEXT:  ; %bb.1: ; %frem.else16
 ; SI-NEXT:    v_mov_b32_e32 v0, s10
@@ -16545,8 +16494,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; SI-NEXT:  .LBB13_10:
 ; SI-NEXT:    v_mov_b32_e32 v2, s4
 ; SI-NEXT:    v_mov_b32_e32 v3, s5
-; SI-NEXT:    v_cmp_ngt_f64_e64 s[0:1], |s[2:3]|, |v[2:3]|
-; SI-NEXT:    s_and_b64 vcc, exec, s[0:1]
+; SI-NEXT:    v_cmp_ngt_f64_e64 vcc, |s[2:3]|, |v[2:3]|
 ; SI-NEXT:    s_cbranch_vccz .LBB13_12
 ; SI-NEXT:  ; %bb.11: ; %frem.else
 ; SI-NEXT:    v_mov_b32_e32 v2, s4
@@ -16702,8 +16650,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    buffer_load_dwordx4 v[0:3], off, s[8:11], 0
 ; CI-NEXT:    buffer_load_dwordx4 v[4:7], off, s[4:7], 0 offset:64
 ; CI-NEXT:    s_waitcnt vmcnt(0)
-; CI-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[0:1]|, |v[4:5]|
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[0:1]|, |v[4:5]|
 ; CI-NEXT:    s_cbranch_vccz .LBB13_2
 ; CI-NEXT:  ; %bb.1: ; %frem.else16
 ; CI-NEXT:    v_cmp_eq_f64_e64 vcc, |v[0:1]|, |v[4:5]|
@@ -16779,8 +16726,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; CI-NEXT:    v_ldexp_f64 v[8:9], v[8:9], v16
 ; CI-NEXT:    v_bfi_b32 v9, s2, v9, v1
 ; CI-NEXT:  .LBB13_9:
-; CI-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[2:3]|, |v[6:7]|
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[2:3]|, |v[6:7]|
 ; CI-NEXT:    s_cbranch_vccz .LBB13_11
 ; CI-NEXT:  ; %bb.10: ; %frem.else
 ; CI-NEXT:    v_cmp_eq_f64_e64 vcc, |v[2:3]|, |v[6:7]|
@@ -16888,8 +16834,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    flat_load_dwordx4 v[0:3], v[0:1]
 ; VI-NEXT:    flat_load_dwordx4 v[4:7], v[4:5]
 ; VI-NEXT:    s_waitcnt vmcnt(0)
-; VI-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[0:1]|, |v[4:5]|
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[0:1]|, |v[4:5]|
 ; VI-NEXT:    s_cbranch_vccz .LBB13_2
 ; VI-NEXT:  ; %bb.1: ; %frem.else16
 ; VI-NEXT:    v_cmp_eq_f64_e64 vcc, |v[0:1]|, |v[4:5]|
@@ -16965,8 +16910,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; VI-NEXT:    v_ldexp_f64 v[8:9], v[8:9], v16
 ; VI-NEXT:    v_bfi_b32 v9, s2, v9, v1
 ; VI-NEXT:  .LBB13_9:
-; VI-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[2:3]|, |v[6:7]|
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[2:3]|, |v[6:7]|
 ; VI-NEXT:    s_cbranch_vccz .LBB13_11
 ; VI-NEXT:  ; %bb.10: ; %frem.else
 ; VI-NEXT:    v_cmp_eq_f64_e64 vcc, |v[2:3]|, |v[6:7]|
@@ -17069,8 +17013,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    global_load_dwordx4 v[0:3], v8, s[2:3]
 ; GFX9-NEXT:    global_load_dwordx4 v[4:7], v8, s[6:7] offset:64
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
-; GFX9-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[0:1]|, |v[4:5]|
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[0:1]|, |v[4:5]|
 ; GFX9-NEXT:    s_cbranch_vccz .LBB13_2
 ; GFX9-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX9-NEXT:    v_cmp_eq_f64_e64 vcc, |v[0:1]|, |v[4:5]|
@@ -17146,8 +17089,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX9-NEXT:    v_ldexp_f64 v[8:9], v[8:9], v16
 ; GFX9-NEXT:    v_bfi_b32 v9, s2, v9, v1
 ; GFX9-NEXT:  .LBB13_9:
-; GFX9-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[2:3]|, |v[6:7]|
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[2:3]|, |v[6:7]|
 ; GFX9-NEXT:    s_cbranch_vccz .LBB13_11
 ; GFX9-NEXT:  ; %bb.10: ; %frem.else
 ; GFX9-NEXT:    v_cmp_eq_f64_e64 vcc, |v[2:3]|, |v[6:7]|
@@ -17251,8 +17193,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX10-NEXT:    global_load_dwordx4 v[0:3], v8, s[2:3]
 ; GFX10-NEXT:    global_load_dwordx4 v[4:7], v8, s[6:7] offset:64
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
-; GFX10-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[4:5]|
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, |v[4:5]|
 ; GFX10-NEXT:    s_cbranch_vccz .LBB13_2
 ; GFX10-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX10-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[4:5]|
@@ -17329,8 +17270,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX10-NEXT:    v_ldexp_f64 v[8:9], v[8:9], v16
 ; GFX10-NEXT:    v_bfi_b32 v9, 0x7fffffff, v9, v1
 ; GFX10-NEXT:  .LBB13_9:
-; GFX10-NEXT:    v_cmp_ngt_f64_e64 s2, |v[2:3]|, |v[6:7]|
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[2:3]|, |v[6:7]|
 ; GFX10-NEXT:    s_cbranch_vccz .LBB13_11
 ; GFX10-NEXT:  ; %bb.10: ; %frem.else
 ; GFX10-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, |v[6:7]|
@@ -17432,8 +17372,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX11-NEXT:    global_load_b128 v[0:3], v4, s[2:3]
 ; GFX11-NEXT:    global_load_b128 v[4:7], v4, s[4:5] offset:64
 ; GFX11-NEXT:    s_waitcnt vmcnt(0)
-; GFX11-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[4:5]|
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, |v[4:5]|
 ; GFX11-NEXT:    s_cbranch_vccz .LBB13_2
 ; GFX11-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX11-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[4:5]|
@@ -17527,8 +17466,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX11-NEXT:    v_bfi_b32 v9, 0x7fffffff, v9, v1
 ; GFX11-NEXT:  .LBB13_9:
-; GFX11-NEXT:    v_cmp_ngt_f64_e64 s2, |v[2:3]|, |v[6:7]|
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[2:3]|, |v[6:7]|
 ; GFX11-NEXT:    s_cbranch_vccz .LBB13_11
 ; GFX11-NEXT:  ; %bb.10: ; %frem.else
 ; GFX11-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, |v[6:7]|
@@ -17645,8 +17583,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX1150-NEXT:    global_load_b128 v[0:3], v4, s[2:3]
 ; GFX1150-NEXT:    global_load_b128 v[4:7], v4, s[4:5] offset:64
 ; GFX1150-NEXT:    s_waitcnt vmcnt(0)
-; GFX1150-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[4:5]|
-; GFX1150-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX1150-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, |v[4:5]|
 ; GFX1150-NEXT:    s_cbranch_vccz .LBB13_2
 ; GFX1150-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX1150-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[4:5]|
@@ -17739,8 +17676,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1150-NEXT:    v_bfi_b32 v9, 0x7fffffff, v9, v1
 ; GFX1150-NEXT:  .LBB13_9:
-; GFX1150-NEXT:    v_cmp_ngt_f64_e64 s2, |v[2:3]|, |v[6:7]|
-; GFX1150-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX1150-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[2:3]|, |v[6:7]|
 ; GFX1150-NEXT:    s_cbranch_vccz .LBB13_11
 ; GFX1150-NEXT:  ; %bb.10: ; %frem.else
 ; GFX1150-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, |v[6:7]|
@@ -17856,8 +17792,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX1200-NEXT:    global_load_b128 v[0:3], v4, s[2:3]
 ; GFX1200-NEXT:    global_load_b128 v[4:7], v4, s[4:5] offset:64
 ; GFX1200-NEXT:    s_wait_loadcnt 0x0
-; GFX1200-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, |v[4:5]|
-; GFX1200-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX1200-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, |v[4:5]|
 ; GFX1200-NEXT:    s_cbranch_vccz .LBB13_2
 ; GFX1200-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX1200-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, |v[4:5]|
@@ -17952,9 +17887,7 @@ define amdgpu_kernel void @frem_v2f64(ptr addrspace(1) %out, ptr addrspace(1) %i
 ; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1200-NEXT:    v_bfi_b32 v9, 0x7fffffff, v9, v1
 ; GFX1200-NEXT:  .LBB13_9:
-; GFX1200-NEXT:    v_cmp_ngt_f64_e64 s2, |v[2:3]|, |v[6:7]|
-; GFX1200-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
-; GFX1200-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX1200-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[2:3]|, |v[6:7]|
 ; GFX1200-NEXT:    s_cbranch_vccz .LBB13_11
 ; GFX1200-NEXT:  ; %bb.10: ; %frem.else
 ; GFX1200-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, |v[6:7]|
@@ -18267,10 +18200,9 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; SI-NEXT:    s_waitcnt vmcnt(0)
 ; SI-NEXT:    v_readfirstlane_b32 s3, v1
 ; SI-NEXT:    v_readfirstlane_b32 s2, v0
-; SI-NEXT:    v_cmp_ngt_f64_e64 s[6:7], |s[2:3]|, 1.0
+; SI-NEXT:    v_cmp_ngt_f64_e64 vcc, |s[2:3]|, 1.0
 ; SI-NEXT:    v_readfirstlane_b32 s4, v2
 ; SI-NEXT:    v_readfirstlane_b32 s5, v3
-; SI-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; SI-NEXT:    s_cbranch_vccz .LBB15_2
 ; SI-NEXT:  ; %bb.1: ; %frem.else16
 ; SI-NEXT:    v_cmp_eq_f64_e64 s[6:7], |s[2:3]|, 1.0
@@ -18287,7 +18219,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; SI-NEXT:    s_and_b64 s[8:9], s[8:9], exec
 ; SI-NEXT:    s_cselect_b32 s8, 1, 0
 ; SI-NEXT:    s_cmp_lg_u32 s8, 1
-; SI-NEXT:    s_cbranch_scc1 .LBB15_9
+; SI-NEXT:    s_cbranch_scc1 .LBB15_10
 ; SI-NEXT:  ; %bb.4: ; %frem.compute15
 ; SI-NEXT:    s_and_b32 s6, s3, 0x7fffffff
 ; SI-NEXT:    v_frexp_mant_f64_e64 v[0:1], |s[2:3]|
@@ -18354,31 +18286,30 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; SI-NEXT:    v_cndmask_b32_e32 v1, v1, v3, vcc
 ; SI-NEXT:    v_mov_b32_e32 v2, s3
 ; SI-NEXT:    v_bfi_b32 v1, s6, v1, v2
-; SI-NEXT:    s_branch .LBB15_10
+; SI-NEXT:    v_cmp_ngt_f64_e64 vcc, |s[4:5]|, 1.0
+; SI-NEXT:    s_cbranch_vccnz .LBB15_11
 ; SI-NEXT:  .LBB15_9:
+; SI-NEXT:    s_mov_b64 s[8:9], -1
+; SI-NEXT:    ; implicit-def: $sgpr6_sgpr7
+; SI-NEXT:    s_branch .LBB15_12
+; SI-NEXT:  .LBB15_10:
 ; SI-NEXT:    v_mov_b32_e32 v0, s6
 ; SI-NEXT:    v_mov_b32_e32 v1, s7
-; SI-NEXT:  .LBB15_10:
-; SI-NEXT:    v_cmp_ngt_f64_e64 s[6:7], |s[4:5]|, 1.0
-; SI-NEXT:    s_and_b64 vcc, exec, s[6:7]
-; SI-NEXT:    s_cbranch_vccz .LBB15_12
-; SI-NEXT:  ; %bb.11: ; %frem.else
+; SI-NEXT:    v_cmp_ngt_f64_e64 vcc, |s[4:5]|, 1.0
+; SI-NEXT:    s_cbranch_vccz .LBB15_9
+; SI-NEXT:  .LBB15_11: ; %frem.else
 ; SI-NEXT:    v_cmp_eq_f64_e64 s[6:7], |s[4:5]|, 1.0
 ; SI-NEXT:    s_and_b32 s8, s5, 0x80000000
 ; SI-NEXT:    s_and_b64 s[6:7], s[6:7], exec
 ; SI-NEXT:    s_cselect_b32 s7, s8, s5
 ; SI-NEXT:    s_cselect_b32 s6, 0, s4
 ; SI-NEXT:    s_mov_b64 s[8:9], 0
-; SI-NEXT:    s_branch .LBB15_13
-; SI-NEXT:  .LBB15_12:
-; SI-NEXT:    s_mov_b64 s[8:9], -1
-; SI-NEXT:    ; implicit-def: $sgpr6_sgpr7
-; SI-NEXT:  .LBB15_13: ; %Flow48
+; SI-NEXT:  .LBB15_12: ; %Flow48
 ; SI-NEXT:    s_and_b64 s[8:9], s[8:9], exec
 ; SI-NEXT:    s_cselect_b32 s8, 1, 0
 ; SI-NEXT:    s_cmp_lg_u32 s8, 1
-; SI-NEXT:    s_cbranch_scc1 .LBB15_19
-; SI-NEXT:  ; %bb.14: ; %frem.compute
+; SI-NEXT:    s_cbranch_scc1 .LBB15_18
+; SI-NEXT:  ; %bb.13: ; %frem.compute
 ; SI-NEXT:    s_and_b32 s6, s5, 0x7fffffff
 ; SI-NEXT:    v_frexp_mant_f64_e64 v[2:3], |s[4:5]|
 ; SI-NEXT:    s_cmp_lt_i32 s6, 0x7ff00000
@@ -18395,13 +18326,13 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; SI-NEXT:    v_ldexp_f64 v[3:4], v[2:3], 26
 ; SI-NEXT:    s_mov_b32 s6, -1
 ; SI-NEXT:    s_cmp_lt_i32 s9, 27
-; SI-NEXT:    s_cbranch_scc1 .LBB15_18
-; SI-NEXT:  ; %bb.15: ; %frem.loop_body.preheader
+; SI-NEXT:    s_cbranch_scc1 .LBB15_17
+; SI-NEXT:  ; %bb.14: ; %frem.loop_body.preheader
 ; SI-NEXT:    s_add_i32 s9, s7, 25
 ; SI-NEXT:    v_mov_b32_e32 v7, 0x43300000
 ; SI-NEXT:    v_mov_b32_e32 v2, 0
 ; SI-NEXT:    s_mov_b32 s7, 0x432fffff
-; SI-NEXT:  .LBB15_16: ; %frem.loop_body
+; SI-NEXT:  .LBB15_15: ; %frem.loop_body
 ; SI-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; SI-NEXT:    v_mov_b32_e32 v6, v4
 ; SI-NEXT:    v_mov_b32_e32 v5, v3
@@ -18419,11 +18350,11 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; SI-NEXT:    v_cndmask_b32_e32 v4, v4, v9, vcc
 ; SI-NEXT:    v_cndmask_b32_e32 v3, v3, v8, vcc
 ; SI-NEXT:    v_ldexp_f64 v[3:4], v[3:4], 26
-; SI-NEXT:    s_cbranch_scc1 .LBB15_16
-; SI-NEXT:  ; %bb.17: ; %Flow
+; SI-NEXT:    s_cbranch_scc1 .LBB15_15
+; SI-NEXT:  ; %bb.16: ; %Flow
 ; SI-NEXT:    v_mov_b32_e32 v3, v5
 ; SI-NEXT:    v_mov_b32_e32 v4, v6
-; SI-NEXT:  .LBB15_18: ; %frem.loop_exit
+; SI-NEXT:  .LBB15_17: ; %frem.loop_exit
 ; SI-NEXT:    s_sub_i32 s6, s9, 25
 ; SI-NEXT:    v_ldexp_f64 v[2:3], v[3:4], s6
 ; SI-NEXT:    s_mov_b32 s6, -1
@@ -18444,11 +18375,11 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; SI-NEXT:    v_cndmask_b32_e32 v3, v3, v5, vcc
 ; SI-NEXT:    v_mov_b32_e32 v4, s5
 ; SI-NEXT:    v_bfi_b32 v3, s6, v3, v4
-; SI-NEXT:    s_branch .LBB15_20
-; SI-NEXT:  .LBB15_19:
+; SI-NEXT:    s_branch .LBB15_19
+; SI-NEXT:  .LBB15_18:
 ; SI-NEXT:    v_mov_b32_e32 v2, s6
 ; SI-NEXT:    v_mov_b32_e32 v3, s7
-; SI-NEXT:  .LBB15_20: ; %Flow49
+; SI-NEXT:  .LBB15_19: ; %Flow49
 ; SI-NEXT:    v_mov_b32_e32 v4, 0
 ; SI-NEXT:    v_mov_b32_e32 v5, 0x7ff00000
 ; SI-NEXT:    v_cmp_nge_f64_e64 vcc, |s[2:3]|, v[4:5]
@@ -18473,8 +18404,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; CI-NEXT:    s_mov_b32 s5, s3
 ; CI-NEXT:    buffer_load_dwordx4 v[0:3], off, s[4:7], 0
 ; CI-NEXT:    s_waitcnt vmcnt(0)
-; CI-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[0:1]|, 1.0
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[0:1]|, 1.0
 ; CI-NEXT:    s_cbranch_vccz .LBB15_2
 ; CI-NEXT:  ; %bb.1: ; %frem.else16
 ; CI-NEXT:    v_cmp_eq_f64_e64 vcc, |v[0:1]|, 1.0
@@ -18529,8 +18459,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; CI-NEXT:    v_cndmask_b32_e32 v4, v4, v6, vcc
 ; CI-NEXT:    v_bfi_b32 v5, s2, v5, v1
 ; CI-NEXT:  .LBB15_9:
-; CI-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[2:3]|, 1.0
-; CI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CI-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[2:3]|, 1.0
 ; CI-NEXT:    s_cbranch_vccz .LBB15_11
 ; CI-NEXT:  ; %bb.10: ; %frem.else
 ; CI-NEXT:    v_cmp_eq_f64_e64 vcc, |v[2:3]|, 1.0
@@ -18607,8 +18536,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; VI-NEXT:    v_mov_b32_e32 v1, s3
 ; VI-NEXT:    flat_load_dwordx4 v[0:3], v[0:1]
 ; VI-NEXT:    s_waitcnt vmcnt(0)
-; VI-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[0:1]|, 1.0
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[0:1]|, 1.0
 ; VI-NEXT:    s_cbranch_vccz .LBB15_2
 ; VI-NEXT:  ; %bb.1: ; %frem.else16
 ; VI-NEXT:    v_cmp_eq_f64_e64 vcc, |v[0:1]|, 1.0
@@ -18663,8 +18591,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; VI-NEXT:    v_cndmask_b32_e32 v4, v4, v6, vcc
 ; VI-NEXT:    v_bfi_b32 v5, s2, v5, v1
 ; VI-NEXT:  .LBB15_9:
-; VI-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[2:3]|, 1.0
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[2:3]|, 1.0
 ; VI-NEXT:    s_cbranch_vccz .LBB15_11
 ; VI-NEXT:  ; %bb.10: ; %frem.else
 ; VI-NEXT:    v_cmp_eq_f64_e64 vcc, |v[2:3]|, 1.0
@@ -18740,8 +18667,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX9-NEXT:    s_waitcnt lgkmcnt(0)
 ; GFX9-NEXT:    global_load_dwordx4 v[0:3], v0, s[2:3]
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
-; GFX9-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[0:1]|, 1.0
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[0:1]|, 1.0
 ; GFX9-NEXT:    s_cbranch_vccz .LBB15_2
 ; GFX9-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX9-NEXT:    v_cmp_eq_f64_e64 vcc, |v[0:1]|, 1.0
@@ -18796,8 +18722,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX9-NEXT:    v_cndmask_b32_e32 v4, v4, v6, vcc
 ; GFX9-NEXT:    v_bfi_b32 v5, s2, v5, v1
 ; GFX9-NEXT:  .LBB15_9:
-; GFX9-NEXT:    v_cmp_ngt_f64_e64 s[2:3], |v[2:3]|, 1.0
-; GFX9-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX9-NEXT:    v_cmp_ngt_f64_e64 vcc, |v[2:3]|, 1.0
 ; GFX9-NEXT:    s_cbranch_vccz .LBB15_11
 ; GFX9-NEXT:  ; %bb.10: ; %frem.else
 ; GFX9-NEXT:    v_cmp_eq_f64_e64 vcc, |v[2:3]|, 1.0
@@ -18872,8 +18797,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX10-NEXT:    s_waitcnt lgkmcnt(0)
 ; GFX10-NEXT:    global_load_dwordx4 v[0:3], v0, s[2:3]
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
-; GFX10-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, 1.0
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, 1.0
 ; GFX10-NEXT:    s_cbranch_vccz .LBB15_2
 ; GFX10-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX10-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, 1.0
@@ -18894,8 +18818,8 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX10-NEXT:    v_frexp_mant_f64_e64 v[4:5], |v[0:1]|
 ; GFX10-NEXT:    v_frexp_exp_i32_f64_e32 v6, v[0:1]
 ; GFX10-NEXT:    v_ldexp_f64 v[4:5], v[4:5], 26
-; GFX10-NEXT:    v_add_nc_u32_e32 v8, -1, v6
 ; GFX10-NEXT:    v_readfirstlane_b32 s2, v6
+; GFX10-NEXT:    v_add_nc_u32_e32 v8, -1, v6
 ; GFX10-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 27, v8
 ; GFX10-NEXT:    s_cbranch_vccnz .LBB15_8
 ; GFX10-NEXT:  ; %bb.5: ; %frem.loop_body23.preheader
@@ -18929,8 +18853,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX10-NEXT:    v_cndmask_b32_e32 v4, v4, v6, vcc_lo
 ; GFX10-NEXT:    v_bfi_b32 v5, 0x7fffffff, v5, v1
 ; GFX10-NEXT:  .LBB15_9:
-; GFX10-NEXT:    v_cmp_ngt_f64_e64 s2, |v[2:3]|, 1.0
-; GFX10-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX10-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[2:3]|, 1.0
 ; GFX10-NEXT:    s_cbranch_vccz .LBB15_11
 ; GFX10-NEXT:  ; %bb.10: ; %frem.else
 ; GFX10-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, 1.0
@@ -18951,8 +18874,8 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX10-NEXT:    v_frexp_mant_f64_e64 v[6:7], |v[2:3]|
 ; GFX10-NEXT:    v_frexp_exp_i32_f64_e32 v8, v[2:3]
 ; GFX10-NEXT:    v_ldexp_f64 v[6:7], v[6:7], 26
-; GFX10-NEXT:    v_add_nc_u32_e32 v10, -1, v8
 ; GFX10-NEXT:    v_readfirstlane_b32 s2, v8
+; GFX10-NEXT:    v_add_nc_u32_e32 v10, -1, v8
 ; GFX10-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 27, v10
 ; GFX10-NEXT:    s_cbranch_vccnz .LBB15_17
 ; GFX10-NEXT:  ; %bb.14: ; %frem.loop_body.preheader
@@ -19003,8 +18926,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX11-NEXT:    s_waitcnt lgkmcnt(0)
 ; GFX11-NEXT:    global_load_b128 v[0:3], v0, s[2:3]
 ; GFX11-NEXT:    s_waitcnt vmcnt(0)
-; GFX11-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, 1.0
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, 1.0
 ; GFX11-NEXT:    s_cbranch_vccz .LBB15_2
 ; GFX11-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX11-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, 1.0
@@ -19028,9 +18950,9 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX11-NEXT:    v_frexp_exp_i32_f64_e32 v6, v[0:1]
 ; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX11-NEXT:    v_ldexp_f64 v[4:5], v[4:5], 26
-; GFX11-NEXT:    v_add_nc_u32_e32 v8, -1, v6
 ; GFX11-NEXT:    v_readfirstlane_b32 s2, v6
-; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX11-NEXT:    v_add_nc_u32_e32 v8, -1, v6
+; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX11-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 27, v8
 ; GFX11-NEXT:    s_cbranch_vccnz .LBB15_8
 ; GFX11-NEXT:  ; %bb.5: ; %frem.loop_body23.preheader
@@ -19068,8 +18990,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX11-NEXT:    v_bfi_b32 v5, 0x7fffffff, v5, v1
 ; GFX11-NEXT:  .LBB15_9:
-; GFX11-NEXT:    v_cmp_ngt_f64_e64 s2, |v[2:3]|, 1.0
-; GFX11-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX11-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[2:3]|, 1.0
 ; GFX11-NEXT:    s_cbranch_vccz .LBB15_11
 ; GFX11-NEXT:  ; %bb.10: ; %frem.else
 ; GFX11-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, 1.0
@@ -19093,9 +19014,9 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX11-NEXT:    v_frexp_exp_i32_f64_e32 v8, v[2:3]
 ; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX11-NEXT:    v_ldexp_f64 v[6:7], v[6:7], 26
-; GFX11-NEXT:    v_add_nc_u32_e32 v10, -1, v8
 ; GFX11-NEXT:    v_readfirstlane_b32 s2, v8
-; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX11-NEXT:    v_add_nc_u32_e32 v10, -1, v8
+; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX11-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 27, v10
 ; GFX11-NEXT:    s_cbranch_vccnz .LBB15_17
 ; GFX11-NEXT:  ; %bb.14: ; %frem.loop_body.preheader
@@ -19149,8 +19070,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX1150-NEXT:    s_waitcnt lgkmcnt(0)
 ; GFX1150-NEXT:    global_load_b128 v[0:3], v0, s[2:3]
 ; GFX1150-NEXT:    s_waitcnt vmcnt(0)
-; GFX1150-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, 1.0
-; GFX1150-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX1150-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, 1.0
 ; GFX1150-NEXT:    s_cbranch_vccz .LBB15_2
 ; GFX1150-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX1150-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, 1.0
@@ -19174,9 +19094,9 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX1150-NEXT:    v_frexp_exp_i32_f64_e32 v6, v[0:1]
 ; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1150-NEXT:    v_ldexp_f64 v[4:5], v[4:5], 26
-; GFX1150-NEXT:    v_add_nc_u32_e32 v8, -1, v6
 ; GFX1150-NEXT:    v_readfirstlane_b32 s2, v6
-; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX1150-NEXT:    v_add_nc_u32_e32 v8, -1, v6
+; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1150-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 27, v8
 ; GFX1150-NEXT:    s_cbranch_vccnz .LBB15_8
 ; GFX1150-NEXT:  ; %bb.5: ; %frem.loop_body23.preheader
@@ -19214,8 +19134,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1150-NEXT:    v_bfi_b32 v5, 0x7fffffff, v5, v1
 ; GFX1150-NEXT:  .LBB15_9:
-; GFX1150-NEXT:    v_cmp_ngt_f64_e64 s2, |v[2:3]|, 1.0
-; GFX1150-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX1150-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[2:3]|, 1.0
 ; GFX1150-NEXT:    s_cbranch_vccz .LBB15_11
 ; GFX1150-NEXT:  ; %bb.10: ; %frem.else
 ; GFX1150-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, 1.0
@@ -19239,9 +19158,9 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX1150-NEXT:    v_frexp_exp_i32_f64_e32 v8, v[2:3]
 ; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1150-NEXT:    v_ldexp_f64 v[6:7], v[6:7], 26
-; GFX1150-NEXT:    v_add_nc_u32_e32 v10, -1, v8
 ; GFX1150-NEXT:    v_readfirstlane_b32 s2, v8
-; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX1150-NEXT:    v_add_nc_u32_e32 v10, -1, v8
+; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1150-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 27, v10
 ; GFX1150-NEXT:    s_cbranch_vccnz .LBB15_17
 ; GFX1150-NEXT:  ; %bb.14: ; %frem.loop_body.preheader
@@ -19295,8 +19214,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX1200-NEXT:    s_wait_kmcnt 0x0
 ; GFX1200-NEXT:    global_load_b128 v[0:3], v0, s[2:3]
 ; GFX1200-NEXT:    s_wait_loadcnt 0x0
-; GFX1200-NEXT:    v_cmp_ngt_f64_e64 s2, |v[0:1]|, 1.0
-; GFX1200-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
+; GFX1200-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[0:1]|, 1.0
 ; GFX1200-NEXT:    s_cbranch_vccz .LBB15_2
 ; GFX1200-NEXT:  ; %bb.1: ; %frem.else16
 ; GFX1200-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[0:1]|, 1.0
@@ -19320,9 +19238,9 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX1200-NEXT:    v_frexp_exp_i32_f64_e32 v6, v[0:1]
 ; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1200-NEXT:    v_ldexp_f64 v[4:5], v[4:5], 26
-; GFX1200-NEXT:    v_add_nc_u32_e32 v8, -1, v6
 ; GFX1200-NEXT:    v_readfirstlane_b32 s2, v6
-; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX1200-NEXT:    v_add_nc_u32_e32 v8, -1, v6
+; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1200-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 27, v8
 ; GFX1200-NEXT:    s_cbranch_vccnz .LBB15_8
 ; GFX1200-NEXT:  ; %bb.5: ; %frem.loop_body23.preheader
@@ -19362,9 +19280,7 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1200-NEXT:    v_bfi_b32 v5, 0x7fffffff, v5, v1
 ; GFX1200-NEXT:  .LBB15_9:
-; GFX1200-NEXT:    v_cmp_ngt_f64_e64 s2, |v[2:3]|, 1.0
-; GFX1200-NEXT:    s_and_b32 vcc_lo, exec_lo, s2
-; GFX1200-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX1200-NEXT:    v_cmp_ngt_f64_e64 vcc_lo, |v[2:3]|, 1.0
 ; GFX1200-NEXT:    s_cbranch_vccz .LBB15_11
 ; GFX1200-NEXT:  ; %bb.10: ; %frem.else
 ; GFX1200-NEXT:    v_cmp_eq_f64_e64 vcc_lo, |v[2:3]|, 1.0
@@ -19390,9 +19306,9 @@ define amdgpu_kernel void @frem_v2f64_const_one_denum(ptr addrspace(1) %out, ptr
 ; GFX1200-NEXT:    v_frexp_exp_i32_f64_e32 v8, v[2:3]
 ; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1200-NEXT:    v_ldexp_f64 v[6:7], v[6:7], 26
-; GFX1200-NEXT:    v_add_nc_u32_e32 v10, -1, v8
 ; GFX1200-NEXT:    v_readfirstlane_b32 s2, v8
-; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX1200-NEXT:    v_add_nc_u32_e32 v10, -1, v8
+; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1200-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 27, v10
 ; GFX1200-NEXT:    s_cbranch_vccnz .LBB15_17
 ; GFX1200-NEXT:  ; %bb.14: ; %frem.loop_body.preheader
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i32.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i32.ll
index 6cc43c0785048b7..685aae11c869594 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i32.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i32.ll
@@ -414,8 +414,7 @@ define amdgpu_ps void @non_cst_non_compare_input(ptr addrspace(1) %out, i32 %tid
 ; GFX10-NEXT:  ; %bb.3: ; %A
 ; GFX10-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v2
 ; GFX10-NEXT:    s_andn2_b32 s0, s0, exec_lo
-; GFX10-NEXT:    s_and_b32 s2, vcc_lo, exec_lo
-; GFX10-NEXT:    s_or_b32 s0, s0, s2
+; GFX10-NEXT:    s_or_b32 s0, s0, vcc_lo
 ; GFX10-NEXT:  ; %bb.4: ; %exit
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s1
 ; GFX10-NEXT:    v_cndmask_b32_e64 v2, 0, 1, s0
@@ -438,8 +437,7 @@ define amdgpu_ps void @non_cst_non_compare_input(ptr addrspace(1) %out, i32 %tid
 ; GFX11-NEXT:  ; %bb.3: ; %A
 ; GFX11-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v2
 ; GFX11-NEXT:    s_and_not1_b32 s0, s0, exec_lo
-; GFX11-NEXT:    s_and_b32 s2, vcc_lo, exec_lo
-; GFX11-NEXT:    s_or_b32 s0, s0, s2
+; GFX11-NEXT:    s_or_b32 s0, s0, vcc_lo
 ; GFX11-NEXT:  ; %bb.4: ; %exit
 ; GFX11-NEXT:    s_or_b32 exec_lo, exec_lo, s1
 ; GFX11-NEXT:    v_cndmask_b32_e64 v2, 0, 1, s0
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i64.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i64.ll
index 68da5a9866184f9..5f824faf9a4650e 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ballot.i64.ll
@@ -417,8 +417,7 @@ define amdgpu_ps void @non_cst_non_compare_input(ptr addrspace(1) %out, i32 %tid
 ; CHECK-NEXT:  ; %bb.3: ; %A
 ; CHECK-NEXT:    v_cmp_ne_u32_e32 vcc, 0, v2
 ; CHECK-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec
-; CHECK-NEXT:    s_and_b64 s[4:5], vcc, exec
-; CHECK-NEXT:    s_or_b64 s[0:1], s[0:1], s[4:5]
+; CHECK-NEXT:    s_or_b64 s[0:1], s[0:1], vcc
 ; CHECK-NEXT:  ; %bb.4: ; %exit
 ; CHECK-NEXT:    s_or_b64 exec, exec, s[2:3]
 ; CHECK-NEXT:    v_cndmask_b32_e64 v2, 0, 1, s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.div.fmas.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.div.fmas.ll
index f4a18171134576c..45e574e013154a8 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.div.fmas.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.div.fmas.ll
@@ -255,11 +255,10 @@ define amdgpu_kernel void @test_div_fmas_f32_i1_phi_vcc(ptr addrspace(1) %out, p
 ; GCN-NEXT:    buffer_load_dword v0, off, s[4:7], 0
 ; GCN-NEXT:    s_waitcnt vmcnt(0)
 ; GCN-NEXT:    v_cmp_ne_u32_e32 vcc, 0, v0
-; GCN-NEXT:    s_and_b64 vcc, vcc, exec
 ; GCN-NEXT:  .LBB9_2: ; %exit
 ; GCN-NEXT:    s_or_b64 exec, exec, s[10:11]
 ; GCN-NEXT:    s_waitcnt vmcnt(0)
-; GCN-NEXT:    s_nop 0
+; GCN-NEXT:    s_nop 1
 ; GCN-NEXT:    v_div_fmas_f32 v0, v1, v2, v3
 ; GCN-NEXT:    s_mov_b32 s10, -1
 ; GCN-NEXT:    s_mov_b32 s11, s3
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.kill.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.kill.ll
index 8973dc21b118f31..d32e9428911a49f 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.kill.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.kill.ll
@@ -1574,8 +1574,7 @@ define amdgpu_ps void @kill_with_loop_exit(float inreg %inp0, float inreg %inp1,
 ; SI-NEXT:    v_mov_b32_e32 v0, 0x43000000
 ; SI-NEXT:    v_cmp_lt_f32_e32 vcc, s0, v0
 ; SI-NEXT:    v_cmp_lt_f32_e64 s[0:1], s1, v0
-; SI-NEXT:    s_and_b64 s[0:1], vcc, s[0:1]
-; SI-NEXT:    s_and_b64 vcc, exec, s[0:1]
+; SI-NEXT:    s_and_b64 vcc, vcc, s[0:1]
 ; SI-NEXT:    v_mov_b32_e32 v0, 1.0
 ; SI-NEXT:    s_cbranch_vccnz .LBB25_5
 ; SI-NEXT:  ; %bb.1: ; %.preheader1.preheader
@@ -1607,8 +1606,7 @@ define amdgpu_ps void @kill_with_loop_exit(float inreg %inp0, float inreg %inp1,
 ; GFX10-NEXT:    v_cmp_gt_f32_e64 s[4:5], 0x43000000, s0
 ; GFX10-NEXT:    v_cmp_gt_f32_e64 s[0:1], 0x43000000, s1
 ; GFX10-NEXT:    v_mov_b32_e32 v0, 1.0
-; GFX10-NEXT:    s_and_b64 s[0:1], s[4:5], s[0:1]
-; GFX10-NEXT:    s_and_b64 vcc, exec, s[0:1]
+; GFX10-NEXT:    s_and_b64 vcc, s[4:5], s[0:1]
 ; GFX10-NEXT:    s_cbranch_vccnz .LBB25_5
 ; GFX10-NEXT:  ; %bb.1: ; %.preheader1.preheader
 ; GFX10-NEXT:    v_cmp_ngt_f32_e64 s[0:1], s6, 0
@@ -1639,8 +1637,7 @@ define amdgpu_ps void @kill_with_loop_exit(float inreg %inp0, float inreg %inp1,
 ; GFX11-NEXT:    v_cmp_gt_f32_e64 s[4:5], 0x43000000, s0
 ; GFX11-NEXT:    v_cmp_gt_f32_e64 s[0:1], 0x43000000, s1
 ; GFX11-NEXT:    v_mov_b32_e32 v0, 1.0
-; GFX11-NEXT:    s_and_b64 s[0:1], s[4:5], s[0:1]
-; GFX11-NEXT:    s_and_b64 vcc, exec, s[0:1]
+; GFX11-NEXT:    s_and_b64 vcc, s[4:5], s[0:1]
 ; GFX11-NEXT:    s_cbranch_vccnz .LBB25_5
 ; GFX11-NEXT:  ; %bb.1: ; %.preheader1.preheader
 ; GFX11-NEXT:    v_cmp_ngt_f32_e64 s[0:1], s6, 0
diff --git a/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fadd.ll b/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fadd.ll
index 50669d1eb6cba7a..c5bdc47f1c50101 100644
--- a/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fadd.ll
+++ b/llvm/test/CodeGen/AMDGPU/local-atomicrmw-fadd.ll
@@ -8404,13 +8404,12 @@ define amdgpu_kernel void @local_ds_fadd(ptr addrspace(1) %out, ptr addrspace(3)
 ; GFX7-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX7-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX7-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX7-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX7-NEXT:    v_readfirstlane_b32 s8, v1
 ; GFX7-NEXT:    v_readlane_b32 s9, v2, s3
 ; GFX7-NEXT:    s_mov_b32 m0, s3
+; GFX7-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7-NEXT:    v_writelane_b32 v0, s8, m0
 ; GFX7-NEXT:    v_add_f32_e32 v1, s9, v1
-; GFX7-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX7-NEXT:    s_cbranch_vccnz .LBB28_8
 ; GFX7-NEXT:  ; %bb.9: ; %ComputeEnd
 ; GFX7-NEXT:    v_mbcnt_lo_u32_b32_e64 v2, exec_lo, 0
@@ -8530,13 +8529,12 @@ define amdgpu_kernel void @local_ds_fadd(ptr addrspace(1) %out, ptr addrspace(3)
 ; GFX6-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX6-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX6-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX6-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX6-NEXT:    v_readfirstlane_b32 s8, v1
 ; GFX6-NEXT:    v_readlane_b32 s9, v2, s3
 ; GFX6-NEXT:    s_mov_b32 m0, s3
+; GFX6-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX6-NEXT:    v_writelane_b32 v0, s8, m0
 ; GFX6-NEXT:    v_add_f32_e32 v1, s9, v1
-; GFX6-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX6-NEXT:    s_cbranch_vccnz .LBB28_8
 ; GFX6-NEXT:  ; %bb.9: ; %ComputeEnd
 ; GFX6-NEXT:    v_mbcnt_lo_u32_b32_e64 v2, exec_lo, 0
@@ -9253,13 +9251,12 @@ define amdgpu_kernel void @local_ds_fadd_one_as(ptr addrspace(1) %out, ptr addrs
 ; GFX7-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX7-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX7-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX7-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX7-NEXT:    v_readfirstlane_b32 s8, v1
 ; GFX7-NEXT:    v_readlane_b32 s9, v2, s3
 ; GFX7-NEXT:    s_mov_b32 m0, s3
+; GFX7-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX7-NEXT:    v_writelane_b32 v0, s8, m0
 ; GFX7-NEXT:    v_add_f32_e32 v1, s9, v1
-; GFX7-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX7-NEXT:    s_cbranch_vccnz .LBB29_8
 ; GFX7-NEXT:  ; %bb.9: ; %ComputeEnd
 ; GFX7-NEXT:    v_mbcnt_lo_u32_b32_e64 v2, exec_lo, 0
@@ -9379,13 +9376,12 @@ define amdgpu_kernel void @local_ds_fadd_one_as(ptr addrspace(1) %out, ptr addrs
 ; GFX6-NEXT:    s_ff1_i32_b64 s3, s[0:1]
 ; GFX6-NEXT:    s_lshl_b64 s[6:7], 1, s3
 ; GFX6-NEXT:    s_andn2_b64 s[0:1], s[0:1], s[6:7]
-; GFX6-NEXT:    v_cmp_ne_u64_e64 s[6:7], s[0:1], 0
 ; GFX6-NEXT:    v_readfirstlane_b32 s8, v1
 ; GFX6-NEXT:    v_readlane_b32 s9, v2, s3
 ; GFX6-NEXT:    s_mov_b32 m0, s3
+; GFX6-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; GFX6-NEXT:    v_writelane_b32 v0, s8, m0
 ; GFX6-NEXT:    v_add_f32_e32 v1, s9, v1
-; GFX6-NEXT:    s_and_b64 vcc, exec, s[6:7]
 ; GFX6-NEXT:    s_cbranch_vccnz .LBB29_8
 ; GFX6-NEXT:  ; %bb.9: ; %ComputeEnd
 ; GFX6-NEXT:    v_mbcnt_lo_u32_b32_e64 v2, exec_lo, 0
diff --git a/llvm/test/CodeGen/AMDGPU/long-branch-reserve-register.ll b/llvm/test/CodeGen/AMDGPU/long-branch-reserve-register.ll
index 8c7dcd9253e8a76..99d53b3cbcc7813 100644
--- a/llvm/test/CodeGen/AMDGPU/long-branch-reserve-register.ll
+++ b/llvm/test/CodeGen/AMDGPU/long-branch-reserve-register.ll
@@ -104,15 +104,14 @@ define amdgpu_kernel void @uniform_conditional_min_long_forward_vcnd_branch(ptr
 ; GCN:       ; %bb.0: ; %bb0
 ; GCN-NEXT:    s_load_dword s0, s[4:5], 0xb
 ; GCN-NEXT:    s_waitcnt lgkmcnt(0)
-; GCN-NEXT:    v_cmp_eq_f32_e64 s[2:3], s0, 0
-; GCN-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GCN-NEXT:    v_cmp_eq_f32_e64 vcc, s0, 0
 ; GCN-NEXT:    s_cbranch_vccz .LBB2_1
 ; GCN-NEXT:  ; %bb.3: ; %bb0
-; GCN-NEXT:    s_getpc_b64 s[8:9]
+; GCN-NEXT:    s_getpc_b64 s[2:3]
 ; GCN-NEXT:  .Lpost_getpc1:
-; GCN-NEXT:    s_add_u32 s8, s8, (.LBB2_2-.Lpost_getpc1)&4294967295
-; GCN-NEXT:    s_addc_u32 s9, s9, (.LBB2_2-.Lpost_getpc1)>>32
-; GCN-NEXT:    s_setpc_b64 s[8:9]
+; GCN-NEXT:    s_add_u32 s2, s2, (.LBB2_2-.Lpost_getpc1)&4294967295
+; GCN-NEXT:    s_addc_u32 s3, s3, (.LBB2_2-.Lpost_getpc1)>>32
+; GCN-NEXT:    s_setpc_b64 s[2:3]
 ; GCN-NEXT:  .LBB2_1: ; %bb2
 ; GCN-NEXT:    ;;#ASMSTART
 ; GCN-NEXT:     ; 32 bytes
diff --git a/llvm/test/CodeGen/AMDGPU/loop_break.ll b/llvm/test/CodeGen/AMDGPU/loop_break.ll
index a63df79e1d1b486..4617ef79f6bf18c 100644
--- a/llvm/test/CodeGen/AMDGPU/loop_break.ll
+++ b/llvm/test/CodeGen/AMDGPU/loop_break.ll
@@ -52,8 +52,7 @@ define amdgpu_kernel void @break_loop(i32 %arg) #0 {
 ; GCN-NEXT:    s_waitcnt vmcnt(0)
 ; GCN-NEXT:    s_andn2_b64 s[4:5], s[4:5], exec
 ; GCN-NEXT:    v_cmp_ge_i32_e32 vcc, v0, v1
-; GCN-NEXT:    s_and_b64 s[8:9], vcc, exec
-; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], s[8:9]
+; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], vcc
 ; GCN-NEXT:  .LBB0_3: ; %Flow
 ; GCN-NEXT:    ; in Loop: Header=BB0_1 Depth=1
 ; GCN-NEXT:    s_and_b64 s[8:9], exec, s[4:5]
@@ -221,8 +220,7 @@ define amdgpu_kernel void @constexpr_phi_cond_break_loop(i32 %arg) #0 {
 ; GCN-NEXT:    s_waitcnt vmcnt(0)
 ; GCN-NEXT:    s_andn2_b64 s[4:5], s[4:5], exec
 ; GCN-NEXT:    v_cmp_ge_i32_e32 vcc, v0, v1
-; GCN-NEXT:    s_and_b64 s[8:9], vcc, exec
-; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], s[8:9]
+; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], vcc
 ; GCN-NEXT:  .LBB2_3: ; %Flow
 ; GCN-NEXT:    ; in Loop: Header=BB2_1 Depth=1
 ; GCN-NEXT:    s_and_b64 s[8:9], exec, s[4:5]
@@ -312,8 +310,7 @@ define amdgpu_kernel void @true_phi_cond_break_loop(i32 %arg) #0 {
 ; GCN-NEXT:    s_waitcnt vmcnt(0)
 ; GCN-NEXT:    s_andn2_b64 s[4:5], s[4:5], exec
 ; GCN-NEXT:    v_cmp_ge_i32_e32 vcc, v0, v1
-; GCN-NEXT:    s_and_b64 s[8:9], vcc, exec
-; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], s[8:9]
+; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], vcc
 ; GCN-NEXT:  .LBB3_3: ; %Flow
 ; GCN-NEXT:    ; in Loop: Header=BB3_1 Depth=1
 ; GCN-NEXT:    s_and_b64 s[8:9], exec, s[4:5]
@@ -402,8 +399,7 @@ define amdgpu_kernel void @false_phi_cond_break_loop(i32 %arg) #0 {
 ; GCN-NEXT:    s_waitcnt vmcnt(0)
 ; GCN-NEXT:    s_andn2_b64 s[4:5], s[4:5], exec
 ; GCN-NEXT:    v_cmp_ge_i32_e32 vcc, v0, v1
-; GCN-NEXT:    s_and_b64 s[8:9], vcc, exec
-; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], s[8:9]
+; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], vcc
 ; GCN-NEXT:  .LBB4_3: ; %Flow
 ; GCN-NEXT:    ; in Loop: Header=BB4_1 Depth=1
 ; GCN-NEXT:    s_and_b64 s[8:9], exec, s[4:5]
@@ -496,8 +492,7 @@ define amdgpu_kernel void @invert_true_phi_cond_break_loop(i32 %arg) #0 {
 ; GCN-NEXT:    s_waitcnt vmcnt(0)
 ; GCN-NEXT:    s_andn2_b64 s[4:5], s[4:5], exec
 ; GCN-NEXT:    v_cmp_ge_i32_e32 vcc, v0, v1
-; GCN-NEXT:    s_and_b64 s[8:9], vcc, exec
-; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], s[8:9]
+; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], vcc
 ; GCN-NEXT:  .LBB5_3: ; %Flow
 ; GCN-NEXT:    ; in Loop: Header=BB5_1 Depth=1
 ; GCN-NEXT:    s_xor_b64 s[8:9], s[4:5], -1
diff --git a/llvm/test/CodeGen/AMDGPU/memcpy-crash-issue63986.ll b/llvm/test/CodeGen/AMDGPU/memcpy-crash-issue63986.ll
index d90c9771c3e4a54..758bd06408415b9 100644
--- a/llvm/test/CodeGen/AMDGPU/memcpy-crash-issue63986.ll
+++ b/llvm/test/CodeGen/AMDGPU/memcpy-crash-issue63986.ll
@@ -19,10 +19,9 @@ define void @issue63986(i64 %0, i64 %idxprom, ptr inreg %ptr) {
 ; CHECK-NEXT:    flat_load_dwordx4 v[10:13], v[6:7]
 ; CHECK-NEXT:    v_add_co_u32_e32 v6, vcc, s4, v8
 ; CHECK-NEXT:    s_add_u32 s4, s4, 16
-; CHECK-NEXT:    s_addc_u32 s5, s5, 0
-; CHECK-NEXT:    v_cmp_lt_u64_e64 s[6:7], s[4:5], 32
 ; CHECK-NEXT:    v_addc_co_u32_e32 v7, vcc, v9, v7, vcc
-; CHECK-NEXT:    s_and_b64 vcc, exec, s[6:7]
+; CHECK-NEXT:    s_addc_u32 s5, s5, 0
+; CHECK-NEXT:    v_cmp_lt_u64_e64 vcc, s[4:5], 32
 ; CHECK-NEXT:    s_waitcnt vmcnt(0) lgkmcnt(0)
 ; CHECK-NEXT:    flat_store_dwordx4 v[6:7], v[10:13]
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB0_1
diff --git a/llvm/test/CodeGen/AMDGPU/memcpy_const_compare.ll b/llvm/test/CodeGen/AMDGPU/memcpy_const_compare.ll
index bb9011a3c43d74c..69974521887e71e 100644
--- a/llvm/test/CodeGen/AMDGPU/memcpy_const_compare.ll
+++ b/llvm/test/CodeGen/AMDGPU/memcpy_const_compare.ll
@@ -34,13 +34,11 @@ define amdgpu_kernel void @_start() {
 ; CHECK-NEXT:    s_add_u32 s0, s0, 16
 ; CHECK-NEXT:    s_addc_u32 s1, s1, 0
 ; CHECK-NEXT:    v_pk_mov_b32 v[0:1], s[2:3], s[2:3] op_sel:[0,1]
-; CHECK-NEXT:    v_cmp_lt_u64_e64 s[2:3], s[0:1], 16
-; CHECK-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; CHECK-NEXT:    v_cmp_lt_u64_e64 vcc, s[0:1], 16
 ; CHECK-NEXT:    s_waitcnt vmcnt(0)
 ; CHECK-NEXT:    flat_store_dwordx4 v[0:1], v[2:5]
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB0_1
 ; CHECK-NEXT:  ; %bb.2: ; %dynamic-memcpy-expansion-residual-cond
-; FIXME: Compare should be evaluated at compile time
 ; CHECK-NEXT:    s_cmp_eq_u64 13, 0
 ; CHECK-NEXT:    s_cbranch_scc1 .LBB0_5
 ; CHECK-NEXT:  ; %bb.3: ; %dynamic-memcpy-expansion-residual-body.preheader
@@ -72,13 +70,13 @@ define amdgpu_kernel void @_start() {
 ; CHECK-NEXT:    s_add_u32 s0, s0, 1
 ; CHECK-NEXT:    s_addc_u32 s1, s1, 0
 ; CHECK-NEXT:    v_pk_mov_b32 v[0:1], s[6:7], s[6:7] op_sel:[0,1]
-; CHECK-NEXT:    v_cmp_lt_u64_e64 s[6:7], s[0:1], 13
-; CHECK-NEXT:    s_and_b64 vcc, exec, s[6:7]
+; CHECK-NEXT:    v_cmp_lt_u64_e64 vcc, s[0:1], 13
 ; CHECK-NEXT:    s_waitcnt vmcnt(0) lgkmcnt(0)
 ; CHECK-NEXT:    flat_store_byte v[0:1], v2
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB0_4
 ; CHECK-NEXT:  .LBB0_5: ; %dynamic-memcpy-post-expansion
 ; CHECK-NEXT:    s_endpgm
+; FIXME: Compare should be evaluated at compile time
   %src_ptr = getelementptr inbounds [128 x i8], ptr @src_array, i64 0, i64 0
   %dst_ptr = getelementptr inbounds [128 x i8], ptr @dst_array, i64 0, i64 0
   call void @llvm.memcpy.p0.p0.i64(ptr %dst_ptr, ptr %src_ptr, i64 add (i64 sub (i64 16, i64 ptrtoint (ptr addrspacecast (ptr addrspace(4) null to ptr) to i64)), i64 13), i1 false)
diff --git a/llvm/test/CodeGen/AMDGPU/memintrinsic-unroll.ll b/llvm/test/CodeGen/AMDGPU/memintrinsic-unroll.ll
index 1d81216dda2995b..adc3f9440feca85 100644
--- a/llvm/test/CodeGen/AMDGPU/memintrinsic-unroll.ll
+++ b/llvm/test/CodeGen/AMDGPU/memintrinsic-unroll.ll
@@ -13,11 +13,11 @@ define void @memcpy_p0_p0_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(0)
 ; CHECK-LABEL: memcpy_p0_p0_sz2048:
 ; CHECK:       ; %bb.0: ; %entry
 ; CHECK-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; CHECK-NEXT:    s_mov_b64 s[4:5], 0
+; CHECK-NEXT:    s_mov_b64 s[6:7], 0
 ; CHECK-NEXT:  .LBB0_1: ; %static-memcpy-expansion-main-body
 ; CHECK-NEXT:    ; =>This Inner Loop Header: Depth=1
-; CHECK-NEXT:    v_add_co_u32 v20, vcc_lo, v2, s4
-; CHECK-NEXT:    v_add_co_ci_u32_e64 v21, null, s5, v3, vcc_lo
+; CHECK-NEXT:    v_add_co_u32 v20, vcc_lo, v2, s6
+; CHECK-NEXT:    v_add_co_ci_u32_e64 v21, null, s7, v3, vcc_lo
 ; CHECK-NEXT:    v_add_co_u32 v96, vcc_lo, v20, 48
 ; CHECK-NEXT:    v_add_co_ci_u32_e64 v97, null, 0, v21, vcc_lo
 ; CHECK-NEXT:    s_clause 0xf
@@ -37,13 +37,13 @@ define void @memcpy_p0_p0_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(0)
 ; CHECK-NEXT:    flat_load_dwordx4 v[80:83], v[96:97] offset:64
 ; CHECK-NEXT:    flat_load_dwordx4 v[84:87], v[96:97] offset:32
 ; CHECK-NEXT:    flat_load_dwordx4 v[96:99], v[96:97]
-; CHECK-NEXT:    v_add_co_u32 v100, vcc_lo, v0, s4
-; CHECK-NEXT:    s_add_u32 s4, s4, 0x100
-; CHECK-NEXT:    v_add_co_ci_u32_e64 v101, null, s5, v1, vcc_lo
-; CHECK-NEXT:    s_addc_u32 s5, s5, 0
-; CHECK-NEXT:    v_add_co_u32 v102, vcc_lo, v100, 48
-; CHECK-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
-; CHECK-NEXT:    v_add_co_ci_u32_e64 v103, null, 0, v101, vcc_lo
+; CHECK-NEXT:    v_add_co_u32 v100, vcc_lo, v0, s6
+; CHECK-NEXT:    s_add_u32 s6, s6, 0x100
+; CHECK-NEXT:    v_add_co_ci_u32_e64 v101, null, s7, v1, vcc_lo
+; CHECK-NEXT:    s_addc_u32 s7, s7, 0
+; CHECK-NEXT:    v_add_co_u32 v102, s4, v100, 48
+; CHECK-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[6:7]
+; CHECK-NEXT:    v_add_co_ci_u32_e64 v103, null, 0, v101, s4
 ; CHECK-NEXT:    s_waitcnt vmcnt(15) lgkmcnt(15)
 ; CHECK-NEXT:    flat_store_dwordx4 v[100:101], v[4:7] offset:64
 ; CHECK-NEXT:    s_waitcnt vmcnt(14) lgkmcnt(15)
@@ -76,7 +76,6 @@ define void @memcpy_p0_p0_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(0)
 ; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[84:87] offset:32
 ; CHECK-NEXT:    s_waitcnt vmcnt(0) lgkmcnt(15)
 ; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[96:99]
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB0_1
 ; CHECK-NEXT:  ; %bb.2: ; %static-memcpy-post-expansion
 ; CHECK-NEXT:    s_waitcnt lgkmcnt(0)
@@ -144,7 +143,7 @@ define void @memcpy_p0_p0_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(0)
 ; ALIGNED-NEXT:    v_add_co_ci_u32_e64 v23, null, 0, v25, vcc_lo
 ; ALIGNED-NEXT:    v_add_co_u32 v20, vcc_lo, v24, 3
 ; ALIGNED-NEXT:    v_add_co_ci_u32_e64 v21, null, 0, v25, vcc_lo
-; ALIGNED-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
+; ALIGNED-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(3)
 ; ALIGNED-NEXT:    flat_store_byte_d16_hi v[22:23], v102 offset:136
 ; ALIGNED-NEXT:    flat_store_byte v[20:21], v102 offset:137
@@ -167,7 +166,6 @@ define void @memcpy_p0_p0_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(0)
 ; ALIGNED-NEXT:    buffer_load_dword v100, off, s[0:3], s32 offset:232
 ; ALIGNED-NEXT:    buffer_load_dword v99, off, s[0:3], s32 offset:228
 ; ALIGNED-NEXT:    buffer_load_dword v98, off, s[0:3], s32 offset:224
-; ALIGNED-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(3)
 ; ALIGNED-NEXT:    flat_store_byte_d16_hi v[22:23], v101 offset:72
 ; ALIGNED-NEXT:    flat_store_byte v[20:21], v101 offset:73
@@ -777,8 +775,7 @@ define void @memcpy_p0_p0_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(0)
 ; UNROLL3-NEXT:    flat_store_dwordx4 v[16:17], v[8:11] offset:16
 ; UNROLL3-NEXT:    s_waitcnt vmcnt(0) lgkmcnt(2)
 ; UNROLL3-NEXT:    flat_store_dwordx4 v[16:17], v[12:15] offset:32
-; UNROLL3-NEXT:    v_cmp_gt_u64_e64 s6, 0x7e0, s[4:5]
-; UNROLL3-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
+; UNROLL3-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    s_cbranch_vccnz .LBB0_1
 ; UNROLL3-NEXT:  ; %bb.2: ; %static-memcpy-post-expansion
 ; UNROLL3-NEXT:    flat_load_dwordx4 v[4:7], v[2:3] offset:2016
@@ -856,8 +853,7 @@ define void @memcpy_p1_p1_sz2048(ptr addrspace(1) align 1 %dst, ptr addrspace(1)
 ; CHECK-NEXT:    global_store_dwordx4 v[100:101], v[84:87], off
 ; CHECK-NEXT:    s_waitcnt vmcnt(0)
 ; CHECK-NEXT:    global_store_dwordx4 v[100:101], v[96:99], off offset:16
-; CHECK-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
+; CHECK-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB1_1
 ; CHECK-NEXT:  ; %bb.2: ; %static-memcpy-post-expansion
 ; CHECK-NEXT:    s_setpc_b64 s[30:31]
@@ -934,7 +930,7 @@ define void @memcpy_p1_p1_sz2048(ptr addrspace(1) align 1 %dst, ptr addrspace(1)
 ; ALIGNED-NEXT:    buffer_load_dword v50, off, s[0:3], s32 offset:104
 ; ALIGNED-NEXT:    buffer_load_dword v49, off, s[0:3], s32 offset:100
 ; ALIGNED-NEXT:    buffer_load_dword v36, off, s[0:3], s32 offset:96
-; ALIGNED-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
+; ALIGNED-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(3)
 ; ALIGNED-NEXT:    global_store_byte_d16_hi v[20:21], v51, off offset:238
 ; ALIGNED-NEXT:    global_store_byte v[20:21], v51, off offset:236
@@ -960,7 +956,6 @@ define void @memcpy_p1_p1_sz2048(ptr addrspace(1) align 1 %dst, ptr addrspace(1)
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v31, 8, v31
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v103, 24, v30
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v30, 8, v30
-; ALIGNED-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(3)
 ; ALIGNED-NEXT:    global_store_byte_d16_hi v[20:21], v71, off offset:222
 ; ALIGNED-NEXT:    global_store_byte v[20:21], v71, off offset:220
@@ -1539,8 +1534,7 @@ define void @memcpy_p1_p1_sz2048(ptr addrspace(1) align 1 %dst, ptr addrspace(1)
 ; UNROLL3-NEXT:    global_store_dwordx4 v[16:17], v[8:11], off offset:16
 ; UNROLL3-NEXT:    s_waitcnt vmcnt(0)
 ; UNROLL3-NEXT:    global_store_dwordx4 v[16:17], v[12:15], off offset:32
-; UNROLL3-NEXT:    v_cmp_gt_u64_e64 s6, 0x7e0, s[4:5]
-; UNROLL3-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
+; UNROLL3-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    s_cbranch_vccnz .LBB1_1
 ; UNROLL3-NEXT:  ; %bb.2: ; %static-memcpy-post-expansion
 ; UNROLL3-NEXT:    global_load_dwordx4 v[4:7], v[2:3], off offset:2016
@@ -1559,13 +1553,13 @@ define void @memcpy_p0_p4_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(4)
 ; CHECK-LABEL: memcpy_p0_p4_sz2048:
 ; CHECK:       ; %bb.0: ; %entry
 ; CHECK-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; CHECK-NEXT:    s_mov_b64 s[4:5], 0
+; CHECK-NEXT:    s_mov_b64 s[6:7], 0
 ; CHECK-NEXT:  .LBB2_1: ; %static-memcpy-expansion-main-body
 ; CHECK-NEXT:    ; =>This Inner Loop Header: Depth=1
-; CHECK-NEXT:    v_add_co_u32 v96, vcc_lo, v2, s4
-; CHECK-NEXT:    v_add_co_ci_u32_e64 v97, null, s5, v3, vcc_lo
-; CHECK-NEXT:    v_add_co_u32 v100, vcc_lo, v0, s4
-; CHECK-NEXT:    s_add_u32 s4, s4, 0x100
+; CHECK-NEXT:    v_add_co_u32 v96, vcc_lo, v2, s6
+; CHECK-NEXT:    v_add_co_ci_u32_e64 v97, null, s7, v3, vcc_lo
+; CHECK-NEXT:    v_add_co_u32 v100, vcc_lo, v0, s6
+; CHECK-NEXT:    v_add_co_ci_u32_e64 v101, null, s7, v1, vcc_lo
 ; CHECK-NEXT:    s_clause 0xf
 ; CHECK-NEXT:    global_load_dwordx4 v[4:7], v[96:97], off offset:240
 ; CHECK-NEXT:    global_load_dwordx4 v[8:11], v[96:97], off offset:224
@@ -1583,11 +1577,11 @@ define void @memcpy_p0_p4_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(4)
 ; CHECK-NEXT:    global_load_dwordx4 v[80:83], v[96:97], off offset:32
 ; CHECK-NEXT:    global_load_dwordx4 v[84:87], v[96:97], off offset:16
 ; CHECK-NEXT:    global_load_dwordx4 v[96:99], v[96:97], off
-; CHECK-NEXT:    v_add_co_ci_u32_e64 v101, null, s5, v1, vcc_lo
-; CHECK-NEXT:    s_addc_u32 s5, s5, 0
-; CHECK-NEXT:    v_add_co_u32 v102, vcc_lo, v100, 48
-; CHECK-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
-; CHECK-NEXT:    v_add_co_ci_u32_e64 v103, null, 0, v101, vcc_lo
+; CHECK-NEXT:    s_add_u32 s6, s6, 0x100
+; CHECK-NEXT:    s_addc_u32 s7, s7, 0
+; CHECK-NEXT:    v_add_co_u32 v102, s4, v100, 48
+; CHECK-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[6:7]
+; CHECK-NEXT:    v_add_co_ci_u32_e64 v103, null, 0, v101, s4
 ; CHECK-NEXT:    s_waitcnt vmcnt(15)
 ; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[4:7] offset:192
 ; CHECK-NEXT:    s_waitcnt vmcnt(14)
@@ -1620,7 +1614,6 @@ define void @memcpy_p0_p4_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(4)
 ; CHECK-NEXT:    flat_store_dwordx4 v[100:101], v[84:87] offset:16
 ; CHECK-NEXT:    s_waitcnt vmcnt(0)
 ; CHECK-NEXT:    flat_store_dwordx4 v[100:101], v[96:99]
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB2_1
 ; CHECK-NEXT:  ; %bb.2: ; %static-memcpy-post-expansion
 ; CHECK-NEXT:    s_waitcnt lgkmcnt(0)
@@ -1632,27 +1625,27 @@ define void @memcpy_p0_p4_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(4)
 ; ALIGNED-NEXT:    s_mov_b64 s[4:5], 0
 ; ALIGNED-NEXT:  .LBB2_1: ; %static-memcpy-expansion-main-body
 ; ALIGNED-NEXT:    ; =>This Inner Loop Header: Depth=1
-; ALIGNED-NEXT:    v_add_co_u32 v8, vcc_lo, v2, s4
-; ALIGNED-NEXT:    v_add_co_ci_u32_e64 v9, null, s5, v3, vcc_lo
+; ALIGNED-NEXT:    v_add_co_u32 v4, vcc_lo, v2, s4
+; ALIGNED-NEXT:    v_add_co_ci_u32_e64 v5, null, s5, v3, vcc_lo
 ; ALIGNED-NEXT:    v_add_co_u32 v84, vcc_lo, v0, s4
 ; ALIGNED-NEXT:    v_add_co_ci_u32_e64 v85, null, s5, v1, vcc_lo
 ; ALIGNED-NEXT:    s_clause 0xf
-; ALIGNED-NEXT:    global_load_dwordx4 v[98:101], v[8:9], off offset:240
-; ALIGNED-NEXT:    global_load_dwordx4 v[112:115], v[8:9], off offset:224
-; ALIGNED-NEXT:    global_load_dwordx4 v[80:83], v[8:9], off offset:208
-; ALIGNED-NEXT:    global_load_dwordx4 v[68:71], v[8:9], off offset:192
-; ALIGNED-NEXT:    global_load_dwordx4 v[64:67], v[8:9], off offset:176
-; ALIGNED-NEXT:    global_load_dwordx4 v[52:55], v[8:9], off offset:160
-; ALIGNED-NEXT:    global_load_dwordx4 v[48:51], v[8:9], off offset:144
-; ALIGNED-NEXT:    global_load_dwordx4 v[36:39], v[8:9], off offset:128
-; ALIGNED-NEXT:    global_load_dwordx4 v[32:35], v[8:9], off offset:112
-; ALIGNED-NEXT:    global_load_dwordx4 v[28:31], v[8:9], off offset:96
-; ALIGNED-NEXT:    global_load_dwordx4 v[24:27], v[8:9], off offset:80
-; ALIGNED-NEXT:    global_load_dwordx4 v[20:23], v[8:9], off offset:64
-; ALIGNED-NEXT:    global_load_dwordx4 v[16:19], v[8:9], off offset:48
-; ALIGNED-NEXT:    global_load_dwordx4 v[12:15], v[8:9], off offset:32
-; ALIGNED-NEXT:    global_load_dwordx4 v[4:7], v[8:9], off
-; ALIGNED-NEXT:    global_load_dwordx4 v[8:11], v[8:9], off offset:16
+; ALIGNED-NEXT:    global_load_dwordx4 v[98:101], v[4:5], off offset:240
+; ALIGNED-NEXT:    global_load_dwordx4 v[112:115], v[4:5], off offset:224
+; ALIGNED-NEXT:    global_load_dwordx4 v[80:83], v[4:5], off offset:208
+; ALIGNED-NEXT:    global_load_dwordx4 v[68:71], v[4:5], off offset:192
+; ALIGNED-NEXT:    global_load_dwordx4 v[64:67], v[4:5], off offset:176
+; ALIGNED-NEXT:    global_load_dwordx4 v[52:55], v[4:5], off offset:160
+; ALIGNED-NEXT:    global_load_dwordx4 v[48:51], v[4:5], off offset:144
+; ALIGNED-NEXT:    global_load_dwordx4 v[36:39], v[4:5], off offset:128
+; ALIGNED-NEXT:    global_load_dwordx4 v[32:35], v[4:5], off offset:112
+; ALIGNED-NEXT:    global_load_dwordx4 v[28:31], v[4:5], off offset:96
+; ALIGNED-NEXT:    global_load_dwordx4 v[24:27], v[4:5], off offset:80
+; ALIGNED-NEXT:    global_load_dwordx4 v[20:23], v[4:5], off offset:64
+; ALIGNED-NEXT:    global_load_dwordx4 v[16:19], v[4:5], off offset:48
+; ALIGNED-NEXT:    global_load_dwordx4 v[12:15], v[4:5], off offset:32
+; ALIGNED-NEXT:    global_load_dwordx4 v[8:11], v[4:5], off offset:16
+; ALIGNED-NEXT:    global_load_dwordx4 v[4:7], v[4:5], off
 ; ALIGNED-NEXT:    v_add_co_u32 v96, vcc_lo, v84, 6
 ; ALIGNED-NEXT:    v_add_co_ci_u32_e64 v97, null, 0, v85, vcc_lo
 ; ALIGNED-NEXT:    v_add_co_u32 v86, vcc_lo, v84, 3
@@ -1867,9 +1860,9 @@ define void @memcpy_p0_p4_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(4)
 ; ALIGNED-NEXT:    flat_store_byte v[86:87], v114 offset:152
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(6)
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v114, 24, v28
-; ALIGNED-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
 ; ALIGNED-NEXT:    flat_store_byte v[86:87], v115 offset:150
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v115, 8, v28
+; ALIGNED-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v82, 24, v30
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v83, 8, v30
 ; ALIGNED-NEXT:    flat_store_byte_d16_hi v[96:97], v50 offset:148
@@ -1952,9 +1945,10 @@ define void @memcpy_p0_p4_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(4)
 ; ALIGNED-NEXT:    flat_store_byte v[86:87], v29 offset:97
 ; ALIGNED-NEXT:    flat_store_byte_d16_hi v[96:97], v28 offset:92
 ; ALIGNED-NEXT:    flat_store_byte v[86:87], v28 offset:93
-; ALIGNED-NEXT:    s_waitcnt vmcnt(0)
+; ALIGNED-NEXT:    s_waitcnt vmcnt(1)
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v30, 24, v10
 ; ALIGNED-NEXT:    flat_store_byte v[86:87], v114 offset:96
+; ALIGNED-NEXT:    s_waitcnt vmcnt(0)
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v114, 24, v6
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v98, 24, v25
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v32, 24, v19
@@ -1962,7 +1956,6 @@ define void @memcpy_p0_p4_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(4)
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v31, 8, v10
 ; ALIGNED-NEXT:    flat_store_byte v[86:87], v115 offset:94
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v115, 8, v6
-; ALIGNED-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v99, 8, v25
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v33, 8, v19
 ; ALIGNED-NEXT:    v_lshrrev_b32_e32 v36, 24, v24
@@ -2151,8 +2144,7 @@ define void @memcpy_p0_p4_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(4)
 ; UNROLL3-NEXT:    flat_store_dwordx4 v[16:17], v[8:11]
 ; UNROLL3-NEXT:    s_waitcnt vmcnt(0)
 ; UNROLL3-NEXT:    flat_store_dwordx4 v[16:17], v[12:15] offset:32
-; UNROLL3-NEXT:    v_cmp_gt_u64_e64 s6, 0x7e0, s[4:5]
-; UNROLL3-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
+; UNROLL3-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    s_cbranch_vccnz .LBB2_1
 ; UNROLL3-NEXT:  ; %bb.2: ; %static-memcpy-post-expansion
 ; UNROLL3-NEXT:    s_clause 0x1
@@ -2244,7 +2236,7 @@ define void @memcpy_p5_p5_sz2048(ptr addrspace(5) align 1 %dst, ptr addrspace(5)
 ; CHECK-NEXT:    s_add_u32 s4, s4, 0x100
 ; CHECK-NEXT:    s_addc_u32 s5, s5, 0
 ; CHECK-NEXT:    v_add_nc_u32_e32 v1, 0x100, v1
-; CHECK-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
+; CHECK-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; CHECK-NEXT:    s_waitcnt vmcnt(62)
 ; CHECK-NEXT:    buffer_store_dword v2, v0, s[0:3], 0 offen offset:252
 ; CHECK-NEXT:    buffer_store_dword v3, v0, s[0:3], 0 offen offset:248
@@ -2373,7 +2365,6 @@ define void @memcpy_p5_p5_sz2048(ptr addrspace(5) align 1 %dst, ptr addrspace(5)
 ; CHECK-NEXT:    s_waitcnt vmcnt(0)
 ; CHECK-NEXT:    buffer_store_dword v97, v0, s[0:3], 0 offen
 ; CHECK-NEXT:    v_add_nc_u32_e32 v0, 0x100, v0
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB3_1
 ; CHECK-NEXT:  ; %bb.2: ; %static-memcpy-post-expansion
 ; CHECK-NEXT:    s_setpc_b64 s[30:31]
@@ -2435,8 +2426,7 @@ define void @memcpy_p5_p5_sz2048(ptr addrspace(5) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    buffer_load_ubyte v2, v1, s[0:3], 0 offen
 ; ALIGNED-NEXT:    s_add_u32 s4, s4, 0x100
 ; ALIGNED-NEXT:    s_addc_u32 s5, s5, 0
-; ALIGNED-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
-; ALIGNED-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
+; ALIGNED-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(0)
 ; ALIGNED-NEXT:    buffer_store_dword v2, off, s[0:3], s32 offset:192 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_load_ubyte v2, v1, s[0:3], 0 offen offset:1
@@ -3555,7 +3545,7 @@ define void @memcpy_p5_p5_sz2048(ptr addrspace(5) align 1 %dst, ptr addrspace(5)
 ; UNROLL3-NEXT:    s_add_u32 s4, s4, 48
 ; UNROLL3-NEXT:    s_addc_u32 s5, s5, 0
 ; UNROLL3-NEXT:    v_add_nc_u32_e32 v3, 48, v3
-; UNROLL3-NEXT:    v_cmp_gt_u64_e64 s6, 0x7e0, s[4:5]
+; UNROLL3-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    s_waitcnt vmcnt(11)
 ; UNROLL3-NEXT:    buffer_store_dword v4, v2, s[0:3], 0 offen offset:44
 ; UNROLL3-NEXT:    s_waitcnt vmcnt(10)
@@ -3581,7 +3571,6 @@ define void @memcpy_p5_p5_sz2048(ptr addrspace(5) align 1 %dst, ptr addrspace(5)
 ; UNROLL3-NEXT:    s_waitcnt vmcnt(0)
 ; UNROLL3-NEXT:    buffer_store_dword v15, v2, s[0:3], 0 offen
 ; UNROLL3-NEXT:    v_add_nc_u32_e32 v2, 48, v2
-; UNROLL3-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; UNROLL3-NEXT:    s_cbranch_vccnz .LBB3_1
 ; UNROLL3-NEXT:  ; %bb.2: ; %static-memcpy-post-expansion
 ; UNROLL3-NEXT:    s_clause 0x3
@@ -3620,7 +3609,7 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; CHECK-LABEL: memcpy_p0_p5_sz2048:
 ; CHECK:       ; %bb.0: ; %entry
 ; CHECK-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; CHECK-NEXT:    s_mov_b64 s[4:5], 0
+; CHECK-NEXT:    s_mov_b64 s[6:7], 0
 ; CHECK-NEXT:  .LBB4_1: ; %static-memcpy-expansion-main-body
 ; CHECK-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; CHECK-NEXT:    s_clause 0x3e
@@ -3640,38 +3629,38 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; CHECK-NEXT:    buffer_load_dword v13, v2, s[0:3], 0 offen offset:72
 ; CHECK-NEXT:    buffer_load_dword v12, v2, s[0:3], 0 offen offset:68
 ; CHECK-NEXT:    buffer_load_dword v11, v2, s[0:3], 0 offen offset:64
-; CHECK-NEXT:    buffer_load_dword v22, v2, s[0:3], 0 offen offset:236
-; CHECK-NEXT:    buffer_load_dword v26, v2, s[0:3], 0 offen offset:252
-; CHECK-NEXT:    buffer_load_dword v25, v2, s[0:3], 0 offen offset:248
-; CHECK-NEXT:    buffer_load_dword v24, v2, s[0:3], 0 offen offset:244
-; CHECK-NEXT:    buffer_load_dword v23, v2, s[0:3], 0 offen offset:240
-; CHECK-NEXT:    buffer_load_dword v21, v2, s[0:3], 0 offen offset:232
-; CHECK-NEXT:    buffer_load_dword v20, v2, s[0:3], 0 offen offset:228
-; CHECK-NEXT:    buffer_load_dword v19, v2, s[0:3], 0 offen offset:224
-; CHECK-NEXT:    buffer_load_dword v30, v2, s[0:3], 0 offen offset:220
-; CHECK-NEXT:    buffer_load_dword v29, v2, s[0:3], 0 offen offset:216
-; CHECK-NEXT:    buffer_load_dword v28, v2, s[0:3], 0 offen offset:212
-; CHECK-NEXT:    buffer_load_dword v27, v2, s[0:3], 0 offen offset:208
-; CHECK-NEXT:    buffer_load_dword v34, v2, s[0:3], 0 offen offset:204
-; CHECK-NEXT:    buffer_load_dword v33, v2, s[0:3], 0 offen offset:200
-; CHECK-NEXT:    buffer_load_dword v32, v2, s[0:3], 0 offen offset:196
-; CHECK-NEXT:    buffer_load_dword v31, v2, s[0:3], 0 offen offset:192
-; CHECK-NEXT:    buffer_load_dword v38, v2, s[0:3], 0 offen offset:108
-; CHECK-NEXT:    buffer_load_dword v51, v2, s[0:3], 0 offen offset:124
-; CHECK-NEXT:    buffer_load_dword v50, v2, s[0:3], 0 offen offset:120
-; CHECK-NEXT:    buffer_load_dword v49, v2, s[0:3], 0 offen offset:116
-; CHECK-NEXT:    buffer_load_dword v48, v2, s[0:3], 0 offen offset:112
-; CHECK-NEXT:    buffer_load_dword v37, v2, s[0:3], 0 offen offset:104
-; CHECK-NEXT:    buffer_load_dword v36, v2, s[0:3], 0 offen offset:100
-; CHECK-NEXT:    buffer_load_dword v35, v2, s[0:3], 0 offen offset:96
-; CHECK-NEXT:    buffer_load_dword v55, v2, s[0:3], 0 offen offset:188
-; CHECK-NEXT:    buffer_load_dword v54, v2, s[0:3], 0 offen offset:184
-; CHECK-NEXT:    buffer_load_dword v53, v2, s[0:3], 0 offen offset:180
-; CHECK-NEXT:    buffer_load_dword v52, v2, s[0:3], 0 offen offset:176
-; CHECK-NEXT:    buffer_load_dword v67, v2, s[0:3], 0 offen offset:172
-; CHECK-NEXT:    buffer_load_dword v66, v2, s[0:3], 0 offen offset:168
-; CHECK-NEXT:    buffer_load_dword v65, v2, s[0:3], 0 offen offset:164
-; CHECK-NEXT:    buffer_load_dword v64, v2, s[0:3], 0 offen offset:160
+; CHECK-NEXT:    buffer_load_dword v22, v2, s[0:3], 0 offen offset:108
+; CHECK-NEXT:    buffer_load_dword v26, v2, s[0:3], 0 offen offset:124
+; CHECK-NEXT:    buffer_load_dword v25, v2, s[0:3], 0 offen offset:120
+; CHECK-NEXT:    buffer_load_dword v24, v2, s[0:3], 0 offen offset:116
+; CHECK-NEXT:    buffer_load_dword v23, v2, s[0:3], 0 offen offset:112
+; CHECK-NEXT:    buffer_load_dword v21, v2, s[0:3], 0 offen offset:104
+; CHECK-NEXT:    buffer_load_dword v20, v2, s[0:3], 0 offen offset:100
+; CHECK-NEXT:    buffer_load_dword v19, v2, s[0:3], 0 offen offset:96
+; CHECK-NEXT:    buffer_load_dword v30, v2, s[0:3], 0 offen offset:236
+; CHECK-NEXT:    buffer_load_dword v34, v2, s[0:3], 0 offen offset:252
+; CHECK-NEXT:    buffer_load_dword v33, v2, s[0:3], 0 offen offset:248
+; CHECK-NEXT:    buffer_load_dword v32, v2, s[0:3], 0 offen offset:244
+; CHECK-NEXT:    buffer_load_dword v31, v2, s[0:3], 0 offen offset:240
+; CHECK-NEXT:    buffer_load_dword v29, v2, s[0:3], 0 offen offset:232
+; CHECK-NEXT:    buffer_load_dword v28, v2, s[0:3], 0 offen offset:228
+; CHECK-NEXT:    buffer_load_dword v27, v2, s[0:3], 0 offen offset:224
+; CHECK-NEXT:    buffer_load_dword v38, v2, s[0:3], 0 offen offset:204
+; CHECK-NEXT:    buffer_load_dword v51, v2, s[0:3], 0 offen offset:220
+; CHECK-NEXT:    buffer_load_dword v50, v2, s[0:3], 0 offen offset:216
+; CHECK-NEXT:    buffer_load_dword v49, v2, s[0:3], 0 offen offset:212
+; CHECK-NEXT:    buffer_load_dword v48, v2, s[0:3], 0 offen offset:208
+; CHECK-NEXT:    buffer_load_dword v37, v2, s[0:3], 0 offen offset:200
+; CHECK-NEXT:    buffer_load_dword v36, v2, s[0:3], 0 offen offset:196
+; CHECK-NEXT:    buffer_load_dword v35, v2, s[0:3], 0 offen offset:192
+; CHECK-NEXT:    buffer_load_dword v55, v2, s[0:3], 0 offen offset:172
+; CHECK-NEXT:    buffer_load_dword v67, v2, s[0:3], 0 offen offset:188
+; CHECK-NEXT:    buffer_load_dword v66, v2, s[0:3], 0 offen offset:184
+; CHECK-NEXT:    buffer_load_dword v65, v2, s[0:3], 0 offen offset:180
+; CHECK-NEXT:    buffer_load_dword v64, v2, s[0:3], 0 offen offset:176
+; CHECK-NEXT:    buffer_load_dword v54, v2, s[0:3], 0 offen offset:168
+; CHECK-NEXT:    buffer_load_dword v53, v2, s[0:3], 0 offen offset:164
+; CHECK-NEXT:    buffer_load_dword v52, v2, s[0:3], 0 offen offset:160
 ; CHECK-NEXT:    buffer_load_dword v71, v2, s[0:3], 0 offen offset:156
 ; CHECK-NEXT:    buffer_load_dword v70, v2, s[0:3], 0 offen offset:152
 ; CHECK-NEXT:    buffer_load_dword v69, v2, s[0:3], 0 offen offset:148
@@ -3688,32 +3677,32 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; CHECK-NEXT:    buffer_load_dword v98, v2, s[0:3], 0 offen offset:24
 ; CHECK-NEXT:    buffer_load_dword v99, v2, s[0:3], 0 offen offset:28
 ; CHECK-NEXT:    buffer_load_dword v87, v2, s[0:3], 0 offen offset:12
-; CHECK-NEXT:    v_add_co_u32 v100, vcc_lo, v0, s4
-; CHECK-NEXT:    s_add_u32 s4, s4, 0x100
-; CHECK-NEXT:    v_add_co_ci_u32_e64 v101, null, s5, v1, vcc_lo
-; CHECK-NEXT:    s_addc_u32 s5, s5, 0
-; CHECK-NEXT:    v_add_co_u32 v102, vcc_lo, v100, 48
-; CHECK-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
+; CHECK-NEXT:    v_add_co_u32 v100, vcc_lo, v0, s6
+; CHECK-NEXT:    v_add_co_ci_u32_e64 v101, null, s7, v1, vcc_lo
+; CHECK-NEXT:    s_add_u32 s6, s6, 0x100
+; CHECK-NEXT:    s_addc_u32 s7, s7, 0
+; CHECK-NEXT:    v_add_co_u32 v102, s4, v100, 48
+; CHECK-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[6:7]
 ; CHECK-NEXT:    v_add_nc_u32_e32 v2, 0x100, v2
-; CHECK-NEXT:    v_add_co_ci_u32_e64 v103, null, 0, v101, vcc_lo
-; CHECK-NEXT:    s_waitcnt vmcnt(43)
-; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[23:26] offset:192
-; CHECK-NEXT:    s_waitcnt vmcnt(40)
-; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[19:22] offset:176
-; CHECK-NEXT:    s_waitcnt vmcnt(36)
-; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[27:30] offset:160
+; CHECK-NEXT:    v_add_co_ci_u32_e64 v103, null, 0, v101, s4
+; CHECK-NEXT:    s_waitcnt vmcnt(35)
+; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[31:34] offset:192
 ; CHECK-NEXT:    s_waitcnt vmcnt(32)
-; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[31:34] offset:144
-; CHECK-NEXT:    s_waitcnt vmcnt(20)
-; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[52:55] offset:128
+; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[27:30] offset:176
+; CHECK-NEXT:    s_waitcnt vmcnt(27)
+; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[48:51] offset:160
+; CHECK-NEXT:    s_waitcnt vmcnt(24)
+; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[35:38] offset:144
+; CHECK-NEXT:    s_waitcnt vmcnt(19)
+; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[64:67] offset:128
 ; CHECK-NEXT:    s_waitcnt vmcnt(16)
-; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[64:67] offset:112
+; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[52:55] offset:112
 ; CHECK-NEXT:    s_waitcnt vmcnt(12)
 ; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[68:71] offset:96
 ; CHECK-NEXT:    s_waitcnt vmcnt(8)
 ; CHECK-NEXT:    flat_store_dwordx4 v[100:101], v[80:83] offset:128
-; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[48:51] offset:64
-; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[35:38] offset:48
+; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[23:26] offset:64
+; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[19:22] offset:48
 ; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[15:18] offset:32
 ; CHECK-NEXT:    flat_store_dwordx4 v[100:101], v[11:14] offset:64
 ; CHECK-NEXT:    flat_store_dwordx4 v[102:103], v[7:10]
@@ -3722,7 +3711,6 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; CHECK-NEXT:    flat_store_dwordx4 v[100:101], v[96:99] offset:16
 ; CHECK-NEXT:    s_waitcnt vmcnt(0)
 ; CHECK-NEXT:    flat_store_dwordx4 v[100:101], v[84:87]
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB4_1
 ; CHECK-NEXT:  ; %bb.2: ; %static-memcpy-post-expansion
 ; CHECK-NEXT:    s_waitcnt lgkmcnt(0)
@@ -3801,11 +3789,11 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    buffer_load_ubyte v14, v2, s[0:3], 0 offen offset:32
 ; ALIGNED-NEXT:    buffer_load_ubyte v15, v2, s[0:3], 0 offen offset:33
 ; ALIGNED-NEXT:    buffer_load_ubyte v19, v2, s[0:3], 0 offen offset:34
-; ALIGNED-NEXT:    buffer_load_ubyte v21, v2, s[0:3], 0 offen offset:35
+; ALIGNED-NEXT:    buffer_load_ubyte v20, v2, s[0:3], 0 offen offset:35
 ; ALIGNED-NEXT:    buffer_load_ubyte v13, v2, s[0:3], 0 offen offset:36
-; ALIGNED-NEXT:    buffer_load_ubyte v17, v2, s[0:3], 0 offen offset:37
-; ALIGNED-NEXT:    buffer_load_ubyte v18, v2, s[0:3], 0 offen offset:38
-; ALIGNED-NEXT:    buffer_load_ubyte v20, v2, s[0:3], 0 offen offset:39
+; ALIGNED-NEXT:    buffer_load_ubyte v16, v2, s[0:3], 0 offen offset:37
+; ALIGNED-NEXT:    buffer_load_ubyte v17, v2, s[0:3], 0 offen offset:38
+; ALIGNED-NEXT:    buffer_load_ubyte v18, v2, s[0:3], 0 offen offset:39
 ; ALIGNED-NEXT:    buffer_load_ubyte v22, v2, s[0:3], 0 offen offset:40
 ; ALIGNED-NEXT:    buffer_load_ubyte v27, v2, s[0:3], 0 offen offset:41
 ; ALIGNED-NEXT:    buffer_load_ubyte v28, v2, s[0:3], 0 offen offset:42
@@ -3822,10 +3810,10 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    buffer_load_ubyte v37, v2, s[0:3], 0 offen offset:53
 ; ALIGNED-NEXT:    buffer_load_ubyte v35, v2, s[0:3], 0 offen offset:54
 ; ALIGNED-NEXT:    buffer_load_ubyte v36, v2, s[0:3], 0 offen offset:55
-; ALIGNED-NEXT:    buffer_load_ubyte v16, v2, s[0:3], 0 offen offset:56
+; ALIGNED-NEXT:    buffer_load_ubyte v38, v2, s[0:3], 0 offen offset:56
 ; ALIGNED-NEXT:    buffer_load_ubyte v50, v2, s[0:3], 0 offen offset:57
 ; ALIGNED-NEXT:    buffer_load_ubyte v51, v2, s[0:3], 0 offen offset:58
-; ALIGNED-NEXT:    buffer_load_ubyte v38, v2, s[0:3], 0 offen offset:60
+; ALIGNED-NEXT:    buffer_load_ubyte v21, v2, s[0:3], 0 offen offset:60
 ; ALIGNED-NEXT:    buffer_load_ubyte v49, v2, s[0:3], 0 offen offset:61
 ; ALIGNED-NEXT:    buffer_load_ubyte v39, v2, s[0:3], 0 offen offset:62
 ; ALIGNED-NEXT:    buffer_load_ubyte v48, v2, s[0:3], 0 offen offset:63
@@ -3878,11 +3866,11 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(45)
 ; ALIGNED-NEXT:    v_lshl_or_b32 v7, v15, 8, v14
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(43)
-; ALIGNED-NEXT:    v_lshl_or_b32 v8, v21, 8, v19
+; ALIGNED-NEXT:    v_lshl_or_b32 v8, v20, 8, v19
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(41)
-; ALIGNED-NEXT:    v_lshl_or_b32 v9, v17, 8, v13
+; ALIGNED-NEXT:    v_lshl_or_b32 v9, v16, 8, v13
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(39)
-; ALIGNED-NEXT:    v_lshl_or_b32 v10, v20, 8, v18
+; ALIGNED-NEXT:    v_lshl_or_b32 v10, v18, 8, v17
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(37)
 ; ALIGNED-NEXT:    v_lshl_or_b32 v11, v27, 8, v22
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v1, 16, v0
@@ -3913,7 +3901,7 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    v_lshl_or_b32 v4, v36, 8, v35
 ; ALIGNED-NEXT:    buffer_store_dword v5, off, s[0:3], s32 offset:580 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(18)
-; ALIGNED-NEXT:    v_lshl_or_b32 v5, v49, 8, v38
+; ALIGNED-NEXT:    v_lshl_or_b32 v5, v49, 8, v21
 ; ALIGNED-NEXT:    buffer_store_dword v6, off, s[0:3], s32 offset:592 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(16)
 ; ALIGNED-NEXT:    v_lshl_or_b32 v6, v48, 8, v39
@@ -3923,7 +3911,7 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    buffer_load_ubyte v4, v2, s[0:3], 0 offen offset:80
 ; ALIGNED-NEXT:    v_lshl_or_b32 v3, v6, 16, v5
 ; ALIGNED-NEXT:    buffer_store_dword v7, off, s[0:3], s32 offset:612 ; 4-byte Folded Spill
-; ALIGNED-NEXT:    v_lshl_or_b32 v7, v50, 8, v16
+; ALIGNED-NEXT:    v_lshl_or_b32 v7, v50, 8, v38
 ; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:652 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(13)
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v52, 8, v51
@@ -3948,12 +3936,12 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(13)
 ; ALIGNED-NEXT:    v_lshl_or_b32 v1, v68, 8, v67
 ; ALIGNED-NEXT:    buffer_store_dword v19, off, s[0:3], s32 offset:532 ; 4-byte Folded Spill
-; ALIGNED-NEXT:    buffer_store_dword v21, off, s[0:3], s32 offset:536 ; 4-byte Folded Spill
-; ALIGNED-NEXT:    buffer_store_dword v17, off, s[0:3], s32 offset:500 ; 4-byte Folded Spill
-; ALIGNED-NEXT:    buffer_store_dword v18, off, s[0:3], s32 offset:516 ; 4-byte Folded Spill
+; ALIGNED-NEXT:    buffer_store_dword v20, off, s[0:3], s32 offset:536 ; 4-byte Folded Spill
+; ALIGNED-NEXT:    buffer_store_dword v16, off, s[0:3], s32 offset:500 ; 4-byte Folded Spill
+; ALIGNED-NEXT:    buffer_store_dword v17, off, s[0:3], s32 offset:516 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v1, 16, v0
 ; ALIGNED-NEXT:    buffer_load_ubyte v1, v2, s[0:3], 0 offen offset:79
-; ALIGNED-NEXT:    buffer_store_dword v20, off, s[0:3], s32 offset:520 ; 4-byte Folded Spill
+; ALIGNED-NEXT:    buffer_store_dword v18, off, s[0:3], s32 offset:520 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v22, off, s[0:3], s32 offset:548 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v27, off, s[0:3], s32 offset:560 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:704 ; 4-byte Folded Spill
@@ -3973,11 +3961,11 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    buffer_store_dword v37, off, s[0:3], s32 offset:616 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v35, off, s[0:3], s32 offset:604 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v36, off, s[0:3], s32 offset:608 ; 4-byte Folded Spill
-; ALIGNED-NEXT:    buffer_store_dword v38, off, s[0:3], s32 offset:624 ; 4-byte Folded Spill
+; ALIGNED-NEXT:    buffer_store_dword v21, off, s[0:3], s32 offset:624 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v49, off, s[0:3], s32 offset:636 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v39, off, s[0:3], s32 offset:628 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v48, off, s[0:3], s32 offset:632 ; 4-byte Folded Spill
-; ALIGNED-NEXT:    buffer_store_dword v16, off, s[0:3], s32 offset:620 ; 4-byte Folded Spill
+; ALIGNED-NEXT:    buffer_store_dword v38, off, s[0:3], s32 offset:620 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v50, off, s[0:3], s32 offset:640 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v51, off, s[0:3], s32 offset:648 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v52, off, s[0:3], s32 offset:644 ; 4-byte Folded Spill
@@ -4231,13 +4219,13 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    buffer_load_ubyte v10, v2, s[0:3], 0 offen offset:144
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v0, 8, v5
 ; ALIGNED-NEXT:    s_clause 0x3e
-; ALIGNED-NEXT:    buffer_load_ubyte v109, v2, s[0:3], 0 offen offset:165
-; ALIGNED-NEXT:    buffer_load_ubyte v110, v2, s[0:3], 0 offen offset:166
+; ALIGNED-NEXT:    buffer_load_ubyte v111, v2, s[0:3], 0 offen offset:165
+; ALIGNED-NEXT:    buffer_load_ubyte v109, v2, s[0:3], 0 offen offset:166
 ; ALIGNED-NEXT:    buffer_load_ubyte v108, v2, s[0:3], 0 offen offset:167
 ; ALIGNED-NEXT:    buffer_load_ubyte v107, v2, s[0:3], 0 offen offset:168
 ; ALIGNED-NEXT:    buffer_load_ubyte v105, v2, s[0:3], 0 offen offset:169
-; ALIGNED-NEXT:    buffer_load_ubyte v94, v2, s[0:3], 0 offen offset:170
-; ALIGNED-NEXT:    buffer_load_ubyte v95, v2, s[0:3], 0 offen offset:171
+; ALIGNED-NEXT:    buffer_load_ubyte v95, v2, s[0:3], 0 offen offset:170
+; ALIGNED-NEXT:    buffer_load_ubyte v104, v2, s[0:3], 0 offen offset:171
 ; ALIGNED-NEXT:    buffer_load_ubyte v93, v2, s[0:3], 0 offen offset:172
 ; ALIGNED-NEXT:    buffer_load_ubyte v89, v2, s[0:3], 0 offen offset:173
 ; ALIGNED-NEXT:    buffer_load_ubyte v88, v2, s[0:3], 0 offen offset:174
@@ -4319,8 +4307,8 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    buffer_load_ubyte v125, v2, s[0:3], 0 offen offset:4
 ; ALIGNED-NEXT:    buffer_load_ubyte v123, v2, s[0:3], 0 offen offset:5
 ; ALIGNED-NEXT:    buffer_load_ubyte v122, v2, s[0:3], 0 offen offset:6
-; ALIGNED-NEXT:    buffer_load_ubyte v111, v2, s[0:3], 0 offen offset:7
-; ALIGNED-NEXT:    buffer_load_ubyte v104, v2, s[0:3], 0 offen offset:12
+; ALIGNED-NEXT:    buffer_load_ubyte v110, v2, s[0:3], 0 offen offset:7
+; ALIGNED-NEXT:    buffer_load_ubyte v94, v2, s[0:3], 0 offen offset:12
 ; ALIGNED-NEXT:    buffer_load_ubyte v91, v2, s[0:3], 0 offen offset:13
 ; ALIGNED-NEXT:    buffer_load_ubyte v92, v2, s[0:3], 0 offen offset:14
 ; ALIGNED-NEXT:    buffer_load_ubyte v90, v2, s[0:3], 0 offen offset:15
@@ -4455,12 +4443,12 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    buffer_load_ubyte v9, v2, s[0:3], 0 offen offset:252
 ; ALIGNED-NEXT:    v_lshl_or_b32 v4, v52, 8, v64
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v1, 16, v0
-; ALIGNED-NEXT:    v_lshl_or_b32 v1, v108, 8, v110
+; ALIGNED-NEXT:    v_lshl_or_b32 v1, v108, 8, v109
 ; ALIGNED-NEXT:    v_lshl_or_b32 v117, v4, 16, v3
 ; ALIGNED-NEXT:    v_lshl_or_b32 v3, v65, 8, v67
 ; ALIGNED-NEXT:    v_lshl_or_b32 v4, v54, 8, v55
 ; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:1164 ; 4-byte Folded Spill
-; ALIGNED-NEXT:    v_lshl_or_b32 v0, v109, 8, v5
+; ALIGNED-NEXT:    v_lshl_or_b32 v0, v111, 8, v5
 ; ALIGNED-NEXT:    v_lshl_or_b32 v103, v4, 16, v3
 ; ALIGNED-NEXT:    v_lshl_or_b32 v3, v49, 8, v66
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v1, 16, v0
@@ -4472,10 +4460,10 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    v_lshl_or_b32 v3, v71, 8, v81
 ; ALIGNED-NEXT:    v_lshl_or_b32 v4, v70, 8, v80
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v1, 16, v0
-; ALIGNED-NEXT:    v_lshl_or_b32 v1, v95, 8, v94
+; ALIGNED-NEXT:    v_lshl_or_b32 v1, v104, 8, v95
 ; ALIGNED-NEXT:    v_lshl_or_b32 v86, v4, 16, v3
 ; ALIGNED-NEXT:    v_lshl_or_b32 v3, v36, 8, v38
-; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:1192 ; 4-byte Folded Spill
+; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:1188 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v105, 8, v107
 ; ALIGNED-NEXT:    v_lshl_or_b32 v4, v34, 8, v35
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v1, 16, v0
@@ -4537,10 +4525,10 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    v_lshl_or_b32 v56, v124, 8, v126
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v57, 16, v56
 ; ALIGNED-NEXT:    v_lshl_or_b32 v56, v123, 8, v125
-; ALIGNED-NEXT:    v_lshl_or_b32 v57, v111, 8, v122
+; ALIGNED-NEXT:    v_lshl_or_b32 v57, v110, 8, v122
 ; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:1172 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v57, 16, v56
-; ALIGNED-NEXT:    v_lshl_or_b32 v56, v91, 8, v104
+; ALIGNED-NEXT:    v_lshl_or_b32 v56, v91, 8, v94
 ; ALIGNED-NEXT:    v_lshl_or_b32 v57, v90, 8, v92
 ; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:1176 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    v_lshl_or_b32 v0, v57, 16, v56
@@ -4553,7 +4541,7 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    buffer_load_ubyte v60, v2, s[0:3], 0 offen offset:16
 ; ALIGNED-NEXT:    buffer_load_ubyte v57, v2, s[0:3], 0 offen offset:17
 ; ALIGNED-NEXT:    v_add_nc_u32_e32 v2, 0x100, v2
-; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:1188 ; 4-byte Folded Spill
+; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:1192 ; 4-byte Folded Spill
 ; ALIGNED-NEXT:    buffer_store_dword v6, off, s[0:3], s32 offset:232
 ; ALIGNED-NEXT:    buffer_store_dword v5, off, s[0:3], s32 offset:236
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:228
@@ -4633,8 +4621,7 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    buffer_store_dword v61, off, s[0:3], s32 offset:220
 ; ALIGNED-NEXT:    buffer_store_dword v106, off, s[0:3], s32 offset:212
 ; ALIGNED-NEXT:    buffer_load_dword v0, off, s[0:3], s32 offset:1220 ; 4-byte Folded Reload
-; ALIGNED-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
-; ALIGNED-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
+; ALIGNED-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(0)
 ; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:208
 ; ALIGNED-NEXT:    buffer_load_dword v0, off, s[0:3], s32 offset:1216 ; 4-byte Folded Reload
@@ -4684,7 +4671,7 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v75 offset:173
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(0)
 ; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:312
-; ALIGNED-NEXT:    buffer_load_dword v0, off, s[0:3], s32 offset:1192 ; 4-byte Folded Reload
+; ALIGNED-NEXT:    buffer_load_dword v0, off, s[0:3], s32 offset:1188 ; 4-byte Folded Reload
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(0)
 ; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:316
 ; ALIGNED-NEXT:    buffer_load_dword v0, off, s[0:3], s32 offset:1180 ; 4-byte Folded Reload
@@ -4694,8 +4681,8 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(0)
 ; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:304
 ; ALIGNED-NEXT:    buffer_load_dword v0, off, s[0:3], s32 offset:1160 ; 4-byte Folded Reload
-; ALIGNED-NEXT:    flat_store_byte v[5:6], v94 offset:167
-; ALIGNED-NEXT:    flat_store_byte v[5:6], v95 offset:168
+; ALIGNED-NEXT:    flat_store_byte v[5:6], v95 offset:167
+; ALIGNED-NEXT:    flat_store_byte v[5:6], v104 offset:168
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v105 offset:166
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v79 offset:172
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v89 offset:170
@@ -4711,8 +4698,8 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(0)
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v0 offset:158
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v108 offset:164
-; ALIGNED-NEXT:    flat_store_byte v[5:6], v109 offset:162
-; ALIGNED-NEXT:    flat_store_byte v[5:6], v110 offset:163
+; ALIGNED-NEXT:    flat_store_byte v[5:6], v111 offset:162
+; ALIGNED-NEXT:    flat_store_byte v[5:6], v109 offset:163
 ; ALIGNED-NEXT:    buffer_load_dword v0, off, s[0:3], s32 offset:1168 ; 4-byte Folded Reload
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(0)
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v0 offset:161
@@ -5251,7 +5238,7 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(0)
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v0 offset:17
 ; ALIGNED-NEXT:    flat_store_byte v[3:4], v60 offset:16
-; ALIGNED-NEXT:    buffer_load_dword v0, off, s[0:3], s32 offset:1188 ; 4-byte Folded Reload
+; ALIGNED-NEXT:    buffer_load_dword v0, off, s[0:3], s32 offset:1192 ; 4-byte Folded Reload
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(0)
 ; ALIGNED-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:408
 ; ALIGNED-NEXT:    buffer_load_dword v0, off, s[0:3], s32 offset:1184 ; 4-byte Folded Reload
@@ -5269,12 +5256,12 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v76 offset:6
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v90 offset:12
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v92 offset:11
-; ALIGNED-NEXT:    flat_store_byte v[5:6], v104 offset:9
+; ALIGNED-NEXT:    flat_store_byte v[5:6], v94 offset:9
 ; ALIGNED-NEXT:    flat_store_byte v[3:4], v78 offset:8
 ; ALIGNED-NEXT:    flat_store_byte v[3:4], v121 offset:2
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v120
 ; ALIGNED-NEXT:    flat_store_byte v[3:4], v124 offset:1
-; ALIGNED-NEXT:    flat_store_byte v[5:6], v111 offset:4
+; ALIGNED-NEXT:    flat_store_byte v[5:6], v110 offset:4
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v123 offset:2
 ; ALIGNED-NEXT:    flat_store_byte v[5:6], v122 offset:3
 ; ALIGNED-NEXT:    flat_store_byte v[3:4], v125 offset:4
@@ -5360,13 +5347,12 @@ define void @memcpy_p0_p5_sz2048(ptr addrspace(0) align 1 %dst, ptr addrspace(5)
 ; UNROLL3-NEXT:    v_add_co_ci_u32_e64 v17, null, s5, v1, vcc_lo
 ; UNROLL3-NEXT:    s_addc_u32 s5, s5, 0
 ; UNROLL3-NEXT:    v_add_nc_u32_e32 v3, 48, v3
-; UNROLL3-NEXT:    v_cmp_gt_u64_e64 s6, 0x7e0, s[4:5]
+; UNROLL3-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    s_waitcnt vmcnt(4)
 ; UNROLL3-NEXT:    flat_store_dwordx4 v[16:17], v[8:11] offset:16
 ; UNROLL3-NEXT:    flat_store_dwordx4 v[16:17], v[4:7]
 ; UNROLL3-NEXT:    s_waitcnt vmcnt(0)
 ; UNROLL3-NEXT:    flat_store_dwordx4 v[16:17], v[12:15] offset:32
-; UNROLL3-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; UNROLL3-NEXT:    s_cbranch_vccnz .LBB4_1
 ; UNROLL3-NEXT:  ; %bb.2: ; %static-memcpy-post-expansion
 ; UNROLL3-NEXT:    s_inst_prefetch 0x2
@@ -15977,18 +15963,18 @@ define void @memset_p0_sz2048(ptr addrspace(0) %dst) #1 {
 ; CHECK-NEXT:    v_mov_b32_e32 v3, s5
 ; CHECK-NEXT:    v_mov_b32_e32 v4, s6
 ; CHECK-NEXT:    v_mov_b32_e32 v5, s7
-; CHECK-NEXT:    s_mov_b64 s[4:5], 0
+; CHECK-NEXT:    s_mov_b64 s[6:7], 0
 ; CHECK-NEXT:    s_inst_prefetch 0x1
 ; CHECK-NEXT:    .p2align 6
 ; CHECK-NEXT:  .LBB10_1: ; %static-memset-expansion-main-body
 ; CHECK-NEXT:    ; =>This Inner Loop Header: Depth=1
-; CHECK-NEXT:    v_add_co_u32 v6, vcc_lo, v0, s4
-; CHECK-NEXT:    s_add_u32 s4, s4, 0x100
-; CHECK-NEXT:    v_add_co_ci_u32_e64 v7, null, s5, v1, vcc_lo
-; CHECK-NEXT:    s_addc_u32 s5, s5, 0
-; CHECK-NEXT:    v_add_co_u32 v8, vcc_lo, v6, 48
-; CHECK-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
-; CHECK-NEXT:    v_add_co_ci_u32_e64 v9, null, 0, v7, vcc_lo
+; CHECK-NEXT:    v_add_co_u32 v6, vcc_lo, v0, s6
+; CHECK-NEXT:    s_add_u32 s6, s6, 0x100
+; CHECK-NEXT:    v_add_co_ci_u32_e64 v7, null, s7, v1, vcc_lo
+; CHECK-NEXT:    s_addc_u32 s7, s7, 0
+; CHECK-NEXT:    v_add_co_u32 v8, s4, v6, 48
+; CHECK-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[6:7]
+; CHECK-NEXT:    v_add_co_ci_u32_e64 v9, null, 0, v7, s4
 ; CHECK-NEXT:    flat_store_dwordx4 v[6:7], v[2:5] offset:128
 ; CHECK-NEXT:    flat_store_dwordx4 v[6:7], v[2:5] offset:64
 ; CHECK-NEXT:    flat_store_dwordx4 v[6:7], v[2:5] offset:32
@@ -16005,7 +15991,6 @@ define void @memset_p0_sz2048(ptr addrspace(0) %dst) #1 {
 ; CHECK-NEXT:    flat_store_dwordx4 v[8:9], v[2:5] offset:48
 ; CHECK-NEXT:    flat_store_dwordx4 v[8:9], v[2:5] offset:32
 ; CHECK-NEXT:    flat_store_dwordx4 v[8:9], v[2:5]
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB10_1
 ; CHECK-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; CHECK-NEXT:    s_inst_prefetch 0x2
@@ -16017,18 +16002,17 @@ define void @memset_p0_sz2048(ptr addrspace(0) %dst) #1 {
 ; ALIGNED-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
 ; ALIGNED-NEXT:    v_mov_b32_e32 v4, 0x41414141
 ; ALIGNED-NEXT:    v_mov_b32_e32 v5, 0x41
-; ALIGNED-NEXT:    s_mov_b64 s[4:5], 0
+; ALIGNED-NEXT:    s_mov_b64 s[6:7], 0
 ; ALIGNED-NEXT:  .LBB10_1: ; %static-memset-expansion-main-body
 ; ALIGNED-NEXT:    ; =>This Inner Loop Header: Depth=1
-; ALIGNED-NEXT:    v_add_co_u32 v2, vcc_lo, v0, s4
-; ALIGNED-NEXT:    v_add_co_ci_u32_e64 v3, null, s5, v1, vcc_lo
-; ALIGNED-NEXT:    s_add_u32 s4, s4, 0x100
-; ALIGNED-NEXT:    s_addc_u32 s5, s5, 0
+; ALIGNED-NEXT:    v_add_co_u32 v2, vcc_lo, v0, s6
+; ALIGNED-NEXT:    v_add_co_ci_u32_e64 v3, null, s7, v1, vcc_lo
+; ALIGNED-NEXT:    s_add_u32 s6, s6, 0x100
+; ALIGNED-NEXT:    s_addc_u32 s7, s7, 0
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:204
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:200
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:196
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:192
-; ALIGNED-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
 ; ALIGNED-NEXT:    flat_store_byte v[2:3], v5 offset:128
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:12
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:8
@@ -16054,9 +16038,9 @@ define void @memset_p0_sz2048(ptr addrspace(0) %dst) #1 {
 ; ALIGNED-NEXT:    flat_store_byte v[2:3], v5 offset:2
 ; ALIGNED-NEXT:    flat_store_byte v[2:3], v5 offset:1
 ; ALIGNED-NEXT:    flat_store_byte v[2:3], v5
-; ALIGNED-NEXT:    v_add_co_u32 v2, vcc_lo, v2, 3
-; ALIGNED-NEXT:    v_add_co_ci_u32_e64 v3, null, 0, v3, vcc_lo
-; ALIGNED-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
+; ALIGNED-NEXT:    v_add_co_u32 v2, s4, v2, 3
+; ALIGNED-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[6:7]
+; ALIGNED-NEXT:    v_add_co_ci_u32_e64 v3, null, 0, v3, s4
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:152
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:156
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:148
@@ -16365,7 +16349,6 @@ define void @memset_p0_sz2048(ptr addrspace(0) %dst) #1 {
 ; UNROLL3-NEXT:    v_mov_b32_e32 v4, s6
 ; UNROLL3-NEXT:    v_mov_b32_e32 v5, s7
 ; UNROLL3-NEXT:    s_mov_b64 s[4:5], 0
-; UNROLL3-NEXT:    .p2align 6
 ; UNROLL3-NEXT:  .LBB10_1: ; %static-memset-expansion-main-body
 ; UNROLL3-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; UNROLL3-NEXT:    v_add_co_u32 v6, vcc_lo, v0, s4
@@ -16374,9 +16357,8 @@ define void @memset_p0_sz2048(ptr addrspace(0) %dst) #1 {
 ; UNROLL3-NEXT:    s_addc_u32 s5, s5, 0
 ; UNROLL3-NEXT:    flat_store_dwordx4 v[6:7], v[2:5] offset:16
 ; UNROLL3-NEXT:    flat_store_dwordx4 v[6:7], v[2:5]
-; UNROLL3-NEXT:    v_cmp_gt_u64_e64 s6, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    flat_store_dwordx4 v[6:7], v[2:5] offset:32
-; UNROLL3-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
+; UNROLL3-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    s_cbranch_vccnz .LBB10_1
 ; UNROLL3-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; UNROLL3-NEXT:    s_mov_b32 s4, 0x41414141
@@ -16432,9 +16414,8 @@ define void @memset_p1_sz2048(ptr addrspace(1) %dst) #1 {
 ; CHECK-NEXT:    global_store_dwordx4 v[6:7], v[2:5], off offset:48
 ; CHECK-NEXT:    global_store_dwordx4 v[6:7], v[2:5], off offset:32
 ; CHECK-NEXT:    global_store_dwordx4 v[6:7], v[2:5], off offset:16
-; CHECK-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
 ; CHECK-NEXT:    global_store_dwordx4 v[6:7], v[2:5], off
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
+; CHECK-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB11_1
 ; CHECK-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; CHECK-NEXT:    s_inst_prefetch 0x2
@@ -16456,7 +16437,7 @@ define void @memset_p1_sz2048(ptr addrspace(1) %dst) #1 {
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:156
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:148
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:144
-; ALIGNED-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
+; ALIGNED-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; ALIGNED-NEXT:    global_store_byte v[2:3], v5, off offset:250
 ; ALIGNED-NEXT:    global_store_byte v[2:3], v5, off offset:251
 ; ALIGNED-NEXT:    global_store_byte v[2:3], v5, off offset:249
@@ -16513,7 +16494,6 @@ define void @memset_p1_sz2048(ptr addrspace(1) %dst) #1 {
 ; ALIGNED-NEXT:    global_store_byte v[2:3], v5, off offset:210
 ; ALIGNED-NEXT:    global_store_byte v[2:3], v5, off offset:209
 ; ALIGNED-NEXT:    global_store_byte v[2:3], v5, off offset:208
-; ALIGNED-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:136
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:140
 ; ALIGNED-NEXT:    buffer_store_dword v4, off, s[0:3], s32 offset:132
@@ -16790,7 +16770,6 @@ define void @memset_p1_sz2048(ptr addrspace(1) %dst) #1 {
 ; UNROLL3-NEXT:    v_mov_b32_e32 v4, s6
 ; UNROLL3-NEXT:    v_mov_b32_e32 v5, s7
 ; UNROLL3-NEXT:    s_mov_b64 s[4:5], 0
-; UNROLL3-NEXT:    .p2align 6
 ; UNROLL3-NEXT:  .LBB11_1: ; %static-memset-expansion-main-body
 ; UNROLL3-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; UNROLL3-NEXT:    v_add_co_u32 v6, vcc_lo, v0, s4
@@ -16799,9 +16778,8 @@ define void @memset_p1_sz2048(ptr addrspace(1) %dst) #1 {
 ; UNROLL3-NEXT:    s_addc_u32 s5, s5, 0
 ; UNROLL3-NEXT:    global_store_dwordx4 v[6:7], v[2:5], off offset:16
 ; UNROLL3-NEXT:    global_store_dwordx4 v[6:7], v[2:5], off
-; UNROLL3-NEXT:    v_cmp_gt_u64_e64 s6, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    global_store_dwordx4 v[6:7], v[2:5], off offset:32
-; UNROLL3-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
+; UNROLL3-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    s_cbranch_vccnz .LBB11_1
 ; UNROLL3-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; UNROLL3-NEXT:    s_mov_b32 s4, 0x41414141
@@ -16850,14 +16828,13 @@ define void @memset_p3_sz2048(ptr addrspace(3) %dst) #1 {
 ; CHECK-NEXT:    ds_write_b128 v0, v[1:4] offset:112
 ; CHECK-NEXT:    ds_write_b128 v0, v[1:4] offset:96
 ; CHECK-NEXT:    ds_write_b128 v0, v[1:4] offset:80
-; CHECK-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
 ; CHECK-NEXT:    ds_write_b128 v0, v[1:4] offset:64
+; CHECK-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; CHECK-NEXT:    ds_write_b128 v0, v[1:4] offset:48
 ; CHECK-NEXT:    ds_write_b128 v0, v[1:4] offset:32
 ; CHECK-NEXT:    ds_write_b128 v0, v[1:4] offset:16
 ; CHECK-NEXT:    ds_write_b128 v0, v[1:4]
 ; CHECK-NEXT:    v_add_nc_u32_e32 v0, 0x100, v0
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB12_1
 ; CHECK-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; CHECK-NEXT:    s_inst_prefetch 0x2
@@ -17124,14 +17101,13 @@ define void @memset_p3_sz2048(ptr addrspace(3) %dst) #1 {
 ; ALIGNED-NEXT:    ds_write_b8 v0, v1 offset:7
 ; ALIGNED-NEXT:    ds_write_b8 v0, v1 offset:6
 ; ALIGNED-NEXT:    ds_write_b8 v0, v1 offset:5
-; ALIGNED-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
 ; ALIGNED-NEXT:    ds_write_b8 v0, v1 offset:4
+; ALIGNED-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; ALIGNED-NEXT:    ds_write_b8 v0, v1 offset:3
 ; ALIGNED-NEXT:    ds_write_b8 v0, v1 offset:2
 ; ALIGNED-NEXT:    ds_write_b8 v0, v1 offset:1
 ; ALIGNED-NEXT:    ds_write_b8 v0, v1
 ; ALIGNED-NEXT:    v_add_nc_u32_e32 v0, 0x100, v0
-; ALIGNED-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; ALIGNED-NEXT:    s_cbranch_vccnz .LBB12_1
 ; ALIGNED-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; ALIGNED-NEXT:    s_waitcnt lgkmcnt(0)
@@ -17157,9 +17133,8 @@ define void @memset_p3_sz2048(ptr addrspace(3) %dst) #1 {
 ; UNROLL3-NEXT:    ds_write_b128 v5, v[1:4] offset:16
 ; UNROLL3-NEXT:    ds_write_b128 v5, v[1:4]
 ; UNROLL3-NEXT:    ds_write_b128 v5, v[1:4] offset:32
-; UNROLL3-NEXT:    v_cmp_gt_u64_e64 s6, 0x7e0, s[4:5]
+; UNROLL3-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    v_add_nc_u32_e32 v5, 48, v5
-; UNROLL3-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; UNROLL3-NEXT:    s_cbranch_vccnz .LBB12_1
 ; UNROLL3-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; UNROLL3-NEXT:    s_mov_b32 s4, 0x41414141
@@ -17248,14 +17223,13 @@ define void @memset_p5_sz2048(ptr addrspace(5) %dst) #1 {
 ; CHECK-NEXT:    buffer_store_dword v1, v0, s[0:3], 0 offen offset:28
 ; CHECK-NEXT:    buffer_store_dword v1, v0, s[0:3], 0 offen offset:24
 ; CHECK-NEXT:    buffer_store_dword v1, v0, s[0:3], 0 offen offset:20
-; CHECK-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
 ; CHECK-NEXT:    buffer_store_dword v1, v0, s[0:3], 0 offen offset:16
+; CHECK-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; CHECK-NEXT:    buffer_store_dword v1, v0, s[0:3], 0 offen offset:12
 ; CHECK-NEXT:    buffer_store_dword v1, v0, s[0:3], 0 offen offset:8
 ; CHECK-NEXT:    buffer_store_dword v1, v0, s[0:3], 0 offen offset:4
 ; CHECK-NEXT:    buffer_store_dword v1, v0, s[0:3], 0 offen
 ; CHECK-NEXT:    v_add_nc_u32_e32 v0, 0x100, v0
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB13_1
 ; CHECK-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; CHECK-NEXT:    s_setpc_b64 s[30:31]
@@ -17520,14 +17494,13 @@ define void @memset_p5_sz2048(ptr addrspace(5) %dst) #1 {
 ; ALIGNED-NEXT:    buffer_store_byte v1, v0, s[0:3], 0 offen offset:7
 ; ALIGNED-NEXT:    buffer_store_byte v1, v0, s[0:3], 0 offen offset:6
 ; ALIGNED-NEXT:    buffer_store_byte v1, v0, s[0:3], 0 offen offset:5
-; ALIGNED-NEXT:    v_cmp_gt_u64_e64 s6, 0x800, s[4:5]
 ; ALIGNED-NEXT:    buffer_store_byte v1, v0, s[0:3], 0 offen offset:4
+; ALIGNED-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x800, s[4:5]
 ; ALIGNED-NEXT:    buffer_store_byte v1, v0, s[0:3], 0 offen offset:3
 ; ALIGNED-NEXT:    buffer_store_byte v1, v0, s[0:3], 0 offen offset:2
 ; ALIGNED-NEXT:    buffer_store_byte v1, v0, s[0:3], 0 offen offset:1
 ; ALIGNED-NEXT:    buffer_store_byte v1, v0, s[0:3], 0 offen
 ; ALIGNED-NEXT:    v_add_nc_u32_e32 v0, 0x100, v0
-; ALIGNED-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; ALIGNED-NEXT:    s_cbranch_vccnz .LBB13_1
 ; ALIGNED-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; ALIGNED-NEXT:    s_setpc_b64 s[30:31]
@@ -17550,14 +17523,13 @@ define void @memset_p5_sz2048(ptr addrspace(5) %dst) #1 {
 ; UNROLL3-NEXT:    buffer_store_dword v1, v2, s[0:3], 0 offen offset:28
 ; UNROLL3-NEXT:    buffer_store_dword v1, v2, s[0:3], 0 offen offset:24
 ; UNROLL3-NEXT:    buffer_store_dword v1, v2, s[0:3], 0 offen offset:20
-; UNROLL3-NEXT:    v_cmp_gt_u64_e64 s6, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    buffer_store_dword v1, v2, s[0:3], 0 offen offset:16
+; UNROLL3-NEXT:    v_cmp_gt_u64_e64 vcc_lo, 0x7e0, s[4:5]
 ; UNROLL3-NEXT:    buffer_store_dword v1, v2, s[0:3], 0 offen offset:12
 ; UNROLL3-NEXT:    buffer_store_dword v1, v2, s[0:3], 0 offen offset:8
 ; UNROLL3-NEXT:    buffer_store_dword v1, v2, s[0:3], 0 offen offset:4
 ; UNROLL3-NEXT:    buffer_store_dword v1, v2, s[0:3], 0 offen
 ; UNROLL3-NEXT:    v_add_nc_u32_e32 v2, 48, v2
-; UNROLL3-NEXT:    s_and_b32 vcc_lo, exec_lo, s6
 ; UNROLL3-NEXT:    s_cbranch_vccnz .LBB13_1
 ; UNROLL3-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; UNROLL3-NEXT:    v_mov_b32_e32 v1, 0x41414141
diff --git a/llvm/test/CodeGen/AMDGPU/memset-param-combinations.ll b/llvm/test/CodeGen/AMDGPU/memset-param-combinations.ll
index e68570f5630e713..19067804cf08d1c 100644
--- a/llvm/test/CodeGen/AMDGPU/memset-param-combinations.ll
+++ b/llvm/test/CodeGen/AMDGPU/memset-param-combinations.ll
@@ -503,9 +503,8 @@ define void @memset_p0_sz1055_align_4_varsetval(ptr addrspace(0) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    v_lshl_add_u64 v[10:11], v[0:1], 0, s[0:1]
 ; GFX942-SDAG-NEXT:    s_add_u32 s0, s0, 0x100
 ; GFX942-SDAG-NEXT:    s_addc_u32 s1, s1, 0
-; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[8:9]
 ; GFX942-SDAG-NEXT:    v_lshl_add_u64 v[12:13], v[10:11], 0, s[2:3]
-; GFX942-SDAG-NEXT:    s_and_b64 vcc, exec, vcc
+; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[8:9]
 ; GFX942-SDAG-NEXT:    v_lshl_add_u64 v[14:15], v[10:11], 0, s[4:5]
 ; GFX942-SDAG-NEXT:    v_lshl_add_u64 v[16:17], v[10:11], 0, s[6:7]
 ; GFX942-SDAG-NEXT:    flat_store_dwordx4 v[10:11], v[4:7] offset:64
@@ -640,9 +639,8 @@ define void @memset_p0_sz2048_align_4_varsetval(ptr addrspace(0) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    v_lshl_add_u64 v[8:9], v[0:1], 0, s[0:1]
 ; GFX942-SDAG-NEXT:    s_add_u32 s0, s0, 0x100
 ; GFX942-SDAG-NEXT:    s_addc_u32 s1, s1, 0
-; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[6:7]
 ; GFX942-SDAG-NEXT:    v_lshl_add_u64 v[10:11], v[8:9], 0, s[2:3]
-; GFX942-SDAG-NEXT:    s_and_b64 vcc, exec, vcc
+; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[6:7]
 ; GFX942-SDAG-NEXT:    v_lshl_add_u64 v[12:13], v[8:9], 0, s[4:5]
 ; GFX942-SDAG-NEXT:    v_lshl_add_u64 v[14:15], v[8:9], 0, s[6:7]
 ; GFX942-SDAG-NEXT:    flat_store_dwordx4 v[8:9], v[2:5] offset:64
@@ -739,7 +737,6 @@ define void @memset_p1_sz1055_align_4_varsetval(ptr addrspace(1) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    s_add_u32 s0, s0, 0x100
 ; GFX942-SDAG-NEXT:    s_addc_u32 s1, s1, 0
 ; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[8:9]
-; GFX942-SDAG-NEXT:    s_and_b64 vcc, exec, vcc
 ; GFX942-SDAG-NEXT:    global_store_dwordx4 v[10:11], v[4:7], off offset:112
 ; GFX942-SDAG-NEXT:    global_store_dwordx4 v[10:11], v[4:7], off offset:96
 ; GFX942-SDAG-NEXT:    global_store_dwordx4 v[10:11], v[4:7], off offset:80
@@ -855,7 +852,6 @@ define void @memset_p1_sz2048_align_4_varsetval(ptr addrspace(1) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    s_add_u32 s0, s0, 0x100
 ; GFX942-SDAG-NEXT:    s_addc_u32 s1, s1, 0
 ; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[6:7]
-; GFX942-SDAG-NEXT:    s_and_b64 vcc, exec, vcc
 ; GFX942-SDAG-NEXT:    global_store_dwordx4 v[8:9], v[2:5], off offset:112
 ; GFX942-SDAG-NEXT:    global_store_dwordx4 v[8:9], v[2:5], off offset:96
 ; GFX942-SDAG-NEXT:    global_store_dwordx4 v[8:9], v[2:5], off offset:80
@@ -938,7 +934,6 @@ define void @memset_p3_sz1055_align_4_varsetval(ptr addrspace(3) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX942-SDAG-NEXT:    s_add_u32 s0, s0, 0x100
 ; GFX942-SDAG-NEXT:    s_addc_u32 s1, s1, 0
-; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[2:3]
 ; GFX942-SDAG-NEXT:    ds_write2_b32 v5, v4, v4 offset0:30 offset1:31
 ; GFX942-SDAG-NEXT:    ds_write2_b32 v5, v4, v4 offset0:28 offset1:29
 ; GFX942-SDAG-NEXT:    ds_write2_b32 v5, v4, v4 offset0:26 offset1:27
@@ -972,6 +967,7 @@ define void @memset_p3_sz1055_align_4_varsetval(ptr addrspace(3) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    ds_write2_b32 v5, v4, v4 offset0:34 offset1:35
 ; GFX942-SDAG-NEXT:    ds_write2_b32 v5, v4, v4 offset0:32 offset1:33
 ; GFX942-SDAG-NEXT:    v_add_u32_e32 v5, 0x100, v5
+; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[2:3]
 ; GFX942-SDAG-NEXT:    s_cbranch_vccnz .LBB8_1
 ; GFX942-SDAG-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; GFX942-SDAG-NEXT:    s_mov_b32 s0, 0x4040404
@@ -1067,7 +1063,6 @@ define void @memset_p3_sz2048_align_4_varsetval(ptr addrspace(3) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX942-SDAG-NEXT:    s_add_u32 s0, s0, 0x100
 ; GFX942-SDAG-NEXT:    s_addc_u32 s1, s1, 0
-; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[2:3]
 ; GFX942-SDAG-NEXT:    ds_write2_b32 v0, v1, v1 offset0:30 offset1:31
 ; GFX942-SDAG-NEXT:    ds_write2_b32 v0, v1, v1 offset0:28 offset1:29
 ; GFX942-SDAG-NEXT:    ds_write2_b32 v0, v1, v1 offset0:26 offset1:27
@@ -1101,6 +1096,7 @@ define void @memset_p3_sz2048_align_4_varsetval(ptr addrspace(3) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    ds_write2_b32 v0, v1, v1 offset0:34 offset1:35
 ; GFX942-SDAG-NEXT:    ds_write2_b32 v0, v1, v1 offset0:32 offset1:33
 ; GFX942-SDAG-NEXT:    v_add_u32_e32 v0, 0x100, v0
+; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[2:3]
 ; GFX942-SDAG-NEXT:    s_cbranch_vccnz .LBB9_1
 ; GFX942-SDAG-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; GFX942-SDAG-NEXT:    s_waitcnt lgkmcnt(0)
@@ -1167,7 +1163,6 @@ define void @memset_p5_sz1055_align_4_varsetval(ptr addrspace(5) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX942-SDAG-NEXT:    s_add_u32 s0, s0, 0x100
 ; GFX942-SDAG-NEXT:    s_addc_u32 s1, s1, 0
-; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[6:7]
 ; GFX942-SDAG-NEXT:    scratch_store_dwordx4 v8, v[2:5], off offset:112
 ; GFX942-SDAG-NEXT:    scratch_store_dwordx4 v8, v[2:5], off offset:96
 ; GFX942-SDAG-NEXT:    scratch_store_dwordx4 v8, v[2:5], off offset:80
@@ -1185,6 +1180,7 @@ define void @memset_p5_sz1055_align_4_varsetval(ptr addrspace(5) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    scratch_store_dwordx4 v8, v[2:5], off offset:144
 ; GFX942-SDAG-NEXT:    scratch_store_dwordx4 v8, v[2:5], off offset:128
 ; GFX942-SDAG-NEXT:    v_add_u32_e32 v8, 0x100, v8
+; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[6:7]
 ; GFX942-SDAG-NEXT:    s_cbranch_vccnz .LBB10_1
 ; GFX942-SDAG-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; GFX942-SDAG-NEXT:    s_mov_b32 s0, 0x4040404
@@ -1283,7 +1279,6 @@ define void @memset_p5_sz2048_align_4_varsetval(ptr addrspace(5) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX942-SDAG-NEXT:    s_add_u32 s0, s0, 0x100
 ; GFX942-SDAG-NEXT:    s_addc_u32 s1, s1, 0
-; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[6:7]
 ; GFX942-SDAG-NEXT:    scratch_store_dwordx4 v0, v[2:5], off offset:112
 ; GFX942-SDAG-NEXT:    scratch_store_dwordx4 v0, v[2:5], off offset:96
 ; GFX942-SDAG-NEXT:    scratch_store_dwordx4 v0, v[2:5], off offset:80
@@ -1301,6 +1296,7 @@ define void @memset_p5_sz2048_align_4_varsetval(ptr addrspace(5) align 4 %dst, i
 ; GFX942-SDAG-NEXT:    scratch_store_dwordx4 v0, v[2:5], off offset:144
 ; GFX942-SDAG-NEXT:    scratch_store_dwordx4 v0, v[2:5], off offset:128
 ; GFX942-SDAG-NEXT:    v_add_u32_e32 v0, 0x100, v0
+; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[0:1], v[6:7]
 ; GFX942-SDAG-NEXT:    s_cbranch_vccnz .LBB11_1
 ; GFX942-SDAG-NEXT:  ; %bb.2: ; %static-memset-post-expansion
 ; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/memset-pattern.ll b/llvm/test/CodeGen/AMDGPU/memset-pattern.ll
index eb6240b10939c10..b6b3c126586d8f2 100644
--- a/llvm/test/CodeGen/AMDGPU/memset-pattern.ll
+++ b/llvm/test/CodeGen/AMDGPU/memset-pattern.ll
@@ -99,10 +99,9 @@ define void @memset_pattern_i128_constlen_mainloop_and_residual_taken(ptr addrsp
 ; GFX942-SDAG-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX942-SDAG-NEXT:    s_add_u32 s0, s0, 1
 ; GFX942-SDAG-NEXT:    s_addc_u32 s1, s1, 0
-; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e64 s[2:3], s[0:1], 3
 ; GFX942-SDAG-NEXT:    global_store_dwordx4 v[4:5], v[0:3], off
 ; GFX942-SDAG-NEXT:    v_lshl_add_u64 v[4:5], v[4:5], 0, 16
-; GFX942-SDAG-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e64 vcc, s[0:1], 3
 ; GFX942-SDAG-NEXT:    s_cbranch_vccnz .LBB3_3
 ; GFX942-SDAG-NEXT:  ; %bb.4: ; %memset.pattern-post-expansion
 ; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0)
@@ -1079,9 +1078,9 @@ define void @memset_pattern_i64_as7_dynlen_unaligned_uniform_len(ptr addrspace(7
 ; GFX942-SDAG-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX942-SDAG-NEXT:    s_add_u32 s6, s6, 2
 ; GFX942-SDAG-NEXT:    s_addc_u32 s7, s7, 0
-; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[6:7], v[6:7]
 ; GFX942-SDAG-NEXT:    buffer_store_dwordx4 v[2:5], v1, s[0:3], 0 offen
 ; GFX942-SDAG-NEXT:    v_add_u32_e32 v1, 32, v1
+; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[6:7], v[6:7]
 ; GFX942-SDAG-NEXT:    s_cbranch_vccnz .LBB15_2
 ; GFX942-SDAG-NEXT:  .LBB15_3: ; %memset.pattern-expansion-residual-cond
 ; GFX942-SDAG-NEXT:    s_cmp_eq_u64 s[4:5], 0
@@ -1099,9 +1098,9 @@ define void @memset_pattern_i64_as7_dynlen_unaligned_uniform_len(ptr addrspace(7
 ; GFX942-SDAG-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX942-SDAG-NEXT:    s_add_u32 s6, s6, 1
 ; GFX942-SDAG-NEXT:    s_addc_u32 s7, s7, 0
-; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[6:7], v[2:3]
 ; GFX942-SDAG-NEXT:    buffer_store_dwordx2 v[0:1], v4, s[0:3], 0 offen
 ; GFX942-SDAG-NEXT:    v_add_u32_e32 v4, 8, v4
+; GFX942-SDAG-NEXT:    v_cmp_lt_u64_e32 vcc, s[6:7], v[2:3]
 ; GFX942-SDAG-NEXT:    s_cbranch_vccnz .LBB15_5
 ; GFX942-SDAG-NEXT:  .LBB15_6: ; %memset.pattern-post-expansion
 ; GFX942-SDAG-NEXT:    s_waitcnt vmcnt(0)
diff --git a/llvm/test/CodeGen/AMDGPU/mul.ll b/llvm/test/CodeGen/AMDGPU/mul.ll
index 29b44d00e9ea24e..38c01fc4ff439f7 100644
--- a/llvm/test/CodeGen/AMDGPU/mul.ll
+++ b/llvm/test/CodeGen/AMDGPU/mul.ll
@@ -3050,8 +3050,7 @@ define amdgpu_kernel void @mul64_in_branch(ptr addrspace(1) %out, ptr addrspace(
 ; SI-NEXT:    s_load_dwordx8 s[0:7], s[4:5], 0x9
 ; SI-NEXT:    s_mov_b64 s[8:9], 0
 ; SI-NEXT:    s_waitcnt lgkmcnt(0)
-; SI-NEXT:    v_cmp_ne_u64_e64 s[10:11], s[4:5], 0
-; SI-NEXT:    s_and_b64 vcc, exec, s[10:11]
+; SI-NEXT:    v_cmp_ne_u64_e64 vcc, s[4:5], 0
 ; SI-NEXT:    s_cbranch_vccz .LBB16_2
 ; SI-NEXT:  ; %bb.1: ; %else
 ; SI-NEXT:    v_mov_b32_e32 v0, s6
diff --git a/llvm/test/CodeGen/AMDGPU/multi-divergent-exit-region.ll b/llvm/test/CodeGen/AMDGPU/multi-divergent-exit-region.ll
index f72cb2327af4c40..cac188ec5ba7fe1 100644
--- a/llvm/test/CodeGen/AMDGPU/multi-divergent-exit-region.ll
+++ b/llvm/test/CodeGen/AMDGPU/multi-divergent-exit-region.ll
@@ -68,9 +68,8 @@
 ; GCN-DAG:  s_xor_b64
 
 ; GCN: ; %LeafBlock1
-; GCN-NEXT: v_cmp_ne_u32_e32    vcc, 2,
 ; GCN-NEXT: s_mov_b64           [[EXIT0]], exec
-; GCN-NEXT: s_and_b64           [[EXIT1]], vcc, exec
+; GCN-NEXT: v_cmp_ne_u32_e64    [[EXIT1]], 2,
 
 ; GCN: ; %Flow
 ; GCN-NEXT: s_andn2_saveexec_b64
@@ -80,10 +79,8 @@
 ; GCN-DAG:  v_cmp_ne_u32_e64    [[INV:s\[[0-9]+:[0-9]+\]]], 1,
 ; GCN-DAG:  s_andn2_b64         [[EXIT0]], [[EXIT0]], exec
 ; GCN-DAG:  s_andn2_b64         [[EXIT1]], [[EXIT1]], exec
-; GCN-DAG:  s_and_b64           [[TMP0:s\[[0-9]+:[0-9]+\]]], vcc, exec
-; GCN-DAG:  s_and_b64           [[TMP1:s\[[0-9]+:[0-9]+\]]], [[INV]], exec
-; GCN-DAG:  s_or_b64            [[EXIT0]], [[EXIT0]], [[TMP0]]
-; GCN-DAG:  s_or_b64            [[EXIT1]], [[EXIT1]], [[TMP1]]
+; GCN-DAG:  s_or_b64            [[EXIT0]], [[EXIT0]], vcc
+; GCN-DAG:  s_or_b64            [[EXIT1]], [[EXIT1]], [[INV]]
 
 ; GCN: ; %Flow4
 ; GCN-NEXT: s_or_b64            exec, exec,
@@ -361,7 +358,7 @@ exit1:                                     ; preds = %LeafBlock, %LeafBlock1
 ; GCN: s_cmp_gt_i32 s0, 1
 ; GCN: s_cbranch_scc0 [[FLOW:.LBB[0-9]+_[0-9]+]]
 
-; GCN: v_cmp_ne_u32_e32 vcc, 7, v0
+; GCN: v_cmp_ne_u32_e64 s{{\[[0-9]+:[0-9]+\]}}, 7, v0
 
 ; GCN: {{^}}[[FLOW]]:
 
diff --git a/llvm/test/CodeGen/AMDGPU/multilevel-break.ll b/llvm/test/CodeGen/AMDGPU/multilevel-break.ll
index 0c5b90d9c2ae0c7..a872344940b5381 100644
--- a/llvm/test/CodeGen/AMDGPU/multilevel-break.ll
+++ b/llvm/test/CodeGen/AMDGPU/multilevel-break.ll
@@ -60,16 +60,16 @@ define amdgpu_vs void @multi_else_break(<4 x float> %vec, i32 %ub, i32 %cont) {
 ; GCN-NEXT:    ; Child Loop BB0_4 Depth 2
 ; GCN-NEXT:    s_mov_b64 s[4:5], 0
 ; GCN-NEXT:    ; implicit-def: $sgpr2_sgpr3
-; GCN-NEXT:    ; implicit-def: $sgpr8_sgpr9
 ; GCN-NEXT:    ; implicit-def: $sgpr6_sgpr7
+; GCN-NEXT:    ; implicit-def: $sgpr8_sgpr9
 ; GCN-NEXT:    s_branch .LBB0_4
 ; GCN-NEXT:  .LBB0_3: ; %Flow
 ; GCN-NEXT:    ; in Loop: Header=BB0_4 Depth=2
 ; GCN-NEXT:    s_or_b64 exec, exec, s[10:11]
-; GCN-NEXT:    s_and_b64 s[10:11], exec, s[8:9]
+; GCN-NEXT:    s_and_b64 s[10:11], exec, s[6:7]
 ; GCN-NEXT:    s_or_b64 s[4:5], s[10:11], s[4:5]
 ; GCN-NEXT:    s_andn2_b64 s[2:3], s[2:3], exec
-; GCN-NEXT:    s_and_b64 s[10:11], s[6:7], exec
+; GCN-NEXT:    s_and_b64 s[10:11], s[8:9], exec
 ; GCN-NEXT:    s_or_b64 s[2:3], s[2:3], s[10:11]
 ; GCN-NEXT:    s_andn2_b64 exec, exec, s[4:5]
 ; GCN-NEXT:    s_cbranch_execz .LBB0_1
@@ -77,18 +77,17 @@ define amdgpu_vs void @multi_else_break(<4 x float> %vec, i32 %ub, i32 %cont) {
 ; GCN-NEXT:    ; Parent Loop BB0_2 Depth=1
 ; GCN-NEXT:    ; => This Inner Loop Header: Depth=2
 ; GCN-NEXT:    v_cmp_lt_i32_e32 vcc, v0, v4
-; GCN-NEXT:    s_or_b64 s[6:7], s[6:7], exec
 ; GCN-NEXT:    s_or_b64 s[8:9], s[8:9], exec
+; GCN-NEXT:    s_or_b64 s[6:7], s[6:7], exec
 ; GCN-NEXT:    s_and_saveexec_b64 s[10:11], vcc
 ; GCN-NEXT:    s_cbranch_execz .LBB0_3
 ; GCN-NEXT:  ; %bb.5: ; %ENDIF
 ; GCN-NEXT:    ; in Loop: Header=BB0_4 Depth=2
 ; GCN-NEXT:    v_add_i32_e32 v0, vcc, 1, v0
 ; GCN-NEXT:    v_cmp_ne_u32_e32 vcc, v5, v0
-; GCN-NEXT:    s_andn2_b64 s[8:9], s[8:9], exec
-; GCN-NEXT:    s_and_b64 s[12:13], vcc, exec
 ; GCN-NEXT:    s_andn2_b64 s[6:7], s[6:7], exec
-; GCN-NEXT:    s_or_b64 s[8:9], s[8:9], s[12:13]
+; GCN-NEXT:    s_andn2_b64 s[8:9], s[8:9], exec
+; GCN-NEXT:    s_or_b64 s[6:7], s[6:7], vcc
 ; GCN-NEXT:    s_branch .LBB0_3
 ; GCN-NEXT:  .LBB0_6: ; %IF
 ; GCN-NEXT:    s_endpgm
@@ -214,8 +213,7 @@ define amdgpu_kernel void @multi_if_break_loop(i32 %arg) #0 {
 ; GCN-NEXT:    s_waitcnt vmcnt(0)
 ; GCN-NEXT:    s_andn2_b64 s[4:5], s[4:5], exec
 ; GCN-NEXT:    v_cmp_ge_i32_e32 vcc, v0, v1
-; GCN-NEXT:    s_and_b64 s[6:7], vcc, exec
-; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], s[6:7]
+; GCN-NEXT:    s_or_b64 s[4:5], s[4:5], vcc
 ; GCN-NEXT:    s_branch .LBB1_1
 ; GCN-NEXT:  .LBB1_9: ; %bb9
 ; GCN-NEXT:    s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/no-dup-inst-prefetch.ll b/llvm/test/CodeGen/AMDGPU/no-dup-inst-prefetch.ll
index 22f17fa4a6dc9f4..27f34f84f23faa3 100644
--- a/llvm/test/CodeGen/AMDGPU/no-dup-inst-prefetch.ll
+++ b/llvm/test/CodeGen/AMDGPU/no-dup-inst-prefetch.ll
@@ -45,7 +45,6 @@ define amdgpu_cs void @_amdgpu_cs_main(float %0, i32 %1) {
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-NEXT:    v_fma_f32 v1, v1, v0, 0
 ; GFX10-NEXT:    v_cmp_le_f32_e64 s0, 0, v1
-; GFX10-NEXT:    s_and_b32 s0, s0, exec_lo
 ; GFX10-NEXT:    s_or_b32 s2, s2, s0
 ; GFX10-NEXT:    s_branch .LBB0_1
 ; GFX10-NEXT:  .LBB0_4: ; %loop0_merge
@@ -92,9 +91,8 @@ define amdgpu_cs void @_amdgpu_cs_main(float %0, i32 %1) {
 ; GFX12-NEXT:    image_sample_lz v1, [v2, v2, v1], s[8:15], s[4:7] dmask:0x1 dim:SQ_RSRC_IMG_3D
 ; GFX12-NEXT:    s_wait_samplecnt 0x0
 ; GFX12-NEXT:    v_fma_f32 v1, v1, v0, 0
-; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX12-NEXT:    v_cmp_le_f32_e64 s0, 0, v1
-; GFX12-NEXT:    s_and_b32 s0, s0, exec_lo
 ; GFX12-NEXT:    s_or_b32 s2, s2, s0
 ; GFX12-NEXT:    s_branch .LBB0_1
 ; GFX12-NEXT:  .LBB0_4: ; %loop0_merge
diff --git a/llvm/test/CodeGen/AMDGPU/select-opt.ll b/llvm/test/CodeGen/AMDGPU/select-opt.ll
index 83eb86c075e6cc7..14c6503004d39f4 100644
--- a/llvm/test/CodeGen/AMDGPU/select-opt.ll
+++ b/llvm/test/CodeGen/AMDGPU/select-opt.ll
@@ -158,7 +158,7 @@ define amdgpu_kernel void @opt_select_i64_or_cmp_f32(ptr addrspace(1) %out, floa
 }
 
 ; GCN-LABEL: {{^}}regression:
-; GCN: v_cmp_neq_f32_e64 s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}}, 1.0
+; GCN: v_cmp_neq_f32_e64 vcc, s{{[0-9]+}}, 1.0
 
 define amdgpu_kernel void @regression(ptr addrspace(1) %out, float %c0, float %c1) #0 {
 entry:
diff --git a/llvm/test/CodeGen/AMDGPU/setcc.ll b/llvm/test/CodeGen/AMDGPU/setcc.ll
index 575b128b1f3af02..dd6f4e11373816d 100644
--- a/llvm/test/CodeGen/AMDGPU/setcc.ll
+++ b/llvm/test/CodeGen/AMDGPU/setcc.ll
@@ -394,7 +394,7 @@ endif:
 ; FUNC-LABEL: setcc-i1-and-xor
 ; GCN-DAG: v_cmp_nge_f32_e64 [[A:s\[[0-9]+:[0-9]+\]]], s{{[0-9]+}}, 0{{$}}
 ; GCN-DAG: v_cmp_nle_f32_e64 [[B:s\[[0-9]+:[0-9]+\]]], s{{[0-9]+}}, 1.0
-; GCN: s_or_b64 s[2:3], [[A]], [[B]]
+; GCN: s_or_b64 vcc, [[A]], [[B]]
 define amdgpu_kernel void @setcc-i1-and-xor(ptr addrspace(1) %out, float %cond) #0 {
 bb0:
   %tmp5 = fcmp oge float %cond, 0.000000e+00
diff --git a/llvm/test/CodeGen/AMDGPU/sgpr-control-flow.ll b/llvm/test/CodeGen/AMDGPU/sgpr-control-flow.ll
index a32cfd0d45950a1..d986c4447421c7f 100644
--- a/llvm/test/CodeGen/AMDGPU/sgpr-control-flow.ll
+++ b/llvm/test/CodeGen/AMDGPU/sgpr-control-flow.ll
@@ -199,8 +199,7 @@ define amdgpu_kernel void @sgpr_if_else_valu_cmp_phi_br(ptr addrspace(1) %out, p
 ; SI-NEXT:    s_andn2_b64 s[2:3], s[8:9], exec
 ; SI-NEXT:    s_waitcnt vmcnt(0)
 ; SI-NEXT:    v_cmp_eq_u32_e32 vcc, 0, v0
-; SI-NEXT:    s_and_b64 s[6:7], vcc, exec
-; SI-NEXT:    s_or_b64 s[8:9], s[2:3], s[6:7]
+; SI-NEXT:    s_or_b64 s[8:9], s[2:3], vcc
 ; SI-NEXT:  .LBB3_4: ; %endif
 ; SI-NEXT:    s_or_b64 exec, exec, s[4:5]
 ; SI-NEXT:    s_mov_b32 s3, 0xf000
diff --git a/llvm/test/CodeGen/AMDGPU/si-unify-exit-return-unreachable.ll b/llvm/test/CodeGen/AMDGPU/si-unify-exit-return-unreachable.ll
index 4f41c239f9482d7..30a9227f6e49ecc 100644
--- a/llvm/test/CodeGen/AMDGPU/si-unify-exit-return-unreachable.ll
+++ b/llvm/test/CodeGen/AMDGPU/si-unify-exit-return-unreachable.ll
@@ -139,20 +139,17 @@ define void @my_func(i32 %0) {
 ; GCN-NEXT:    s_and_saveexec_b64 s[4:5], vcc
 ; GCN-NEXT:    s_xor_b64 s[4:5], exec, s[4:5]
 ; GCN-NEXT:  ; %bb.6: ; %LeafBlock5
-; GCN-NEXT:    v_cmp_eq_u32_e32 vcc, 2, v0
 ; GCN-NEXT:    s_mov_b64 s[6:7], exec
-; GCN-NEXT:    s_and_b64 s[8:9], vcc, exec
+; GCN-NEXT:    v_cmp_eq_u32_e64 s[8:9], 2, v0
 ; GCN-NEXT:  ; %bb.7: ; %Flow13
 ; GCN-NEXT:    s_andn2_saveexec_b64 s[10:11], s[4:5]
 ; GCN-NEXT:  ; %bb.8: ; %LeafBlock3
 ; GCN-NEXT:    v_cmp_ne_u32_e64 s[4:5], 0, v0
-; GCN-NEXT:    v_cmp_eq_u32_e32 vcc, 0, v0
 ; GCN-NEXT:    s_andn2_b64 s[6:7], s[6:7], exec
-; GCN-NEXT:    s_and_b64 s[4:5], s[4:5], exec
+; GCN-NEXT:    v_cmp_eq_u32_e32 vcc, 0, v0
 ; GCN-NEXT:    s_or_b64 s[6:7], s[6:7], s[4:5]
 ; GCN-NEXT:    s_andn2_b64 s[4:5], s[8:9], exec
-; GCN-NEXT:    s_and_b64 s[8:9], vcc, exec
-; GCN-NEXT:    s_or_b64 s[8:9], s[4:5], s[8:9]
+; GCN-NEXT:    s_or_b64 s[8:9], s[4:5], vcc
 ; GCN-NEXT:  ; %bb.9: ; %Flow14
 ; GCN-NEXT:    s_or_b64 exec, exec, s[10:11]
 ; GCN-NEXT:    s_mov_b64 s[4:5], 0
diff --git a/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll b/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll
index c3dd793907bebf6..3481028351a676b 100644
--- a/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll
+++ b/llvm/test/CodeGen/AMDGPU/skip-if-dead.ll
@@ -1293,8 +1293,7 @@ end:
 define amdgpu_ps void @no_skip_no_successors(float inreg %arg, float inreg %arg1) #0 {
 ; SI-LABEL: no_skip_no_successors:
 ; SI:       ; %bb.0: ; %bb
-; SI-NEXT:    v_cmp_nge_f32_e64 s[4:5], s1, 0
-; SI-NEXT:    s_and_b64 vcc, exec, s[4:5]
+; SI-NEXT:    v_cmp_nge_f32_e64 vcc, s1, 0
 ; SI-NEXT:    s_cbranch_vccz .LBB12_3
 ; SI-NEXT:  ; %bb.1: ; %bb6
 ; SI-NEXT:    s_mov_b64 s[2:3], exec
@@ -1305,7 +1304,6 @@ define amdgpu_ps void @no_skip_no_successors(float inreg %arg, float inreg %arg1
 ; SI-NEXT:  .LBB12_3: ; %bb3
 ; SI-NEXT:    v_mov_b32_e32 v0, 0x3e7ae148
 ; SI-NEXT:    v_cmp_nge_f32_e32 vcc, s0, v0
-; SI-NEXT:    s_and_b64 vcc, exec, vcc
 ; SI-NEXT:  ; %bb.4: ; %bb5
 ; SI-NEXT:  .LBB12_5:
 ; SI-NEXT:    s_mov_b64 exec, 0
@@ -1314,8 +1312,7 @@ define amdgpu_ps void @no_skip_no_successors(float inreg %arg, float inreg %arg1
 ;
 ; GFX10-WAVE64-LABEL: no_skip_no_successors:
 ; GFX10-WAVE64:       ; %bb.0: ; %bb
-; GFX10-WAVE64-NEXT:    v_cmp_nge_f32_e64 s[4:5], s1, 0
-; GFX10-WAVE64-NEXT:    s_and_b64 vcc, exec, s[4:5]
+; GFX10-WAVE64-NEXT:    v_cmp_nge_f32_e64 vcc, s1, 0
 ; GFX10-WAVE64-NEXT:    s_cbranch_vccz .LBB12_3
 ; GFX10-WAVE64-NEXT:  ; %bb.1: ; %bb6
 ; GFX10-WAVE64-NEXT:    s_mov_b64 s[2:3], exec
@@ -1324,8 +1321,7 @@ define amdgpu_ps void @no_skip_no_successors(float inreg %arg, float inreg %arg1
 ; GFX10-WAVE64-NEXT:  ; %bb.2: ; %bb6
 ; GFX10-WAVE64-NEXT:    s_mov_b64 exec, 0
 ; GFX10-WAVE64-NEXT:  .LBB12_3: ; %bb3
-; GFX10-WAVE64-NEXT:    v_cmp_nle_f32_e64 s[0:1], 0x3e7ae148, s0
-; GFX10-WAVE64-NEXT:    s_and_b64 vcc, exec, s[0:1]
+; GFX10-WAVE64-NEXT:    v_cmp_nle_f32_e64 vcc, 0x3e7ae148, s0
 ; GFX10-WAVE64-NEXT:  ; %bb.4: ; %bb5
 ; GFX10-WAVE64-NEXT:  .LBB12_5:
 ; GFX10-WAVE64-NEXT:    s_mov_b64 exec, 0
@@ -1334,8 +1330,7 @@ define amdgpu_ps void @no_skip_no_successors(float inreg %arg, float inreg %arg1
 ;
 ; GFX10-WAVE32-LABEL: no_skip_no_successors:
 ; GFX10-WAVE32:       ; %bb.0: ; %bb
-; GFX10-WAVE32-NEXT:    v_cmp_nge_f32_e64 s1, s1, 0
-; GFX10-WAVE32-NEXT:    s_and_b32 vcc_lo, exec_lo, s1
+; GFX10-WAVE32-NEXT:    v_cmp_nge_f32_e64 vcc_lo, s1, 0
 ; GFX10-WAVE32-NEXT:    s_cbranch_vccz .LBB12_3
 ; GFX10-WAVE32-NEXT:  ; %bb.1: ; %bb6
 ; GFX10-WAVE32-NEXT:    s_mov_b32 s2, exec_lo
@@ -1344,8 +1339,7 @@ define amdgpu_ps void @no_skip_no_successors(float inreg %arg, float inreg %arg1
 ; GFX10-WAVE32-NEXT:  ; %bb.2: ; %bb6
 ; GFX10-WAVE32-NEXT:    s_mov_b32 exec_lo, 0
 ; GFX10-WAVE32-NEXT:  .LBB12_3: ; %bb3
-; GFX10-WAVE32-NEXT:    v_cmp_nle_f32_e64 s0, 0x3e7ae148, s0
-; GFX10-WAVE32-NEXT:    s_and_b32 vcc_lo, exec_lo, s0
+; GFX10-WAVE32-NEXT:    v_cmp_nle_f32_e64 vcc_lo, 0x3e7ae148, s0
 ; GFX10-WAVE32-NEXT:  ; %bb.4: ; %bb5
 ; GFX10-WAVE32-NEXT:  .LBB12_5:
 ; GFX10-WAVE32-NEXT:    s_mov_b32 exec_lo, 0
@@ -1354,8 +1348,7 @@ define amdgpu_ps void @no_skip_no_successors(float inreg %arg, float inreg %arg1
 ;
 ; GFX11-LABEL: no_skip_no_successors:
 ; GFX11:       ; %bb.0: ; %bb
-; GFX11-NEXT:    v_cmp_nge_f32_e64 s[4:5], s1, 0
-; GFX11-NEXT:    s_and_b64 vcc, exec, s[4:5]
+; GFX11-NEXT:    v_cmp_nge_f32_e64 vcc, s1, 0
 ; GFX11-NEXT:    s_cbranch_vccz .LBB12_3
 ; GFX11-NEXT:  ; %bb.1: ; %bb6
 ; GFX11-NEXT:    s_mov_b64 s[2:3], exec
@@ -1365,8 +1358,7 @@ define amdgpu_ps void @no_skip_no_successors(float inreg %arg, float inreg %arg1
 ; GFX11-NEXT:  ; %bb.2: ; %bb6
 ; GFX11-NEXT:    s_mov_b64 exec, 0
 ; GFX11-NEXT:  .LBB12_3: ; %bb3
-; GFX11-NEXT:    v_cmp_nle_f32_e64 s[0:1], 0x3e7ae148, s0
-; GFX11-NEXT:    s_and_b64 vcc, exec, s[0:1]
+; GFX11-NEXT:    v_cmp_nle_f32_e64 vcc, 0x3e7ae148, s0
 ; GFX11-NEXT:  ; %bb.4: ; %bb5
 ; GFX11-NEXT:  .LBB12_5:
 ; GFX11-NEXT:    s_mov_b64 exec, 0
diff --git a/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll b/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll
index 5a904be43ddbc9d..d998e263ef73461 100644
--- a/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll
+++ b/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll
@@ -623,7 +623,6 @@ define void @f(ptr %p, <16 x i1> %m, <16 x i63> %pt, <16 x i1> %sc,
 ; CHECK-NEXT:  ; %bb.5: ; %LeafBlock
 ; CHECK-NEXT:    v_cmp_ne_u32_e64 s[38:39], 2, v22
 ; CHECK-NEXT:    s_andn2_b64 s[52:53], s[52:53], exec
-; CHECK-NEXT:    s_and_b64 s[38:39], s[38:39], exec
 ; CHECK-NEXT:    s_mov_b64 s[50:51], exec
 ; CHECK-NEXT:    s_or_b64 s[52:53], s[52:53], s[38:39]
 ; CHECK-NEXT:  ; %bb.6: ; %Flow53
diff --git a/llvm/test/CodeGen/AMDGPU/structurize-hoist.ll b/llvm/test/CodeGen/AMDGPU/structurize-hoist.ll
index 981dd9a81243af6..134855cef6d76fa 100644
--- a/llvm/test/CodeGen/AMDGPU/structurize-hoist.ll
+++ b/llvm/test/CodeGen/AMDGPU/structurize-hoist.ll
@@ -120,9 +120,8 @@ define amdgpu_kernel void @test_loop_with_if( ptr %ptr, i1 %cond) #0 {
 ; GFX900-NEXT:    ; in Loop: Header=BB2_2 Depth=1
 ; GFX900-NEXT:    v_cmp_gt_i32_e32 vcc, 11, v5
 ; GFX900-NEXT:    s_andn2_b64 s[8:9], s[4:5], exec
-; GFX900-NEXT:    s_and_b64 s[12:13], vcc, exec
 ; GFX900-NEXT:    s_mov_b64 s[6:7], -1
-; GFX900-NEXT:    s_or_b64 s[8:9], s[8:9], s[12:13]
+; GFX900-NEXT:    s_or_b64 s[8:9], s[8:9], vcc
 ; GFX900-NEXT:  .LBB2_4: ; %Flow
 ; GFX900-NEXT:    ; in Loop: Header=BB2_2 Depth=1
 ; GFX900-NEXT:    s_and_saveexec_b64 s[12:13], s[8:9]
diff --git a/llvm/test/CodeGen/AMDGPU/sub_i1.ll b/llvm/test/CodeGen/AMDGPU/sub_i1.ll
index f817b19cae0a463..be7206073ab7b39 100644
--- a/llvm/test/CodeGen/AMDGPU/sub_i1.ll
+++ b/llvm/test/CodeGen/AMDGPU/sub_i1.ll
@@ -142,8 +142,7 @@ define amdgpu_kernel void @sub_i1_cf(ptr addrspace(1) %out, ptr addrspace(1) %a,
 ; GFX9-NEXT:    s_andn2_b64 s[2:3], s[4:5], exec
 ; GFX9-NEXT:    v_and_b32_e32 v0, 1, v0
 ; GFX9-NEXT:    v_cmp_eq_u32_e32 vcc, 1, v0
-; GFX9-NEXT:    s_and_b64 s[4:5], vcc, exec
-; GFX9-NEXT:    s_or_b64 s[4:5], s[2:3], s[4:5]
+; GFX9-NEXT:    s_or_b64 s[4:5], s[2:3], vcc
 ; GFX9-NEXT:  .LBB2_4: ; %endif
 ; GFX9-NEXT:    s_or_b64 exec, exec, s[6:7]
 ; GFX9-NEXT:    s_waitcnt lgkmcnt(0)
@@ -182,8 +181,7 @@ define amdgpu_kernel void @sub_i1_cf(ptr addrspace(1) %out, ptr addrspace(1) %a,
 ; GFX10-NEXT:    s_andn2_b32 s2, s4, exec_lo
 ; GFX10-NEXT:    v_and_b32_e32 v0, 1, v0
 ; GFX10-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 1, v0
-; GFX10-NEXT:    s_and_b32 s3, vcc_lo, exec_lo
-; GFX10-NEXT:    s_or_b32 s4, s2, s3
+; GFX10-NEXT:    s_or_b32 s4, s2, vcc_lo
 ; GFX10-NEXT:  .LBB2_4: ; %endif
 ; GFX10-NEXT:    s_or_b32 exec_lo, exec_lo, s5
 ; GFX10-NEXT:    s_waitcnt lgkmcnt(0)
@@ -223,10 +221,9 @@ define amdgpu_kernel void @sub_i1_cf(ptr addrspace(1) %out, ptr addrspace(1) %a,
 ; GFX11-NEXT:    s_waitcnt vmcnt(0)
 ; GFX11-NEXT:    s_and_not1_b32 s2, s6, exec_lo
 ; GFX11-NEXT:    v_and_b32_e32 v0, 1, v0
-; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX11-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 1, v0
-; GFX11-NEXT:    s_and_b32 s3, vcc_lo, exec_lo
-; GFX11-NEXT:    s_or_b32 s6, s2, s3
+; GFX11-NEXT:    s_or_b32 s6, s2, vcc_lo
 ; GFX11-NEXT:  .LBB2_4: ; %endif
 ; GFX11-NEXT:    s_or_b32 exec_lo, exec_lo, s4
 ; GFX11-NEXT:    s_xor_b32 s2, s6, -1
diff --git a/llvm/test/CodeGen/AMDGPU/uaddo.ll b/llvm/test/CodeGen/AMDGPU/uaddo.ll
index b2326ee18d8067c..8b010f1d7636327 100644
--- a/llvm/test/CodeGen/AMDGPU/uaddo.ll
+++ b/llvm/test/CodeGen/AMDGPU/uaddo.ll
@@ -1086,8 +1086,8 @@ define amdgpu_kernel void @v_uaddo_clamp_bit(ptr addrspace(1) %out, ptr addrspac
 ; SI-NEXT:    buffer_load_dword v2, off, s[8:11], 0
 ; SI-NEXT:    s_mov_b64 s[8:9], 0
 ; SI-NEXT:    s_waitcnt vmcnt(0)
-; SI-NEXT:    v_cmp_eq_u32_e32 vcc, v1, v2
 ; SI-NEXT:    v_add_i32_e64 v0, s[0:1], v1, v2
+; SI-NEXT:    v_cmp_eq_u32_e32 vcc, v1, v2
 ; SI-NEXT:    s_cbranch_vccnz .LBB9_2
 ; SI-NEXT:  ; %bb.1: ; %if
 ; SI-NEXT:    s_xor_b64 s[8:9], s[0:1], -1
@@ -1118,8 +1118,8 @@ define amdgpu_kernel void @v_uaddo_clamp_bit(ptr addrspace(1) %out, ptr addrspac
 ; VI-NEXT:    flat_load_dword v1, v[0:1]
 ; VI-NEXT:    flat_load_dword v2, v[2:3]
 ; VI-NEXT:    s_waitcnt vmcnt(0)
-; VI-NEXT:    v_cmp_eq_u32_e32 vcc, v1, v2
 ; VI-NEXT:    v_add_u32_e64 v0, s[0:1], v1, v2
+; VI-NEXT:    v_cmp_eq_u32_e32 vcc, v1, v2
 ; VI-NEXT:    s_cbranch_vccnz .LBB9_2
 ; VI-NEXT:  ; %bb.1: ; %if
 ; VI-NEXT:    s_xor_b64 s[2:3], s[0:1], -1
@@ -1144,8 +1144,8 @@ define amdgpu_kernel void @v_uaddo_clamp_bit(ptr addrspace(1) %out, ptr addrspac
 ; GFX9-NEXT:    global_load_dword v2, v0, s[12:13]
 ; GFX9-NEXT:    global_load_dword v3, v0, s[14:15]
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
-; GFX9-NEXT:    v_cmp_eq_u32_e32 vcc, v2, v3
 ; GFX9-NEXT:    v_add_co_u32_e64 v1, s[0:1], v2, v3
+; GFX9-NEXT:    v_cmp_eq_u32_e32 vcc, v2, v3
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB9_2
 ; GFX9-NEXT:  ; %bb.1: ; %if
 ; GFX9-NEXT:    s_xor_b64 s[2:3], s[0:1], -1
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-cfg.ll b/llvm/test/CodeGen/AMDGPU/uniform-cfg.ll
index eaeb4396f73f26a..61b58e931531341 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-cfg.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-cfg.ll
@@ -60,8 +60,7 @@ define amdgpu_kernel void @uniform_if_vcc(float %cond, ptr addrspace(1) %out) {
 ; SI-NEXT:    s_load_dword s1, s[4:5], 0x9
 ; SI-NEXT:    s_mov_b32 s0, 0
 ; SI-NEXT:    s_waitcnt lgkmcnt(0)
-; SI-NEXT:    v_cmp_eq_f32_e64 s[2:3], s1, 0
-; SI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; SI-NEXT:    v_cmp_eq_f32_e64 vcc, s1, 0
 ; SI-NEXT:    s_cbranch_vccnz .LBB1_2
 ; SI-NEXT:  ; %bb.1: ; %else
 ; SI-NEXT:    s_mov_b32 s0, 1
@@ -79,8 +78,7 @@ define amdgpu_kernel void @uniform_if_vcc(float %cond, ptr addrspace(1) %out) {
 ; VI-NEXT:    s_load_dword s1, s[4:5], 0x24
 ; VI-NEXT:    s_mov_b32 s0, 0
 ; VI-NEXT:    s_waitcnt lgkmcnt(0)
-; VI-NEXT:    v_cmp_eq_f32_e64 s[2:3], s1, 0
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_eq_f32_e64 vcc, s1, 0
 ; VI-NEXT:    s_cbranch_vccnz .LBB1_2
 ; VI-NEXT:  ; %bb.1: ; %else
 ; VI-NEXT:    s_mov_b32 s0, 1
@@ -166,8 +164,7 @@ define amdgpu_kernel void @uniform_if_swap_br_targets_vcc(float %cond, ptr addrs
 ; SI-NEXT:    s_load_dword s1, s[4:5], 0x9
 ; SI-NEXT:    s_mov_b32 s0, 0
 ; SI-NEXT:    s_waitcnt lgkmcnt(0)
-; SI-NEXT:    v_cmp_neq_f32_e64 s[2:3], s1, 0
-; SI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; SI-NEXT:    v_cmp_neq_f32_e64 vcc, s1, 0
 ; SI-NEXT:    s_cbranch_vccnz .LBB3_2
 ; SI-NEXT:  ; %bb.1: ; %else
 ; SI-NEXT:    s_mov_b32 s0, 1
@@ -185,8 +182,7 @@ define amdgpu_kernel void @uniform_if_swap_br_targets_vcc(float %cond, ptr addrs
 ; VI-NEXT:    s_load_dword s1, s[4:5], 0x24
 ; VI-NEXT:    s_mov_b32 s0, 0
 ; VI-NEXT:    s_waitcnt lgkmcnt(0)
-; VI-NEXT:    v_cmp_neq_f32_e64 s[2:3], s1, 0
-; VI-NEXT:    s_and_b64 vcc, exec, s[2:3]
+; VI-NEXT:    v_cmp_neq_f32_e64 vcc, s1, 0
 ; VI-NEXT:    s_cbranch_vccnz .LBB3_2
 ; VI-NEXT:  ; %bb.1: ; %else
 ; VI-NEXT:    s_mov_b32 s0, 1
@@ -875,9 +871,8 @@ define amdgpu_kernel void @uniform_if_scc_i64_eq(i64 %cond, ptr addrspace(1) %ou
 ; SI:       ; %bb.0: ; %entry
 ; SI-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x9
 ; SI-NEXT:    s_waitcnt lgkmcnt(0)
-; SI-NEXT:    v_cmp_eq_u64_e64 s[4:5], s[0:1], 0
+; SI-NEXT:    v_cmp_eq_u64_e64 vcc, s[0:1], 0
 ; SI-NEXT:    s_mov_b32 s0, 0
-; SI-NEXT:    s_and_b64 vcc, exec, s[4:5]
 ; SI-NEXT:    s_cbranch_vccnz .LBB15_2
 ; SI-NEXT:  ; %bb.1: ; %else
 ; SI-NEXT:    s_mov_b32 s0, 1
@@ -929,9 +924,8 @@ define amdgpu_kernel void @uniform_if_scc_i64_ne(i64 %cond, ptr addrspace(1) %ou
 ; SI:       ; %bb.0: ; %entry
 ; SI-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x9
 ; SI-NEXT:    s_waitcnt lgkmcnt(0)
-; SI-NEXT:    v_cmp_ne_u64_e64 s[4:5], s[0:1], 0
+; SI-NEXT:    v_cmp_ne_u64_e64 vcc, s[0:1], 0
 ; SI-NEXT:    s_mov_b32 s0, 0
-; SI-NEXT:    s_and_b64 vcc, exec, s[4:5]
 ; SI-NEXT:    s_cbranch_vccnz .LBB16_2
 ; SI-NEXT:  ; %bb.1: ; %else
 ; SI-NEXT:    s_mov_b32 s0, 1
@@ -983,9 +977,8 @@ define amdgpu_kernel void @uniform_if_scc_i64_sgt(i64 %cond, ptr addrspace(1) %o
 ; SI:       ; %bb.0: ; %entry
 ; SI-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x9
 ; SI-NEXT:    s_waitcnt lgkmcnt(0)
-; SI-NEXT:    v_cmp_gt_i64_e64 s[4:5], s[0:1], 0
+; SI-NEXT:    v_cmp_gt_i64_e64 vcc, s[0:1], 0
 ; SI-NEXT:    s_mov_b32 s0, 0
-; SI-NEXT:    s_and_b64 vcc, exec, s[4:5]
 ; SI-NEXT:    s_cbranch_vccnz .LBB17_2
 ; SI-NEXT:  ; %bb.1: ; %else
 ; SI-NEXT:    s_mov_b32 s0, 1
@@ -1002,9 +995,8 @@ define amdgpu_kernel void @uniform_if_scc_i64_sgt(i64 %cond, ptr addrspace(1) %o
 ; VI:       ; %bb.0: ; %entry
 ; VI-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x24
 ; VI-NEXT:    s_waitcnt lgkmcnt(0)
-; VI-NEXT:    v_cmp_gt_i64_e64 s[4:5], s[0:1], 0
+; VI-NEXT:    v_cmp_gt_i64_e64 vcc, s[0:1], 0
 ; VI-NEXT:    s_mov_b32 s0, 0
-; VI-NEXT:    s_and_b64 vcc, exec, s[4:5]
 ; VI-NEXT:    s_cbranch_vccnz .LBB17_2
 ; VI-NEXT:  ; %bb.1: ; %else
 ; VI-NEXT:    s_mov_b32 s0, 1
diff --git a/llvm/test/CodeGen/AMDGPU/uniform-inside-divergent-cfg.ll b/llvm/test/CodeGen/AMDGPU/uniform-inside-divergent-cfg.ll
index a2e272d9127e4ab..18a3e6c1e8c833e 100644
--- a/llvm/test/CodeGen/AMDGPU/uniform-inside-divergent-cfg.ll
+++ b/llvm/test/CodeGen/AMDGPU/uniform-inside-divergent-cfg.ll
@@ -13,9 +13,8 @@ define amdgpu_kernel void @div_unif_div(ptr addrspace(1) %out, float %ubeta, i32
 ; CHECK-NEXT:    s_and_saveexec_b32 s2, vcc_lo
 ; CHECK-NEXT:    s_cbranch_execz .LBB0_8
 ; CHECK-NEXT:  ; %bb.1: ; %A
-; CHECK-NEXT:    v_cmp_eq_f32_e64 s0, s0, 0
+; CHECK-NEXT:    v_cmp_eq_f32_e64 vcc_lo, s0, 0
 ; CHECK-NEXT:    s_mov_b32 s1, exec_lo
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s0
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB0_7
 ; CHECK-NEXT:  ; %bb.2: ; %B
 ; CHECK-NEXT:    s_mov_b32 s0, exec_lo
@@ -94,12 +93,10 @@ define amdgpu_kernel void @unif_div(ptr addrspace(1) %out, float %u1, float %u2,
 ; CHECK-NEXT:    s_and_saveexec_b32 s2, vcc_lo
 ; CHECK-NEXT:    s_cbranch_execz .LBB1_7
 ; CHECK-NEXT:  ; %bb.1: ; %A
-; CHECK-NEXT:    v_cmp_eq_f32_e64 s0, s0, 0
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s0
+; CHECK-NEXT:    v_cmp_eq_f32_e64 vcc_lo, s0, 0
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB1_4
 ; CHECK-NEXT:  ; %bb.2: ; %B
-; CHECK-NEXT:    v_cmp_eq_f32_e64 s0, s1, 0
-; CHECK-NEXT:    s_and_b32 vcc_lo, exec_lo, s0
+; CHECK-NEXT:    v_cmp_eq_f32_e64 vcc_lo, s1, 0
 ; CHECK-NEXT:    s_cbranch_vccnz .LBB1_5
 ; CHECK-NEXT:  ; %bb.3: ; %C
 ; CHECK-NEXT:    v_add_nc_u32_e32 v1, 7, v0
diff --git a/llvm/test/CodeGen/AMDGPU/unstructured-cfg-def-use-issue.ll b/llvm/test/CodeGen/AMDGPU/unstructured-cfg-def-use-issue.ll
index 8b968f34fbfee54..710c61e16a1ac58 100644
--- a/llvm/test/CodeGen/AMDGPU/unstructured-cfg-def-use-issue.ll
+++ b/llvm/test/CodeGen/AMDGPU/unstructured-cfg-def-use-issue.ll
@@ -26,8 +26,8 @@ define hidden void @widget() #0 {
 ; GCN-NEXT:    s_mov_b64 s[18:19], 0
 ; GCN-NEXT:    s_mov_b64 s[20:21], 0
 ; GCN-NEXT:    s_waitcnt vmcnt(0)
-; GCN-NEXT:    v_cmp_gt_i32_e32 vcc, 21, v0
 ; GCN-NEXT:    v_readfirstlane_b32 s24, v0
+; GCN-NEXT:    v_cmp_gt_i32_e32 vcc, 21, v0
 ; GCN-NEXT:    s_cbranch_vccnz .LBB0_2
 ; GCN-NEXT:  ; %bb.1: ; %bb2
 ; GCN-NEXT:    s_cmp_eq_u32 s24, 21
@@ -83,13 +83,12 @@ define hidden void @widget() #0 {
 ; GCN-NEXT:    s_mov_b64 s[34:35], s[18:19]
 ; GCN-NEXT:    s_swappc_b64 s[30:31], s[16:17]
 ; GCN-NEXT:    buffer_load_dword v31, off, s[0:3], s33 ; 4-byte Folded Reload
-; GCN-NEXT:    v_cmp_nlt_f32_e32 vcc, 0, v0
 ; GCN-NEXT:    v_readlane_b32 s4, v41, 0
 ; GCN-NEXT:    v_readlane_b32 s6, v41, 2
 ; GCN-NEXT:    v_readlane_b32 s8, v41, 4
 ; GCN-NEXT:    v_readlane_b32 s10, v41, 6
+; GCN-NEXT:    v_cmp_nlt_f32_e32 vcc, 0, v0
 ; GCN-NEXT:    s_andn2_b64 s[18:19], s[34:35], exec
-; GCN-NEXT:    s_and_b64 s[20:21], vcc, exec
 ; GCN-NEXT:    v_readlane_b32 s12, v41, 8
 ; GCN-NEXT:    v_readlane_b32 s13, v41, 9
 ; GCN-NEXT:    v_readlane_b32 s14, v41, 10
@@ -99,7 +98,7 @@ define hidden void @widget() #0 {
 ; GCN-NEXT:    v_readlane_b32 s9, v41, 5
 ; GCN-NEXT:    v_readlane_b32 s11, v41, 7
 ; GCN-NEXT:    s_mov_b64 s[16:17], 0
-; GCN-NEXT:    s_or_b64 s[18:19], s[18:19], s[20:21]
+; GCN-NEXT:    s_or_b64 s[18:19], s[18:19], vcc
 ; GCN-NEXT:  .LBB0_6: ; %Flow2
 ; GCN-NEXT:    s_and_saveexec_b64 s[20:21], s[18:19]
 ; GCN-NEXT:    s_xor_b64 s[18:19], exec, s[20:21]
diff --git a/llvm/test/CodeGen/AMDGPU/usubo.ll b/llvm/test/CodeGen/AMDGPU/usubo.ll
index d3579bc2983cb61..b98a80c2e4ca25e 100644
--- a/llvm/test/CodeGen/AMDGPU/usubo.ll
+++ b/llvm/test/CodeGen/AMDGPU/usubo.ll
@@ -1086,8 +1086,8 @@ define amdgpu_kernel void @v_usubo_clamp_bit(ptr addrspace(1) %out, ptr addrspac
 ; SI-NEXT:    buffer_load_dword v2, off, s[8:11], 0
 ; SI-NEXT:    s_mov_b64 s[8:9], 0
 ; SI-NEXT:    s_waitcnt vmcnt(0)
-; SI-NEXT:    v_cmp_eq_u32_e32 vcc, v1, v2
 ; SI-NEXT:    v_sub_i32_e64 v0, s[0:1], v1, v2
+; SI-NEXT:    v_cmp_eq_u32_e32 vcc, v1, v2
 ; SI-NEXT:    s_cbranch_vccnz .LBB9_2
 ; SI-NEXT:  ; %bb.1: ; %if
 ; SI-NEXT:    s_xor_b64 s[8:9], s[0:1], -1
@@ -1118,8 +1118,8 @@ define amdgpu_kernel void @v_usubo_clamp_bit(ptr addrspace(1) %out, ptr addrspac
 ; VI-NEXT:    flat_load_dword v1, v[0:1]
 ; VI-NEXT:    flat_load_dword v2, v[2:3]
 ; VI-NEXT:    s_waitcnt vmcnt(0)
-; VI-NEXT:    v_cmp_eq_u32_e32 vcc, v1, v2
 ; VI-NEXT:    v_sub_u32_e64 v0, s[0:1], v1, v2
+; VI-NEXT:    v_cmp_eq_u32_e32 vcc, v1, v2
 ; VI-NEXT:    s_cbranch_vccnz .LBB9_2
 ; VI-NEXT:  ; %bb.1: ; %if
 ; VI-NEXT:    s_xor_b64 s[2:3], s[0:1], -1
@@ -1144,8 +1144,8 @@ define amdgpu_kernel void @v_usubo_clamp_bit(ptr addrspace(1) %out, ptr addrspac
 ; GFX9-NEXT:    global_load_dword v2, v0, s[12:13]
 ; GFX9-NEXT:    global_load_dword v3, v0, s[14:15]
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
-; GFX9-NEXT:    v_cmp_eq_u32_e32 vcc, v2, v3
 ; GFX9-NEXT:    v_sub_co_u32_e64 v1, s[0:1], v2, v3
+; GFX9-NEXT:    v_cmp_eq_u32_e32 vcc, v2, v3
 ; GFX9-NEXT:    s_cbranch_vccnz .LBB9_2
 ; GFX9-NEXT:  ; %bb.1: ; %if
 ; GFX9-NEXT:    s_xor_b64 s[2:3], s[0:1], -1
diff --git a/llvm/test/CodeGen/AMDGPU/valu-i1.ll b/llvm/test/CodeGen/AMDGPU/valu-i1.ll
index aa38b29b0bf2adf..0c7ce4e8b8c9e50 100644
--- a/llvm/test/CodeGen/AMDGPU/valu-i1.ll
+++ b/llvm/test/CodeGen/AMDGPU/valu-i1.ll
@@ -33,9 +33,8 @@ define amdgpu_kernel void @test_if(i32 %b, ptr addrspace(1) %src, ptr addrspace(
 ; SI-NEXT:  ; %bb.4: ; %LeafBlock
 ; SI-NEXT:    v_cmp_ne_u32_e32 vcc, 1, v0
 ; SI-NEXT:    s_andn2_b64 s[2:3], s[2:3], exec
-; SI-NEXT:    s_and_b64 s[6:7], vcc, exec
 ; SI-NEXT:    s_mov_b64 s[10:11], exec
-; SI-NEXT:    s_or_b64 s[2:3], s[2:3], s[6:7]
+; SI-NEXT:    s_or_b64 s[2:3], s[2:3], vcc
 ; SI-NEXT:  ; %bb.5: ; %Flow8
 ; SI-NEXT:    s_or_b64 exec, exec, s[4:5]
 ; SI-NEXT:    s_and_saveexec_b64 s[4:5], s[2:3]
@@ -355,8 +354,7 @@ define amdgpu_kernel void @multi_vcond_loop(ptr addrspace(1) noalias nocapture %
 ; SI-NEXT:    v_addc_u32_e32 v7, vcc, 0, v7, vcc
 ; SI-NEXT:    v_cmp_ge_i64_e32 vcc, s[6:7], v[0:1]
 ; SI-NEXT:    s_andn2_b64 s[4:5], s[4:5], exec
-; SI-NEXT:    s_and_b64 s[8:9], vcc, exec
-; SI-NEXT:    s_or_b64 s[4:5], s[4:5], s[8:9]
+; SI-NEXT:    s_or_b64 s[4:5], s[4:5], vcc
 ; SI-NEXT:  .LBB5_4: ; %Flow
 ; SI-NEXT:    ; in Loop: Header=BB5_2 Depth=1
 ; SI-NEXT:    s_or_b64 exec, exec, s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/vni8-across-blocks.ll b/llvm/test/CodeGen/AMDGPU/vni8-across-blocks.ll
index 61d2b8458bf83b5..0b0f30c862522ef 100644
--- a/llvm/test/CodeGen/AMDGPU/vni8-across-blocks.ll
+++ b/llvm/test/CodeGen/AMDGPU/vni8-across-blocks.ll
@@ -403,8 +403,7 @@ define amdgpu_kernel void @v8i8_phi_chain(ptr addrspace(1) %src1, ptr addrspace(
 ; GFX942-NEXT:    global_load_dwordx2 v[2:3], v1, s[10:11]
 ; GFX942-NEXT:    v_cmp_gt_u32_e32 vcc, 7, v0
 ; GFX942-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec
-; GFX942-NEXT:    s_and_b64 s[4:5], vcc, exec
-; GFX942-NEXT:    s_or_b64 s[0:1], s[0:1], s[4:5]
+; GFX942-NEXT:    s_or_b64 s[0:1], s[0:1], vcc
 ; GFX942-NEXT:  .LBB8_2: ; %Flow
 ; GFX942-NEXT:    s_or_b64 exec, exec, s[2:3]
 ; GFX942-NEXT:    s_and_saveexec_b64 s[2:3], s[0:1]
@@ -462,9 +461,8 @@ define amdgpu_kernel void @v8i8_phi_zeroinit(ptr addrspace(1) %src1, ptr addrspa
 ; GFX942-NEXT:    s_waitcnt vmcnt(1)
 ; GFX942-NEXT:    v_mov_b32_e32 v4, 0
 ; GFX942-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec
-; GFX942-NEXT:    s_and_b64 s[4:5], vcc, exec
 ; GFX942-NEXT:    v_mov_b32_e32 v5, v4
-; GFX942-NEXT:    s_or_b64 s[0:1], s[0:1], s[4:5]
+; GFX942-NEXT:    s_or_b64 s[0:1], s[0:1], vcc
 ; GFX942-NEXT:  .LBB9_2: ; %Flow
 ; GFX942-NEXT:    s_or_b64 exec, exec, s[2:3]
 ; GFX942-NEXT:    s_and_saveexec_b64 s[2:3], s[0:1]
@@ -533,7 +531,6 @@ define amdgpu_kernel void @v8i8_phi_const(ptr addrspace(1) %src1, ptr addrspace(
 ; GFX942-NEXT:    global_load_dwordx2 v[2:3], v3, s[10:11]
 ; GFX942-NEXT:    v_cmp_gt_u32_e32 vcc, 7, v16
 ; GFX942-NEXT:    s_andn2_b64 s[0:1], s[0:1], exec
-; GFX942-NEXT:    s_and_b64 s[4:5], vcc, exec
 ; GFX942-NEXT:    v_mov_b32_e32 v4, 8
 ; GFX942-NEXT:    v_mov_b32_e32 v5, 7
 ; GFX942-NEXT:    v_mov_b32_e32 v6, 6
@@ -542,7 +539,7 @@ define amdgpu_kernel void @v8i8_phi_const(ptr addrspace(1) %src1, ptr addrspace(
 ; GFX942-NEXT:    v_mov_b32_e32 v8, 3
 ; GFX942-NEXT:    v_mov_b32_e32 v9, 2
 ; GFX942-NEXT:    v_mov_b32_e32 v0, 1
-; GFX942-NEXT:    s_or_b64 s[0:1], s[0:1], s[4:5]
+; GFX942-NEXT:    s_or_b64 s[0:1], s[0:1], vcc
 ; GFX942-NEXT:    s_waitcnt vmcnt(0)
 ; GFX942-NEXT:    v_lshrrev_b32_e32 v15, 24, v3
 ; GFX942-NEXT:    v_lshrrev_b32_e32 v14, 16, v3
@@ -721,22 +718,21 @@ define amdgpu_kernel void @v8i8_multiuse_multiblock(ptr addrspace(1) %src1, ptr
 ; GFX942-NEXT:    s_and_saveexec_b64 s[4:5], vcc
 ; GFX942-NEXT:    s_cbranch_execz .LBB13_2
 ; GFX942-NEXT:  ; %bb.1: ; %bb.1
-; GFX942-NEXT:    s_movk_i32 s6, 0xff00
 ; GFX942-NEXT:    v_mov_b32_e32 v6, 8
-; GFX942-NEXT:    v_and_b32_sdwa v5, v0, s6 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_1 src1_sel:DWORD
+; GFX942-NEXT:    s_movk_i32 s6, 0xff00
 ; GFX942-NEXT:    v_lshlrev_b16_sdwa v6, v6, v0 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
-; GFX942-NEXT:    s_mov_b32 s6, 0x6070504
-; GFX942-NEXT:    v_cmp_gt_u32_e32 vcc, 7, v2
 ; GFX942-NEXT:    v_and_b32_e32 v4, 0xffffff00, v0
+; GFX942-NEXT:    v_and_b32_sdwa v5, v0, s6 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:WORD_1 src1_sel:DWORD
 ; GFX942-NEXT:    v_or_b32_sdwa v6, v0, v6 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:BYTE_0 src1_sel:DWORD
-; GFX942-NEXT:    v_perm_b32 v7, v0, v0, s6
+; GFX942-NEXT:    v_cmp_gt_u32_e32 vcc, 7, v2
 ; GFX942-NEXT:    s_andn2_b64 s[2:3], s[2:3], exec
-; GFX942-NEXT:    s_and_b64 s[6:7], vcc, exec
 ; GFX942-NEXT:    v_mov_b32_e32 v3, 0
 ; GFX942-NEXT:    v_or_b32_sdwa v4, v0, v4 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:BYTE_3 src1_sel:DWORD
 ; GFX942-NEXT:    v_or_b32_sdwa v5, v0, v5 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:BYTE_1 src1_sel:DWORD
+; GFX942-NEXT:    s_mov_b32 s6, 0x6070504
 ; GFX942-NEXT:    v_and_b32_e32 v6, 0xffff, v6
-; GFX942-NEXT:    s_or_b64 s[2:3], s[2:3], s[6:7]
+; GFX942-NEXT:    s_or_b64 s[2:3], s[2:3], vcc
+; GFX942-NEXT:    v_perm_b32 v7, v0, v0, s6
 ; GFX942-NEXT:    v_lshl_or_b32 v5, v5, 16, v6
 ; GFX942-NEXT:    v_lshl_or_b32 v4, v4, 16, v6
 ; GFX942-NEXT:    global_store_dword v3, v0, s[12:13]
diff --git a/llvm/test/CodeGen/AMDGPU/wave32.ll b/llvm/test/CodeGen/AMDGPU/wave32.ll
index 5a28e411cfab4cc..ec7d6346ff7e8e7 100644
--- a/llvm/test/CodeGen/AMDGPU/wave32.ll
+++ b/llvm/test/CodeGen/AMDGPU/wave32.ll
@@ -392,8 +392,7 @@ define amdgpu_kernel void @test_loop_with_if(ptr addrspace(1) %arg) #0 {
 ; GFX1032-NEXT:    global_load_dword v4, v[2:3], off
 ; GFX1032-NEXT:    s_waitcnt vmcnt(0)
 ; GFX1032-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 11, v4
-; GFX1032-NEXT:    s_and_b32 s6, vcc_lo, exec_lo
-; GFX1032-NEXT:    s_or_b32 s4, s4, s6
+; GFX1032-NEXT:    s_or_b32 s4, s4, vcc_lo
 ; GFX1032-NEXT:  .LBB10_4: ; %Flow
 ; GFX1032-NEXT:    ; in Loop: Header=BB10_2 Depth=1
 ; GFX1032-NEXT:    s_or_b32 exec_lo, exec_lo, s5
@@ -454,8 +453,7 @@ define amdgpu_kernel void @test_loop_with_if(ptr addrspace(1) %arg) #0 {
 ; GFX1064-NEXT:    global_load_dword v4, v[2:3], off
 ; GFX1064-NEXT:    s_waitcnt vmcnt(0)
 ; GFX1064-NEXT:    v_cmp_gt_i32_e32 vcc, 11, v4
-; GFX1064-NEXT:    s_and_b64 s[10:11], vcc, exec
-; GFX1064-NEXT:    s_or_b64 s[6:7], s[6:7], s[10:11]
+; GFX1064-NEXT:    s_or_b64 s[6:7], s[6:7], vcc
 ; GFX1064-NEXT:  .LBB10_4: ; %Flow
 ; GFX1064-NEXT:    ; in Loop: Header=BB10_2 Depth=1
 ; GFX1064-NEXT:    s_or_b64 exec, exec, s[8:9]
@@ -527,22 +525,21 @@ define amdgpu_kernel void @test_loop_with_if_else_break(ptr addrspace(1) %arg) #
 ; GFX1032-NEXT:    s_load_dwordx2 s[0:1], s[4:5], 0x24
 ; GFX1032-NEXT:    v_min_u32_e32 v1, 0x100, v0
 ; GFX1032-NEXT:    v_mov_b32_e32 v2, 0
-; GFX1032-NEXT:    s_mov_b32 s3, 0
-; GFX1032-NEXT:    ; implicit-def: $sgpr4
+; GFX1032-NEXT:    s_mov_b32 s4, 0
+; GFX1032-NEXT:    ; implicit-def: $sgpr3
 ; GFX1032-NEXT:    s_branch .LBB11_4
 ; GFX1032-NEXT:  .LBB11_2: ; %bb8
 ; GFX1032-NEXT:    ; in Loop: Header=BB11_4 Depth=1
-; GFX1032-NEXT:    s_add_i32 s3, s3, 1
+; GFX1032-NEXT:    s_add_i32 s4, s4, 1
 ; GFX1032-NEXT:    global_store_dword v2, v0, s[0:1]
-; GFX1032-NEXT:    v_cmp_ge_u32_e32 vcc_lo, s3, v1
+; GFX1032-NEXT:    v_cmp_ge_u32_e32 vcc_lo, s4, v1
 ; GFX1032-NEXT:    s_add_u32 s0, s0, 4
 ; GFX1032-NEXT:    s_addc_u32 s1, s1, 0
-; GFX1032-NEXT:    s_andn2_b32 s4, s4, exec_lo
-; GFX1032-NEXT:    s_and_b32 s5, vcc_lo, exec_lo
-; GFX1032-NEXT:    s_or_b32 s4, s4, s5
+; GFX1032-NEXT:    s_andn2_b32 s3, s3, exec_lo
+; GFX1032-NEXT:    s_or_b32 s3, s3, vcc_lo
 ; GFX1032-NEXT:  .LBB11_3: ; %Flow
 ; GFX1032-NEXT:    ; in Loop: Header=BB11_4 Depth=1
-; GFX1032-NEXT:    s_and_b32 s5, exec_lo, s4
+; GFX1032-NEXT:    s_and_b32 s5, exec_lo, s3
 ; GFX1032-NEXT:    s_or_b32 s2, s5, s2
 ; GFX1032-NEXT:    s_andn2_b32 exec_lo, exec_lo, s2
 ; GFX1032-NEXT:    s_cbranch_execz .LBB11_6
@@ -550,12 +547,12 @@ define amdgpu_kernel void @test_loop_with_if_else_break(ptr addrspace(1) %arg) #
 ; GFX1032-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; GFX1032-NEXT:    s_waitcnt lgkmcnt(0)
 ; GFX1032-NEXT:    global_load_dword v3, v2, s[0:1]
-; GFX1032-NEXT:    s_or_b32 s4, s4, exec_lo
+; GFX1032-NEXT:    s_or_b32 s3, s3, exec_lo
 ; GFX1032-NEXT:    s_waitcnt vmcnt(0)
 ; GFX1032-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 11, v3
 ; GFX1032-NEXT:    s_cbranch_vccz .LBB11_2
 ; GFX1032-NEXT:  ; %bb.5: ; in Loop: Header=BB11_4 Depth=1
-; GFX1032-NEXT:    ; implicit-def: $sgpr3
+; GFX1032-NEXT:    ; implicit-def: $sgpr4
 ; GFX1032-NEXT:    ; implicit-def: $sgpr0_sgpr1
 ; GFX1032-NEXT:    s_branch .LBB11_3
 ; GFX1032-NEXT:  .LBB11_6: ; %.loopexit
@@ -582,8 +579,7 @@ define amdgpu_kernel void @test_loop_with_if_else_break(ptr addrspace(1) %arg) #
 ; GFX1064-NEXT:    s_add_u32 s0, s0, 4
 ; GFX1064-NEXT:    s_addc_u32 s1, s1, 0
 ; GFX1064-NEXT:    s_andn2_b64 s[4:5], s[4:5], exec
-; GFX1064-NEXT:    s_and_b64 s[8:9], vcc, exec
-; GFX1064-NEXT:    s_or_b64 s[4:5], s[4:5], s[8:9]
+; GFX1064-NEXT:    s_or_b64 s[4:5], s[4:5], vcc
 ; GFX1064-NEXT:  .LBB11_3: ; %Flow
 ; GFX1064-NEXT:    ; in Loop: Header=BB11_4 Depth=1
 ; GFX1064-NEXT:    s_and_b64 s[8:9], exec, s[4:5]
@@ -1273,7 +1269,6 @@ define amdgpu_kernel void @test_div_fmas_f32_i1_phi_vcc(ptr addrspace(1) %out, p
 ; GFX1032-NEXT:    global_load_dword v0, v0, s[2:3] glc dlc
 ; GFX1032-NEXT:    s_waitcnt vmcnt(0)
 ; GFX1032-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v0
-; GFX1032-NEXT:    s_and_b32 vcc_lo, vcc_lo, exec_lo
 ; GFX1032-NEXT:  .LBB22_2: ; %exit
 ; GFX1032-NEXT:    s_waitcnt_depctr depctr_vm_vsrc(0)
 ; GFX1032-NEXT:    s_or_b32 exec_lo, exec_lo, s1
@@ -1287,23 +1282,22 @@ define amdgpu_kernel void @test_div_fmas_f32_i1_phi_vcc(ptr addrspace(1) %out, p
 ; GFX1064:       ; %bb.0: ; %entry
 ; GFX1064-NEXT:    s_clause 0x1
 ; GFX1064-NEXT:    s_load_dwordx4 s[8:11], s[4:5], 0x24
-; GFX1064-NEXT:    s_load_dwordx2 s[6:7], s[4:5], 0x34
+; GFX1064-NEXT:    s_load_dwordx2 s[2:3], s[4:5], 0x34
 ; GFX1064-NEXT:    v_lshlrev_b32_e32 v1, 2, v0
 ; GFX1064-NEXT:    v_cmp_eq_u32_e64 s[0:1], 0, v0
 ; GFX1064-NEXT:    s_mov_b64 vcc, 0
 ; GFX1064-NEXT:    s_waitcnt lgkmcnt(0)
 ; GFX1064-NEXT:    global_load_dwordx3 v[1:3], v1, s[10:11]
-; GFX1064-NEXT:    s_and_saveexec_b64 s[2:3], s[0:1]
+; GFX1064-NEXT:    s_and_saveexec_b64 s[4:5], s[0:1]
 ; GFX1064-NEXT:    s_cbranch_execz .LBB22_2
 ; GFX1064-NEXT:  ; %bb.1: ; %bb
 ; GFX1064-NEXT:    v_mov_b32_e32 v0, 0
-; GFX1064-NEXT:    global_load_dword v0, v0, s[6:7] glc dlc
+; GFX1064-NEXT:    global_load_dword v0, v0, s[2:3] glc dlc
 ; GFX1064-NEXT:    s_waitcnt vmcnt(0)
 ; GFX1064-NEXT:    v_cmp_ne_u32_e32 vcc, 0, v0
-; GFX1064-NEXT:    s_and_b64 vcc, vcc, exec
 ; GFX1064-NEXT:  .LBB22_2: ; %exit
 ; GFX1064-NEXT:    s_waitcnt_depctr depctr_vm_vsrc(0)
-; GFX1064-NEXT:    s_or_b64 exec, exec, s[2:3]
+; GFX1064-NEXT:    s_or_b64 exec, exec, s[4:5]
 ; GFX1064-NEXT:    v_mov_b32_e32 v0, 0
 ; GFX1064-NEXT:    s_waitcnt vmcnt(0)
 ; GFX1064-NEXT:    v_div_fmas_f32 v1, v1, v2, v3
@@ -1475,8 +1469,7 @@ define amdgpu_kernel void @test_preserve_condition_undef_flag(float %arg, i32 %a
 ; GFX1032-NEXT:    v_cmp_nlt_f32_e64 s1, s1, 1.0
 ; GFX1032-NEXT:    v_cmp_ngt_f32_e64 s0, s0, 0
 ; GFX1032-NEXT:    s_or_b32 s1, s2, s1
-; GFX1032-NEXT:    s_or_b32 s0, s1, s0
-; GFX1032-NEXT:    s_and_b32 vcc_lo, exec_lo, s0
+; GFX1032-NEXT:    s_or_b32 vcc_lo, s1, s0
 ; GFX1032-NEXT:    s_cbranch_vccnz .LBB26_2
 ; GFX1032-NEXT:  ; %bb.1: ; %bb1
 ; GFX1032-NEXT:    v_mov_b32_e32 v0, 0
@@ -1495,8 +1488,7 @@ define amdgpu_kernel void @test_preserve_condition_undef_flag(float %arg, i32 %a
 ; GFX1064-NEXT:    v_cmp_nlt_f32_e64 s[2:3], s2, 1.0
 ; GFX1064-NEXT:    v_cmp_ngt_f32_e64 s[4:5], s6, 0
 ; GFX1064-NEXT:    s_or_b64 s[0:1], s[0:1], s[2:3]
-; GFX1064-NEXT:    s_or_b64 s[0:1], s[0:1], s[4:5]
-; GFX1064-NEXT:    s_and_b64 vcc, exec, s[0:1]
+; GFX1064-NEXT:    s_or_b64 vcc, s[0:1], s[4:5]
 ; GFX1064-NEXT:    s_cbranch_vccnz .LBB26_2
 ; GFX1064-NEXT:  ; %bb.1: ; %bb1
 ; GFX1064-NEXT:    v_mov_b32_e32 v0, 0
@@ -1552,8 +1544,7 @@ define amdgpu_kernel void @test_invert_true_phi_cond_break_loop(i32 %arg) #0 {
 ; GFX1032-NEXT:    s_waitcnt vmcnt(0)
 ; GFX1032-NEXT:    s_andn2_b32 s1, s1, exec_lo
 ; GFX1032-NEXT:    v_cmp_ge_i32_e32 vcc_lo, v0, v1
-; GFX1032-NEXT:    s_and_b32 s3, vcc_lo, exec_lo
-; GFX1032-NEXT:    s_or_b32 s1, s1, s3
+; GFX1032-NEXT:    s_or_b32 s1, s1, vcc_lo
 ; GFX1032-NEXT:    s_branch .LBB27_1
 ; GFX1032-NEXT:  .LBB27_4: ; %bb9
 ; GFX1032-NEXT:    s_or_b32 exec_lo, exec_lo, s0
@@ -1590,8 +1581,7 @@ define amdgpu_kernel void @test_invert_true_phi_cond_break_loop(i32 %arg) #0 {
 ; GFX1064-NEXT:    s_waitcnt vmcnt(0)
 ; GFX1064-NEXT:    s_andn2_b64 s[2:3], s[2:3], exec
 ; GFX1064-NEXT:    v_cmp_ge_i32_e32 vcc, v0, v1
-; GFX1064-NEXT:    s_and_b64 s[6:7], vcc, exec
-; GFX1064-NEXT:    s_or_b64 s[2:3], s[2:3], s[6:7]
+; GFX1064-NEXT:    s_or_b64 s[2:3], s[2:3], vcc
 ; GFX1064-NEXT:    s_branch .LBB27_1
 ; GFX1064-NEXT:  .LBB27_4: ; %bb9
 ; GFX1064-NEXT:    s_or_b64 exec, exec, s[0:1]
@@ -1815,12 +1805,12 @@ define amdgpu_ps <4 x float> @test_loop_vcc(<4 x float> %in) #0 {
 ; GFX1032-NEXT:    s_cbranch_scc0 .LBB33_5
 ; GFX1032-NEXT:  .LBB33_3: ; %loop
 ; GFX1032-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX1032-NEXT:    v_cmp_lt_f32_e32 vcc_lo, 0x40e00000, v8
 ; GFX1032-NEXT:    s_waitcnt vmcnt(0)
 ; GFX1032-NEXT:    v_mov_b32_e32 v7, v3
 ; GFX1032-NEXT:    v_mov_b32_e32 v6, v2
 ; GFX1032-NEXT:    v_mov_b32_e32 v5, v1
 ; GFX1032-NEXT:    v_mov_b32_e32 v4, v0
+; GFX1032-NEXT:    v_cmp_lt_f32_e32 vcc_lo, 0x40e00000, v8
 ; GFX1032-NEXT:    s_cbranch_vccz .LBB33_1
 ; GFX1032-NEXT:  ; %bb.4: ; in Loop: Header=BB33_3 Depth=1
 ; GFX1032-NEXT:    s_mov_b32 s1, -1
@@ -2257,8 +2247,7 @@ define amdgpu_kernel void @test_vccnz_ifcvt_triangle64(ptr addrspace(1) %out, pt
 ; GFX1032-NEXT:    s_waitcnt lgkmcnt(0)
 ; GFX1032-NEXT:    s_load_dwordx2 s[2:3], s[2:3], 0x0
 ; GFX1032-NEXT:    s_waitcnt lgkmcnt(0)
-; GFX1032-NEXT:    v_cmp_neq_f64_e64 s4, s[2:3], 1.0
-; GFX1032-NEXT:    s_and_b32 vcc_lo, exec_lo, s4
+; GFX1032-NEXT:    v_cmp_neq_f64_e64 vcc_lo, s[2:3], 1.0
 ; GFX1032-NEXT:    s_cbranch_vccnz .LBB43_2
 ; GFX1032-NEXT:  ; %bb.1: ; %if
 ; GFX1032-NEXT:    v_add_f64 v[0:1], s[2:3], s[2:3]
@@ -2277,8 +2266,7 @@ define amdgpu_kernel void @test_vccnz_ifcvt_triangle64(ptr addrspace(1) %out, pt
 ; GFX1064-NEXT:    s_waitcnt lgkmcnt(0)
 ; GFX1064-NEXT:    s_load_dwordx2 s[2:3], s[2:3], 0x0
 ; GFX1064-NEXT:    s_waitcnt lgkmcnt(0)
-; GFX1064-NEXT:    v_cmp_neq_f64_e64 s[4:5], s[2:3], 1.0
-; GFX1064-NEXT:    s_and_b64 vcc, exec, s[4:5]
+; GFX1064-NEXT:    v_cmp_neq_f64_e64 vcc, s[2:3], 1.0
 ; GFX1064-NEXT:    s_cbranch_vccnz .LBB43_2
 ; GFX1064-NEXT:  ; %bb.1: ; %if
 ; GFX1064-NEXT:    v_add_f64 v[0:1], s[2:3], s[2:3]
diff --git a/llvm/test/CodeGen/AMDGPU/while-break.ll b/llvm/test/CodeGen/AMDGPU/while-break.ll
index ed1655015722598..18534ac06078cbe 100644
--- a/llvm/test/CodeGen/AMDGPU/while-break.ll
+++ b/llvm/test/CodeGen/AMDGPU/while-break.ll
@@ -23,8 +23,7 @@ define amdgpu_ps float @while_break(i32 %z, float %v, i32 %x, i32 %y) #0 {
 ; GCN-NEXT:    s_xor_b32 s3, exec_lo, s3
 ; GCN-NEXT:  ; %bb.3: ; %else
 ; GCN-NEXT:    ; in Loop: Header=BB0_2 Depth=1
-; GCN-NEXT:    v_cmp_lt_i32_e32 vcc_lo, s1, v3
-; GCN-NEXT:    s_and_b32 s2, vcc_lo, exec_lo
+; GCN-NEXT:    v_cmp_lt_i32_e64 s2, s1, v3
 ; GCN-NEXT:  ; %bb.4: ; %Flow
 ; GCN-NEXT:    ; in Loop: Header=BB0_2 Depth=1
 ; GCN-NEXT:    s_andn2_saveexec_b32 s3, s3
@@ -107,8 +106,7 @@ define amdgpu_ps float @while_break2(i32 %z, float %v, i32 %x, i32 %y) #0 {
 ; GCN-NEXT:    ; in Loop: Header=BB1_2 Depth=1
 ; GCN-NEXT:    v_cmp_lt_i32_e32 vcc_lo, s1, v3
 ; GCN-NEXT:    s_andn2_b32 s2, s2, exec_lo
-; GCN-NEXT:    s_and_b32 s4, vcc_lo, exec_lo
-; GCN-NEXT:    s_or_b32 s2, s2, s4
+; GCN-NEXT:    s_or_b32 s2, s2, vcc_lo
 ; GCN-NEXT:  ; %bb.6: ; %Flow1
 ; GCN-NEXT:    ; in Loop: Header=BB1_2 Depth=1
 ; GCN-NEXT:    s_or_b32 exec_lo, exec_lo, s3
@@ -182,8 +180,7 @@ define amdgpu_ps < 2 x float> @while_break_two_chains_of_phi(float %v, i32 %x, i
 ; GCN-NEXT:    v_add_co_ci_u32_e32 v7, vcc_lo, s7, v5, vcc_lo
 ; GCN-NEXT:    v_cmp_lt_i32_e32 vcc_lo, s0, v2
 ; GCN-NEXT:    global_load_dword v0, v[6:7], off
-; GCN-NEXT:    s_and_b32 s3, vcc_lo, exec_lo
-; GCN-NEXT:    s_or_b32 s3, s1, s3
+; GCN-NEXT:    s_or_b32 s3, s1, vcc_lo
 ; GCN-NEXT:    s_waitcnt vmcnt(0)
 ; GCN-NEXT:    v_add_f32_e32 v6, 1.0, v0
 ; GCN-NEXT:    v_mov_b32_e32 v0, v6
diff --git a/llvm/test/CodeGen/AMDGPU/wqm.ll b/llvm/test/CodeGen/AMDGPU/wqm.ll
index 3161821dd69a2ef..dda0e30bc386f3f 100644
--- a/llvm/test/CodeGen/AMDGPU/wqm.ll
+++ b/llvm/test/CodeGen/AMDGPU/wqm.ll
@@ -1937,12 +1937,12 @@ define amdgpu_ps <4 x float> @test_loop_vcc(<4 x float> %in) nounwind {
 ; GFX9-W64-NEXT:    s_cbranch_scc0 .LBB35_5
 ; GFX9-W64-NEXT:  .LBB35_3: ; %loop
 ; GFX9-W64-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX9-W64-NEXT:    v_cmp_lt_f32_e32 vcc, s4, v8
 ; GFX9-W64-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9-W64-NEXT:    v_mov_b32_e32 v7, v3
 ; GFX9-W64-NEXT:    v_mov_b32_e32 v6, v2
 ; GFX9-W64-NEXT:    v_mov_b32_e32 v5, v1
 ; GFX9-W64-NEXT:    v_mov_b32_e32 v4, v0
+; GFX9-W64-NEXT:    v_cmp_lt_f32_e32 vcc, s4, v8
 ; GFX9-W64-NEXT:    s_cbranch_vccz .LBB35_1
 ; GFX9-W64-NEXT:  ; %bb.4: ; in Loop: Header=BB35_3 Depth=1
 ; GFX9-W64-NEXT:    s_mov_b64 s[2:3], -1
@@ -1981,12 +1981,12 @@ define amdgpu_ps <4 x float> @test_loop_vcc(<4 x float> %in) nounwind {
 ; GFX10-W32-NEXT:    s_cbranch_scc0 .LBB35_5
 ; GFX10-W32-NEXT:  .LBB35_3: ; %loop
 ; GFX10-W32-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX10-W32-NEXT:    v_cmp_lt_f32_e32 vcc_lo, 0x40e00000, v8
 ; GFX10-W32-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-W32-NEXT:    v_mov_b32_e32 v7, v3
 ; GFX10-W32-NEXT:    v_mov_b32_e32 v6, v2
 ; GFX10-W32-NEXT:    v_mov_b32_e32 v5, v1
 ; GFX10-W32-NEXT:    v_mov_b32_e32 v4, v0
+; GFX10-W32-NEXT:    v_cmp_lt_f32_e32 vcc_lo, 0x40e00000, v8
 ; GFX10-W32-NEXT:    s_cbranch_vccz .LBB35_1
 ; GFX10-W32-NEXT:  ; %bb.4: ; in Loop: Header=BB35_3 Depth=1
 ; GFX10-W32-NEXT:    s_mov_b32 s1, -1
diff --git a/llvm/test/CodeGen/AMDGPU/xor.ll b/llvm/test/CodeGen/AMDGPU/xor.ll
index 9a93d54fb95baee..52ae751798ae35b 100644
--- a/llvm/test/CodeGen/AMDGPU/xor.ll
+++ b/llvm/test/CodeGen/AMDGPU/xor.ll
@@ -550,8 +550,7 @@ define amdgpu_kernel void @xor_cf(ptr addrspace(1) %out, ptr addrspace(1) %in, i
 ; SI-NEXT:    s_load_dwordx8 s[0:7], s[4:5], 0x9
 ; SI-NEXT:    s_mov_b64 s[8:9], 0
 ; SI-NEXT:    s_waitcnt lgkmcnt(0)
-; SI-NEXT:    v_cmp_ne_u64_e64 s[10:11], s[4:5], 0
-; SI-NEXT:    s_and_b64 vcc, exec, s[10:11]
+; SI-NEXT:    v_cmp_ne_u64_e64 vcc, s[4:5], 0
 ; SI-NEXT:    s_cbranch_vccz .LBB14_2
 ; SI-NEXT:  ; %bb.1: ; %else
 ; SI-NEXT:    s_mov_b32 s15, 0xf000



More information about the llvm-commits mailing list