[llvm-branch-commits] [llvm] [AMDGPU] Form VOPD3 pairs with pair-local literal moves (PR #223431)

Shilei Tian via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Wed Sep 16 16:54:17 PDT 2026


https://github.com/shiltian updated https://github.com/llvm/llvm-project/pull/223431

>From d64b699e926fd2796d27d058a2f1c56a5bceccc6 Mon Sep 17 00:00:00 2001
From: Shilei Tian <i at tianshilei.me>
Date: Mon, 14 Sep 2026 10:36:18 -0400
Subject: [PATCH] [AMDGPU] Form VOPD3 pairs with pair-local literal moves

This PR lets GCNCreateVOPD form a VOPD3 pair when its components use one
distinct non-inline constant. VOPD3 cannot encode literal operands, but src0
can read scalar registers, so we move the value to a free SGPR. If both
components use the same value, one move serves both.

We reject pairs that need two values because two moves add more instructions
than one fusion removes. We also reject functions without tracked liveness and
functions optimized for size.

We use one reverse liveness walk to find an SGPR that is free over each
pair-local range. We exclude reserved registers and VCC. Disjoint selected
pairs can reuse the same SGPR, and each accepted pair adds at most one
S_MOV_B32 for the one instruction removed by fusion.

When overlapping candidates form the same number of pairs, we prefer the set
that needs fewer scalar moves. We keep pair count as the primary objective.

The post-RA scheduler uses the same matcher policy, so it does not cluster a
two-value pair that the create pass cannot build.
---
 llvm/lib/Target/AMDGPU/GCNCreateVOPD.cpp      | 225 ++++-
 llvm/lib/Target/AMDGPU/GCNVOPDUtils.cpp       | 104 ++-
 llvm/lib/Target/AMDGPU/GCNVOPDUtils.h         |  31 +-
 .../CodeGen/AMDGPU/GlobalISel/fadd.bf16.ll    |  10 +-
 .../AMDGPU/GlobalISel/fcanonicalize.bf16.ll   |   2 +-
 .../CodeGen/AMDGPU/GlobalISel/fma.bf16.ll     |  25 +-
 .../CodeGen/AMDGPU/GlobalISel/fmul.bf16.ll    |  10 +-
 .../CodeGen/AMDGPU/GlobalISel/fptrunc.bf16.ll |  14 +-
 .../test/CodeGen/AMDGPU/GlobalISel/fptrunc.ll |   6 +-
 .../AMDGPU/atomic_optimizations_buffer.ll     |  83 +-
 .../AMDGPU/atomic_optimizations_raw_buffer.ll |  83 +-
 .../atomic_optimizations_struct_buffer.ll     |  83 +-
 llvm/test/CodeGen/AMDGPU/bf16.ll              | 406 +++++----
 .../AMDGPU/branch-relaxation-gfx1250.ll       |  12 +-
 .../CodeGen/AMDGPU/calling-conventions.ll     |  78 +-
 .../test/CodeGen/AMDGPU/carryout-selection.ll |  88 +-
 llvm/test/CodeGen/AMDGPU/ds_read2-gfx1250.ll  |  36 +-
 llvm/test/CodeGen/AMDGPU/ds_write2.ll         |  25 +-
 .../test/CodeGen/AMDGPU/fcanonicalize.bf16.ll |  22 +-
 llvm/test/CodeGen/AMDGPU/fcanonicalize.ll     |   5 +-
 llvm/test/CodeGen/AMDGPU/flat-saddr-load.ll   |   5 +-
 .../float-to-arbitrary-fp-fp8-f16-hw.ll       | 202 +++--
 .../AMDGPU/float-to-arbitrary-fp-fp8-hw.ll    | 610 +++++++-------
 .../AMDGPU/float-to-arbitrary-fp-widen.ll     | 101 +--
 .../CodeGen/AMDGPU/integer-mad-patterns.ll    |  45 +-
 .../AMDGPU/llvm.amdgcn.av.load.b128.ll        |  40 +-
 .../CodeGen/AMDGPU/llvm.amdgcn.permlane.ll    | 776 +++++++++---------
 .../llvm.amdgcn.ptr.s.buffer.load-gfx12.ll    |  10 +-
 .../llvm.amdgcn.raw.atomic.buffer.load.ll     | 119 +--
 .../llvm.amdgcn.raw.ptr.atomic.buffer.load.ll | 119 +--
 .../CodeGen/AMDGPU/llvm.amdgcn.rcp.bf16.ll    |  34 +-
 .../llvm.amdgcn.sched.group.barrier.gfx12.ll  |  14 +-
 .../llvm.amdgcn.struct.atomic.buffer.load.ll  | 144 +---
 ...vm.amdgcn.struct.ptr.atomic.buffer.load.ll | 144 +---
 llvm/test/CodeGen/AMDGPU/llvm.sqrt.bf16.ll    |   9 +-
 llvm/test/CodeGen/AMDGPU/load-constant-i1.ll  |  53 +-
 .../test/CodeGen/AMDGPU/loop-prefetch-data.ll |   8 +-
 llvm/test/CodeGen/AMDGPU/mad-mix-bf16.ll      |  78 +-
 llvm/test/CodeGen/AMDGPU/mad-mix-lo-bf16.ll   | 190 +++--
 llvm/test/CodeGen/AMDGPU/mul.ll               |  12 +-
 llvm/test/CodeGen/AMDGPU/packed-fp32.ll       |  36 +-
 llvm/test/CodeGen/AMDGPU/packed-fp64.ll       |  35 +-
 llvm/test/CodeGen/AMDGPU/packed-u64.ll        |  12 +-
 .../promote-constOffset-to-imm-gfx12.ll       |  23 +-
 .../promote-constOffset-to-imm-gfx12.mir      |   4 +-
 .../AMDGPU/reassoc-mul-add-1-to-mad.ll        |  37 +-
 llvm/test/CodeGen/AMDGPU/v_ashr_pk.ll         |  20 +-
 .../CodeGen/AMDGPU/vopd-combine-gfx1250.mir   | 200 ++++-
 llvm/test/CodeGen/AMDGPU/vopd-combine.mir     |  51 ++
 llvm/test/CodeGen/AMDGPU/vopd3-imm-fold.ll    | 373 +++++++++
 llvm/test/CodeGen/AMDGPU/vopd3-imm-fold.mir   | 433 ++++++++++
 .../waitcnt-loop-ds-prefetch-flushed.ll       |   6 +-
 .../waitcnt-loop-ds-prefetch-pattern.ll       |   7 +-
 53 files changed, 3258 insertions(+), 2040 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/vopd3-imm-fold.ll
 create mode 100644 llvm/test/CodeGen/AMDGPU/vopd3-imm-fold.mir

diff --git a/llvm/lib/Target/AMDGPU/GCNCreateVOPD.cpp b/llvm/lib/Target/AMDGPU/GCNCreateVOPD.cpp
index 015e862bab776..2ca4fa50f7f29 100644
--- a/llvm/lib/Target/AMDGPU/GCNCreateVOPD.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNCreateVOPD.cpp
@@ -7,12 +7,19 @@
 //===----------------------------------------------------------------------===//
 //
 /// \file
-/// Combine VALU pairs into VOPD instructions
-/// Only works on wave32
-/// Has register requirements, we reject creating VOPD if the requirements are
-/// not met.
-/// shouldCombineVOPD mutator in postRA machine scheduler puts candidate
-/// instructions for VOPD back-to-back
+/// Form VOPD instructions from adjacent VALU operations on wave32. The post-RA
+/// scheduler puts likely component pairs next to each other. This pass checks
+/// their final physical-register constraints and selects a non-overlapping set.
+///
+/// VOPD3 components cannot encode literal operands. When all non-inline
+/// immediates in a pair have the same 32-bit value, the pass can materialize
+/// that value in an SGPR which is free over the pair. The move and its register
+/// stay pair-local, so fusion adds at most one move and does not extend
+/// register pressure across pairs.
+///
+/// The pass considers every adjacent candidate. It first maximizes the number
+/// of pairs, then minimizes scalar moves among equal-size matchings. The
+/// earlier candidate wins an exact tie.
 ///
 //
 //===----------------------------------------------------------------------===//
@@ -22,24 +29,200 @@
 #include "GCNVOPDUtils.h"
 #include "SIInstrInfo.h"
 #include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/ADT/STLExtras.h"
+#include "llvm/ADT/SmallBitVector.h"
 #include "llvm/ADT/Statistic.h"
+#include "llvm/CodeGen/LiveRegUnits.h"
 #include "llvm/CodeGen/MachineBasicBlock.h"
 #include "llvm/CodeGen/MachineInstr.h"
 #include "llvm/CodeGen/MachineOperand.h"
 #include "llvm/CodeGen/MachinePassManager.h"
+#include "llvm/CodeGen/MachineRegisterInfo.h"
 #include "llvm/Support/Debug.h"
 
 #define DEBUG_TYPE "gcn-create-vopd"
 STATISTIC(NumVOPDCreated, "Number of VOPD Insts Created.");
+STATISTIC(NumLiteralsMaterialized,
+          "Number of immediates moved into a scalar register to allow VOPD3 "
+          "pairing.");
+STATISTIC(NumCandidateEdgesWithoutFreeSGPR,
+          "Number of VOPD3 candidate edges skipped because no scalar register "
+          "was free for their immediate.");
 
 using namespace llvm;
 
 namespace {
 
+struct VOPDCandidate {
+  VOPDMatchInfo Match;
+  /// The register for Match.LiteralFixups, or a null register if none is free.
+  Register MaterializationReg;
+
+  bool needsMaterialization() const { return !Match.LiteralFixups.empty(); }
+
+  bool isFeasible() const {
+    return !needsMaterialization() || MaterializationReg;
+  }
+};
+
+} // namespace
+
+/// Add everything the instructions in [\p Begin, \p RangeEnd] touch to
+/// \p Live, which already holds what is live after \p RangeEnd.
+static void addRangeUses(LiveRegUnits &Live, MachineBasicBlock::iterator Begin,
+                         MachineInstr &RangeEnd) {
+  MachineBasicBlock::iterator After =
+      std::next(MachineBasicBlock::iterator(&RangeEnd));
+  for (MachineInstr &MI : make_range(Begin, After)) {
+    if (!MI.isDebugInstr())
+      Live.accumulate(MI);
+  }
+}
+
+/// Return a scalar register which every fixup can read and which no value in
+/// \p Live occupies, or a null register. Low registers are preferred, because
+/// those are most likely in use already, so the function's register count does
+/// not grow.
+static Register takeFreeSGPR(const GCNSubtarget &ST,
+                             const MachineRegisterInfo &MRI,
+                             const LiveRegUnits &Live,
+                             ArrayRef<VOPDLiteralFixup> Fixups) {
+  assert(!Fixups.empty());
+  const SIRegisterInfo *TRI = ST.getRegisterInfo();
+  for (MCPhysReg Reg : AMDGPU::SGPR_32RegClass) {
+    // SGPR_32 also holds the halves of VCC. Writing those changes VCCZ,
+    // which is not modelled by \p Live, so a free half is not safe to use.
+    if (MRI.isReserved(Reg) || TRI->isSubRegisterEq(AMDGPU::VCC, Reg) ||
+        !Live.available(Reg))
+      continue;
+    if (!all_of(Fixups, [Reg](const VOPDLiteralFixup &Fixup) {
+          return Fixup.SlotRC->contains(Reg);
+        }))
+      continue;
+    return Reg;
+  }
+  return Register();
+}
+
+namespace {
+
 class GCNCreateVOPD {
 public:
   const GCNSubtarget *ST = nullptr;
 
+  void
+  assignMaterializationRegisters(MachineBasicBlock &MBB,
+                                 MutableArrayRef<VOPDCandidate> Candidates) {
+    auto Candidate = Candidates.rbegin();
+    auto SkipPlainCandidates = [&] {
+      while (Candidate != Candidates.rend() &&
+             !Candidate->needsMaterialization())
+        ++Candidate;
+    };
+    SkipPlainCandidates();
+    if (Candidate == Candidates.rend())
+      return;
+
+    const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
+    LiveRegUnits Walk(*ST->getRegisterInfo());
+    Walk.addLiveOuts(MBB);
+
+    // Before stepping over an instruction, Walk holds what is live immediately
+    // after it. This answers every pair-local range in one backward walk.
+    for (MachineInstr &MI : reverse(MBB)) {
+      if (Candidate != Candidates.rend() &&
+          Candidate->Match.InOrder[1] == &MI) {
+        LiveRegUnits RangeLive = Walk;
+        addRangeUses(RangeLive, Candidate->Match.InOrder[0]->getIterator(),
+                     *Candidate->Match.InOrder[1]);
+        Candidate->MaterializationReg =
+            takeFreeSGPR(*ST, MRI, RangeLive, Candidate->Match.LiteralFixups);
+        ++Candidate;
+        SkipPlainCandidates();
+      }
+      if (!MI.isDebugInstr())
+        Walk.stepBackward(MI);
+    }
+    assert(Candidate == Candidates.rend() &&
+           "every candidate range must end in this block");
+  }
+
+  static SmallVector<VOPDCandidate *, 8>
+  selectCandidates(MutableArrayRef<VOPDCandidate> Candidates) {
+    struct Score {
+      unsigned NumPairs = 0;
+      unsigned NumMoves = 0;
+    };
+
+    const size_t NumCandidates = Candidates.size();
+    SmallVector<Score, 8> Best(NumCandidates + 1);
+    SmallBitVector Take(NumCandidates);
+    auto NextNonOverlapping = [&](size_t I) {
+      size_t Next = I + 1;
+      if (Next != NumCandidates &&
+          Candidates[Next].Match.InOrder[0] == Candidates[I].Match.InOrder[1])
+        ++Next;
+      return Next;
+    };
+
+    // Maximize the number of pairs, then minimize the moves they need. Taking
+    // the current edge on an exact tie preserves the old left-to-right choice.
+    for (size_t I = NumCandidates; I-- != 0;) {
+      Best[I] = Best[I + 1];
+      if (!Candidates[I].isFeasible()) {
+        ++NumCandidateEdgesWithoutFreeSGPR;
+        continue;
+      }
+
+      Score With = Best[NextNonOverlapping(I)];
+      ++With.NumPairs;
+      With.NumMoves += Candidates[I].needsMaterialization();
+      if (With.NumPairs > Best[I].NumPairs ||
+          (With.NumPairs == Best[I].NumPairs &&
+           With.NumMoves <= Best[I].NumMoves)) {
+        Best[I] = With;
+        Take.set(I);
+      }
+    }
+
+    SmallVector<VOPDCandidate *, 8> Selected;
+    for (size_t I = 0; I != NumCandidates;) {
+      if (!Take[I]) {
+        ++I;
+        continue;
+      }
+      Selected.push_back(&Candidates[I]);
+      I = NextNonOverlapping(I);
+    }
+    return Selected;
+  }
+
+  void materializeLiteral(const SIInstrInfo &TII, VOPDCandidate &Candidate) {
+    if (!Candidate.needsMaterialization())
+      return;
+
+    ArrayRef<VOPDLiteralFixup> Fixups = Candidate.Match.LiteralFixups;
+    assert(Candidate.MaterializationReg);
+    assert(all_of(Fixups,
+                  [Imm = Fixups.front().Imm](const VOPDLiteralFixup &Fixup) {
+                    return Fixup.Imm == Imm;
+                  }));
+
+    MachineInstr *InsertPt = Candidate.Match.InOrder[0];
+    BuildMI(*InsertPt->getParent(), InsertPt, DebugLoc(),
+            TII.get(AMDGPU::S_MOV_B32), Candidate.MaterializationReg)
+        .addImm(Fixups.front().Imm);
+    ++NumLiteralsMaterialized;
+
+    for (const VOPDLiteralFixup &Fixup : Fixups) {
+      MachineInstr *MI = Fixup.CompIdx == AMDGPU::VOPD::X
+                             ? Candidate.Match.getMIX()
+                             : Candidate.Match.getMIY();
+      MI->getOperand(Fixup.OpIdx)
+          .ChangeToRegister(Candidate.MaterializationReg, /*isDef=*/false);
+    }
+  }
+
   bool doReplace(const SIInstrInfo *SII, VOPDMatchInfo &Match) {
     MachineInstr *MIX = Match.getMIX();
     MachineInstr *MIY = Match.getMIY();
@@ -123,38 +306,30 @@ class GCNCreateVOPD {
     const SIInstrInfo *SII = ST->getInstrInfo();
     bool Changed = false;
 
-    SmallVector<VOPDMatchInfo> ReplaceCandidates;
-
-    for (auto &MBB : MF) {
+    for (MachineBasicBlock &MBB : MF) {
+      SmallVector<VOPDCandidate, 8> Candidates;
       auto MII = MBB.begin(), E = MBB.end();
       while (MII != E) {
-        auto *FirstMI = &*MII;
+        MachineInstr *FirstMI = &*MII;
         MII = next_nodbg(MII, MBB.end());
         if (MII == MBB.end())
           break;
         if (FirstMI->isDebugInstr())
           continue;
-        auto *SecondMI = &*MII;
+        MachineInstr *SecondMI = &*MII;
 
         if (std::optional<VOPDMatchInfo> Match =
                 tryMatchVOPDPair(*SII, *FirstMI, *SecondMI))
-          ReplaceCandidates.push_back(std::move(*Match));
+          Candidates.push_back({std::move(*Match), Register()});
       }
-    }
 
-    SmallVector<VOPDMatchInfo *> Selected;
-    // An instruction can be the second MI of one candidate and the first MI of
-    // the next. Remember the second MI of the last selected candidate to reject
-    // that overlap.
-    MachineInstr *LastSelectedSecond = nullptr;
-    for (VOPDMatchInfo &Match : ReplaceCandidates) {
-      if (Match.InOrder[0] == LastSelectedSecond)
-        continue;
-      Selected.push_back(&Match);
-      LastSelectedSecond = Match.InOrder[1];
+      assignMaterializationRegisters(MBB, Candidates);
+      SmallVector<VOPDCandidate *, 8> Selected = selectCandidates(Candidates);
+      for (VOPDCandidate *Candidate : Selected) {
+        materializeLiteral(*SII, *Candidate);
+        Changed |= doReplace(SII, Candidate->Match);
+      }
     }
-    for (VOPDMatchInfo *Match : Selected)
-      Changed |= doReplace(SII, *Match);
 
     return Changed;
   }
diff --git a/llvm/lib/Target/AMDGPU/GCNVOPDUtils.cpp b/llvm/lib/Target/AMDGPU/GCNVOPDUtils.cpp
index 4f0e84584988d..63b28bf4ad645 100644
--- a/llvm/lib/Target/AMDGPU/GCNVOPDUtils.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNVOPDUtils.cpp
@@ -19,6 +19,7 @@
 #include "Utils/AMDGPUBaseInfo.h"
 #include "llvm/ADT/SmallVector.h"
 #include "llvm/CodeGen/MachineBasicBlock.h"
+#include "llvm/CodeGen/MachineFunction.h"
 #include "llvm/CodeGen/MachineInstr.h"
 #include "llvm/CodeGen/MachineOperand.h"
 #include "llvm/CodeGen/MachineRegisterInfo.h"
@@ -32,12 +33,13 @@ using namespace llvm;
 
 #define DEBUG_TYPE "gcn-vopd-utils"
 
-// Check if physical register from src<SrcIdx> operand of MI<CompIdx> matches
-// register class constraints in corresponding VOPDOpc operand with name
-// src/vsrc<SrcIdx><CompIdx>.
-static bool isValidVOPDSrc(const SIInstrInfo &TII, int VOPDOpc,
-                           unsigned CompIdx, unsigned SrcIdx,
-                           Register PhysSrcReg) {
+// Return the register class of the VOPDOpc operand named
+// src/vsrc<SrcIdx><CompIdx>, which is the slot src<SrcIdx> of MI<CompIdx> maps
+// to.
+static const TargetRegisterClass *getVOPDSrcRegClass(const SIInstrInfo &TII,
+                                                     int VOPDOpc,
+                                                     unsigned CompIdx,
+                                                     unsigned SrcIdx) {
   using namespace AMDGPU;
   int OpIdx = -1;
   const bool IsX = CompIdx == VOPD::X;
@@ -58,7 +60,17 @@ static bool isValidVOPDSrc(const SIInstrInfo &TII, int VOPDOpc,
   }
 
   assert(OpIdx != -1);
-  return TII.getRegClass(TII.get(VOPDOpc), OpIdx)->contains(PhysSrcReg);
+  return TII.getRegClass(TII.get(VOPDOpc), OpIdx);
+}
+
+// Check if physical register from src<SrcIdx> operand of MI<CompIdx> matches
+// register class constraints in corresponding VOPDOpc operand with name
+// src/vsrc<SrcIdx><CompIdx>.
+static bool isValidVOPDSrc(const SIInstrInfo &TII, int VOPDOpc,
+                           unsigned CompIdx, unsigned SrcIdx,
+                           Register PhysSrcReg) {
+  return getVOPDSrcRegClass(TII, VOPDOpc, CompIdx, SrcIdx)
+      ->contains(PhysSrcReg);
 }
 
 static const MachineOperand &getNamedOp(const MachineInstr &MI,
@@ -93,10 +105,18 @@ static bool canMapVOP3PToVOPD(const MachineInstr &MI) {
          getNamedOp(MI, AMDGPU::OpName::src2).getReg();
 }
 
-bool llvm::checkVOPDRegConstraints(const SIInstrInfo &TII,
-                                   const MachineInstr &MIX,
-                                   const MachineInstr &MIY, bool IsVOPD3,
-                                   bool AllowSameVGPR) {
+static bool canMaterializeVOPDLiterals(const MachineFunction &MF) {
+  // A free register cannot be found without liveness. A move also makes the
+  // code longer, so a function which asked for small code keeps its literals.
+  return MF.getProperties().hasTracksLiveness() &&
+         !MF.getFunction().hasOptSize();
+}
+
+static bool
+checkVOPDRegConstraints(const SIInstrInfo &TII, const MachineInstr &MIX,
+                        const MachineInstr &MIY, bool IsVOPD3,
+                        bool AllowSameVGPR,
+                        SmallVectorImpl<VOPDLiteralFixup> &LiteralFixups) {
   namespace VOPD = AMDGPU::VOPD;
 
   const MachineFunction *MF = MIX.getMF();
@@ -110,6 +130,10 @@ bool llvm::checkVOPDRegConstraints(const SIInstrInfo &TII,
   if (TII.isDPP(MIX) || TII.isDPP(MIY))
     return false;
 
+  // Collected here and handed over only on success, so a failed check cannot
+  // leave anything behind.
+  SmallVector<VOPDLiteralFixup, 2> Fixups;
+
   const SIRegisterInfo *TRI = ST.getRegisterInfo();
   const MachineRegisterInfo &MRI = MF->getRegInfo();
   // Literals also count against scalar bus limit
@@ -121,6 +145,9 @@ bool llvm::checkVOPDRegConstraints(const SIInstrInfo &TII,
     }
     UniqueLiterals.push_back(&Op);
   };
+  // Immediates which the caller will move into a scalar register. Identical
+  // values share one register, so they count like one scalar operand each.
+  SmallSet<int32_t, 2> MaterializedLiterals;
   SmallSet<Register, 4> UniqueScalarRegs;
 
   unsigned EncodingFamily = AMDGPU::getVOPDEncodingFamily(ST);
@@ -138,12 +165,33 @@ bool llvm::checkVOPDRegConstraints(const SIInstrInfo &TII,
     if (Src0.isReg()) {
       if (!isValidVOPDSrc(TII, VOPDOpc, CompIdx, 0, Src0.getReg()))
         return false;
-      if (!TRI->isVectorRegister(MRI, Src0.getReg()))
+      if (TII.regUsesConstantBus(Src0, MRI))
         UniqueScalarRegs.insert(Src0.getReg());
     } else if (!TII.isInlineConstant(Src0)) {
-      if (IsVOPD3)
-        return false;
-      AddLiteral(Src0);
+      if (!IsVOPD3) {
+        AddLiteral(Src0);
+      } else {
+        // A VOPD3 component cannot encode a literal, but src0 can read a
+        // scalar register. The pair is therefore still possible if the caller
+        // moves the value into one.
+        if (!canMaterializeVOPDLiterals(*MF) || !Src0.isImm())
+          return false;
+        // Only a 32-bit slot is handled, because the caller produces the value
+        // with a single S_MOV_B32.
+        int OpIdx = getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::src0);
+        if (TII.getOpSize(MI, OpIdx) != 4)
+          return false;
+        const TargetRegisterClass *SlotRC =
+            TRI->getCommonSubClass(getVOPDSrcRegClass(TII, VOPDOpc, CompIdx, 0),
+                                   &AMDGPU::SGPR_32RegClass);
+        if (!SlotRC)
+          return false;
+        // Only the low bits reach the register, so two operands which name
+        // the same value share one move whichever way they were written.
+        int32_t Imm = static_cast<int32_t>(Src0.getImm());
+        MaterializedLiterals.insert(Imm);
+        Fixups.push_back({CompIdx, static_cast<unsigned>(OpIdx), Imm, SlotRC});
+      }
     }
 
     // V_FMAMK_F32 (src1) and V_FMAAK_F32 (src2) have a mandatory literal.
@@ -180,7 +228,7 @@ bool llvm::checkVOPDRegConstraints(const SIInstrInfo &TII,
             return false;
           if (!isValidVOPDSrc(TII, VOPDOpc, CompIdx, 2, Src2->getReg()))
             return false;
-          if (!TRI->isVectorRegister(MRI, Src2->getReg())) {
+          if (TII.regUsesConstantBus(*Src2, MRI)) {
             assert(MI.getOpcode() == AMDGPU::V_CNDMASK_B32_e64);
             UniqueScalarRegs.insert(Src2->getReg());
           }
@@ -206,7 +254,12 @@ bool llvm::checkVOPDRegConstraints(const SIInstrInfo &TII,
 
   if (UniqueLiterals.size() > 1)
     return false;
-  if ((UniqueLiterals.size() + UniqueScalarRegs.size()) > 2)
+  // Keep materialization pair-local and do not increase instruction count or
+  // register pressure by producing two different values for one pair.
+  if (MaterializedLiterals.size() > 1)
+    return false;
+  if ((UniqueLiterals.size() + MaterializedLiterals.size() +
+       UniqueScalarRegs.size()) > 2)
     return false;
 
   auto GetVRegIdx = [&](unsigned OpcodeIdx, unsigned OperandIdx) {
@@ -230,6 +283,7 @@ bool llvm::checkVOPDRegConstraints(const SIInstrInfo &TII,
 
   LLVM_DEBUG(dbgs() << "VOPD Reg Constraints Passed\n\tX: " << MIX
                     << "\n\tY: " << MIY << "\n");
+  LiteralFixups.assign(Fixups);
   return true;
 }
 
@@ -257,9 +311,15 @@ tryMatchVOPDPairVariant(const SIInstrInfo &TII, unsigned EncodingFamily,
   const GCNSubtarget &ST = TII.getSubtarget();
   bool AllowSameVGPR = ST.hasGFX12Insts();
 
+  // Only a VOPD3 component can need a fixup; a plain one may hold a literal.
+  // checkVOPDRegConstraints() only writes this when it succeeds.
+  SmallVector<VOPDLiteralFixup, 2> Fixups;
+
   if (FirstCanBeVOPD.X && SecondCanBeVOPD.Y) {
-    if (checkVOPDRegConstraints(TII, FirstMI, SecondMI, IsVOPD3, AllowSameVGPR))
-      return VOPDMatchInfo{{&FirstMI, &SecondMI}, 0, IsVOPD3};
+    if (checkVOPDRegConstraints(TII, FirstMI, SecondMI, IsVOPD3, AllowSameVGPR,
+                                Fixups))
+      return VOPDMatchInfo{
+          {&FirstMI, &SecondMI}, 0, IsVOPD3, std::move(Fixups)};
   }
 
   if (FirstCanBeVOPD.Y && SecondCanBeVOPD.X) {
@@ -269,8 +329,10 @@ tryMatchVOPDPairVariant(const SIInstrInfo &TII, unsigned EncodingFamily,
     AllowSameVGPR &= !IsAntiDep;
     if (IsAntiDep && !TII.isVOPDAntidependencyAllowed(SecondMI))
       return std::nullopt;
-    if (checkVOPDRegConstraints(TII, SecondMI, FirstMI, IsVOPD3, AllowSameVGPR))
-      return VOPDMatchInfo{{&FirstMI, &SecondMI}, 1, IsVOPD3};
+    if (checkVOPDRegConstraints(TII, SecondMI, FirstMI, IsVOPD3, AllowSameVGPR,
+                                Fixups))
+      return VOPDMatchInfo{
+          {&FirstMI, &SecondMI}, 1, IsVOPD3, std::move(Fixups)};
   }
 
   return std::nullopt;
diff --git a/llvm/lib/Target/AMDGPU/GCNVOPDUtils.h b/llvm/lib/Target/AMDGPU/GCNVOPDUtils.h
index 9d815e7d13a7e..dbb4d074bf6e9 100644
--- a/llvm/lib/Target/AMDGPU/GCNVOPDUtils.h
+++ b/llvm/lib/Target/AMDGPU/GCNVOPDUtils.h
@@ -15,6 +15,7 @@
 #ifndef LLVM_LIB_TARGET_AMDGPU_VOPDUTILS_H
 #define LLVM_LIB_TARGET_AMDGPU_VOPDUTILS_H
 
+#include "llvm/ADT/SmallVector.h"
 #include "llvm/CodeGen/MachineScheduler.h"
 #include <optional>
 
@@ -22,11 +23,22 @@ namespace llvm {
 
 class MachineInstr;
 class SIInstrInfo;
+class MCRegisterClass;
 
-bool checkVOPDRegConstraints(const SIInstrInfo &TII,
-                             const MachineInstr &FirstMI,
-                             const MachineInstr &SecondMI, bool IsVOPD3,
-                             bool AllowSameVGPR);
+/// A 32-bit immediate which the VOPD encoding cannot hold. The pair only
+/// becomes legal after the operand is replaced by a scalar register holding
+/// \p Imm.
+struct VOPDLiteralFixup {
+  /// Component holding the immediate, AMDGPU::VOPD::X or AMDGPU::VOPD::Y.
+  unsigned CompIdx;
+  /// Index of the immediate operand within that component.
+  unsigned OpIdx;
+  /// Value which has to be placed in a register.
+  int32_t Imm;
+  /// Scalar registers the VOPD source slot can read. This is the slot class
+  /// narrowed to SGPR_32, so every register in it can be used.
+  const MCRegisterClass *SlotRC;
+};
 
 /// Describes a matched VOPD pair.
 struct VOPDMatchInfo {
@@ -35,15 +47,18 @@ struct VOPDMatchInfo {
   /// Which entry in \p InOrder is the X component.
   unsigned XIdx;
   bool IsVOPD3;
+  /// Immediates which have to be moved into scalar registers before the pair
+  /// can be built. They all have the same 32-bit value, so one register serves
+  /// the whole pair. Only a VOPD3 pair can need this.
+  SmallVector<VOPDLiteralFixup, 2> LiteralFixups;
 
   MachineInstr *getMIX() const { return InOrder[XIdx]; }
   MachineInstr *getMIY() const { return InOrder[1 - XIdx]; }
 };
 
-/// Check whether \p FirstMI and \p SecondMI, which are next to each other in
-/// program order, can be combined into a VOPD instruction. Returns the match
-/// info (program order, X/Y assignment, and encoding variant) on success, or
-/// std::nullopt if they cannot be paired.
+/// Check whether FirstMI and SecondMI can be
+/// combined into a VOPD instruction.  Returns the match info (X/Y assignment
+/// and encoding variant) on success, or std::nullopt if they cannot be paired.
 std::optional<VOPDMatchInfo> tryMatchVOPDPair(const SIInstrInfo &TII,
                                               MachineInstr &FirstMI,
                                               MachineInstr &SecondMI);
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fadd.bf16.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fadd.bf16.ll
index 5b6edff208fd4..512163fdc3272 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fadd.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fadd.bf16.ll
@@ -167,12 +167,12 @@ define amdgpu_ps <2 x bfloat> @fadd_v2bf16_vv(<2 x bfloat> %a, <2 x bfloat> %b)
 ; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v2 :: v_dual_lshlrev_b32 v3, 16, v3
 ; GFX1250-NEXT:    v_dual_add_f32 v0, v0, v1 :: v_dual_add_f32 v1, v2, v3
 ; GFX1250-NEXT:    v_bfe_u32 v2, v0, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_bfe_u32 v3, v1, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
@@ -245,12 +245,12 @@ define amdgpu_ps <2 x bfloat> @fadd_v2bf16_vs(<2 x bfloat> %a, <2 x bfloat> inre
 ; GFX1250-NEXT:    v_dual_add_f32 v0, s1, v0 :: v_dual_lshlrev_b32 v1, 16, v1
 ; GFX1250-NEXT:    v_add_f32_e32 v1, s0, v1
 ; GFX1250-NEXT:    v_bfe_u32 v2, v0, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_bfe_u32 v3, v1, 16, 1
 ; GFX1250-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
 ; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
@@ -409,12 +409,12 @@ define amdgpu_ps <2 x bfloat> @fadd_v2bf16_vc(<2 x bfloat> %a) {
 ; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v0 :: v_dual_lshlrev_b32 v1, 16, v1
 ; GFX1250-NEXT:    v_dual_add_f32 v0, 2.0, v0 :: v_dual_add_f32 v1, 2.0, v1
 ; GFX1250-NEXT:    v_bfe_u32 v2, v0, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_bfe_u32 v3, v1, 16, 1
 ; GFX1250-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
 ; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
@@ -477,12 +477,12 @@ define amdgpu_ps <2 x bfloat> @fadd_v2bf16_vl(<2 x bfloat> %a) {
 ; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v0 :: v_dual_lshlrev_b32 v1, 16, v1
 ; GFX1250-NEXT:    v_dual_add_f32 v0, 1.0, v0 :: v_dual_add_f32 v1, 0x42c80000, v1
 ; GFX1250-NEXT:    v_bfe_u32 v2, v0, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_bfe_u32 v3, v1, 16, 1
 ; GFX1250-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
 ; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fcanonicalize.bf16.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fcanonicalize.bf16.ll
index 086d7e295c006..cc029aa82e8e3 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fcanonicalize.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fcanonicalize.bf16.ll
@@ -160,12 +160,12 @@ define amdgpu_ps <2 x bfloat> @fcanonicalize_v2bf16_v(<2 x bfloat> %src) {
 ; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v0 :: v_dual_lshlrev_b32 v1, 16, v1
 ; GFX1250-NEXT:    v_dual_max_num_f32 v0, v0, v0 :: v_dual_max_num_f32 v1, v1, v1
 ; GFX1250-NEXT:    v_bfe_u32 v2, v0, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_bfe_u32 v3, v1, 16, 1
 ; GFX1250-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
 ; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fma.bf16.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fma.bf16.ll
index f411349cf92f8..61bc3fe49a716 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fma.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fma.bf16.ll
@@ -183,12 +183,12 @@ define amdgpu_ps <2 x bfloat> @fma_v2bf16_vvv(<2 x bfloat> %a, <2 x bfloat> %b,
 ; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v4 :: v_dual_lshlrev_b32 v5, 16, v5
 ; GFX1250-NEXT:    v_dual_fmac_f32 v2, v3, v1 :: v_dual_fmac_f32 v5, v0, v4
 ; GFX1250-NEXT:    v_bfe_u32 v0, v2, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v3, 0x400000, v2
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v2
 ; GFX1250-NEXT:    v_bfe_u32 v1, v5, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v5
 ; GFX1250-NEXT:    v_add3_u32 v0, v0, v2, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v5
 ; GFX1250-NEXT:    v_add3_u32 v1, v1, v5, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v3, 0x400000, v2
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v0, v3, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v5
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v1, v4, vcc_lo
@@ -267,19 +267,19 @@ define amdgpu_ps <2 x bfloat> @fma_v2bf16_vss(<2 x bfloat> %a, <2 x bfloat> inre
 ; GFX1250-NEXT:    s_lshr_b32 s2, s0, 16
 ; GFX1250-NEXT:    s_lshr_b32 s3, s1, 16
 ; GFX1250-NEXT:    s_lshl_b32 s0, s0, 16
-; GFX1250-NEXT:    s_lshl_b32 s1, s1, 16
 ; GFX1250-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
-; GFX1250-NEXT:    v_fma_f32 v0, v0, s0, s1
+; GFX1250-NEXT:    s_lshl_b32 s1, s1, 16
 ; GFX1250-NEXT:    s_lshl_b32 s2, s2, 16
+; GFX1250-NEXT:    v_fma_f32 v0, v0, s0, s1
 ; GFX1250-NEXT:    s_lshl_b32 s0, s3, 16
 ; GFX1250-NEXT:    v_fma_f32 v1, v1, s2, s0
 ; GFX1250-NEXT:    v_bfe_u32 v2, v0, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_bfe_u32 v3, v1, 16, 1
 ; GFX1250-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
 ; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
@@ -460,12 +460,12 @@ define amdgpu_ps <2 x bfloat> @fma_v2bf16_vsc(<2 x bfloat> %a, <2 x bfloat> inre
 ; GFX1250-NEXT:    v_fma_f32 v0, v0, s1, 0.5
 ; GFX1250-NEXT:    v_fma_f32 v1, v1, s0, 0.5
 ; GFX1250-NEXT:    v_bfe_u32 v2, v0, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_bfe_u32 v3, v1, 16, 1
 ; GFX1250-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
 ; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
@@ -528,19 +528,20 @@ define amdgpu_ps <2 x bfloat> @fma_v2bf16_vll(<2 x bfloat> %a) {
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    v_mov_b16_e32 v1.l, v0.h
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v0 :: v_dual_lshlrev_b32 v1, 16, v1
+; GFX1250-NEXT:    s_mov_b32 s0, 0x43480000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v0 :: v_dual_mov_b32 v2, s0
 ; GFX1250-NEXT:    v_fma_f32 v0, v0, 1.0, 2.0
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
+; GFX1250-NEXT:    s_mov_b32 s0, 0x400000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v1, 16, v1 :: v_dual_bitop2_b32 v4, s0, v0 bitop3:0x54
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
-; GFX1250-NEXT:    v_mov_b32_e32 v2, 0x43480000
 ; GFX1250-NEXT:    v_fmac_f32_e32 v2, 0x42c80000, v1
 ; GFX1250-NEXT:    v_bfe_u32 v1, v0, 16, 1
-; GFX1250-NEXT:    v_bfe_u32 v3, v2, 16, 1
 ; GFX1250-NEXT:    v_add3_u32 v1, v1, v0, 0x7fff
-; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v2
-; GFX1250-NEXT:    v_add3_u32 v3, v3, v2, 0x7fff
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v1, v1, v4, vcc_lo
+; GFX1250-NEXT:    v_bfe_u32 v3, v2, 16, 1
+; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v2
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v2
+; GFX1250-NEXT:    v_add3_u32 v3, v3, v2, 0x7fff
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
 ; GFX1250-NEXT:    v_mov_b16_e32 v0.l, v1.h
 ; GFX1250-NEXT:    ; return to shader part epilog
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fmul.bf16.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fmul.bf16.ll
index 9c02c096aaf4c..0cf8015d3b325 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fmul.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fmul.bf16.ll
@@ -167,12 +167,12 @@ define amdgpu_ps <2 x bfloat> @fmul_v2bf16_vv(<2 x bfloat> %a, <2 x bfloat> %b)
 ; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v2 :: v_dual_lshlrev_b32 v3, 16, v3
 ; GFX1250-NEXT:    v_dual_mul_f32 v0, v0, v1 :: v_dual_mul_f32 v1, v2, v3
 ; GFX1250-NEXT:    v_bfe_u32 v2, v0, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_bfe_u32 v3, v1, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
@@ -245,12 +245,12 @@ define amdgpu_ps <2 x bfloat> @fmul_v2bf16_vs(<2 x bfloat> %a, <2 x bfloat> inre
 ; GFX1250-NEXT:    v_dual_mul_f32 v0, s1, v0 :: v_dual_lshlrev_b32 v1, 16, v1
 ; GFX1250-NEXT:    v_mul_f32_e32 v1, s0, v1
 ; GFX1250-NEXT:    v_bfe_u32 v2, v0, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_bfe_u32 v3, v1, 16, 1
 ; GFX1250-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
 ; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
@@ -409,12 +409,12 @@ define amdgpu_ps <2 x bfloat> @fmul_v2bf16_vc(<2 x bfloat> %a) {
 ; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v0 :: v_dual_lshlrev_b32 v1, 16, v1
 ; GFX1250-NEXT:    v_dual_mul_f32 v0, 0.5, v0 :: v_dual_mul_f32 v1, 0.5, v1
 ; GFX1250-NEXT:    v_bfe_u32 v2, v0, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_bfe_u32 v3, v1, 16, 1
 ; GFX1250-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
 ; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
@@ -477,12 +477,12 @@ define amdgpu_ps <2 x bfloat> @fmul_v2bf16_vl(<2 x bfloat> %a) {
 ; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v0 :: v_dual_lshlrev_b32 v1, 16, v1
 ; GFX1250-NEXT:    v_dual_mul_f32 v0, 1.0, v0 :: v_dual_mul_f32 v1, 0x42c80000, v1
 ; GFX1250-NEXT:    v_bfe_u32 v2, v0, 16, 1
-; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_bfe_u32 v3, v1, 16, 1
 ; GFX1250-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
 ; GFX1250-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
+; GFX1250-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.bf16.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.bf16.ll
index f9e9b145aa955..06c2495ab4bff 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.bf16.ll
@@ -646,12 +646,13 @@ define amdgpu_ps <2 x bfloat> @fptrunc_v2f32_to_v2bf16_v(<2 x float> %a) {
 ; GFX1250-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v2, v0, 16, 1
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v3, v1, 16, 1
-; GFX1250-FAKE16-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-FAKE16-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
-; GFX1250-FAKE16-NEXT:    v_or_b32_e32 v5, 0x400000, v1
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1250-FAKE16-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
+; GFX1250-FAKE16-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-FAKE16-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT:    v_or_b32_e32 v4, 0x400000, v0
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e32 v0, v2, v4, vcc_lo
 ; GFX1250-FAKE16-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e32 v1, v3, v5, vcc_lo
@@ -667,12 +668,13 @@ define amdgpu_ps <2 x bfloat> @fptrunc_v2f32_to_v2bf16_v(<2 x float> %a) {
 ; GFX1250-TRUE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-TRUE16-NEXT:    v_bfe_u32 v2, v0, 16, 1
 ; GFX1250-TRUE16-NEXT:    v_bfe_u32 v3, v1, 16, 1
-; GFX1250-TRUE16-NEXT:    v_or_b32_e32 v4, 0x400000, v0
 ; GFX1250-TRUE16-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v0
-; GFX1250-TRUE16-NEXT:    v_or_b32_e32 v5, 0x400000, v1
+; GFX1250-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1250-TRUE16-NEXT:    v_add3_u32 v2, v2, v0, 0x7fff
+; GFX1250-TRUE16-NEXT:    v_or_b32_e32 v5, 0x400000, v1
 ; GFX1250-TRUE16-NEXT:    v_add3_u32 v3, v3, v1, 0x7fff
-; GFX1250-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1250-TRUE16-NEXT:    v_or_b32_e32 v4, 0x400000, v0
+; GFX1250-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1250-TRUE16-NEXT:    v_cndmask_b32_e32 v2, v2, v4, vcc_lo
 ; GFX1250-TRUE16-NEXT:    v_cmp_u_f32_e32 vcc_lo, 0, v1
 ; GFX1250-TRUE16-NEXT:    v_cndmask_b32_e32 v0, v3, v5, vcc_lo
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.ll
index d34b222eb56ea..6ac4f39f138fa 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fptrunc.ll
@@ -403,12 +403,12 @@ define amdgpu_ps half @fptrunc_f64_to_f16_div(double %a) {
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    v_and_or_b32 v0, 0x1ff, v1, v0
 ; GFX1250-NEXT:    v_bfe_u32 v2, v1, 20, 11
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_dual_lshrrev_b32 v3, 8, v1 :: v_dual_lshrrev_b32 v1, 16, v1
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v0
 ; GFX1250-NEXT:    v_add_nc_u32_e32 v2, 0xfffffc10, v2
-; GFX1250-NEXT:    v_dual_lshrrev_b32 v3, 8, v1 :: v_dual_lshrrev_b32 v1, 16, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v0, 0, 1, vcc_lo
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_sub_nc_u32_e32 v4, 1, v2
 ; GFX1250-NEXT:    v_and_or_b32 v0, 0xffe, v3, v0
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
index ebec9537f3f1a..f5fee7177d98d 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_buffer.ll
@@ -1735,19 +1735,34 @@ define amdgpu_kernel void @add_i32_varying_offset(ptr addrspace(1) %out, ptr add
 ; GFX12W32-NEXT:    global_store_b32 v0, v1, s[0:1]
 ; GFX12W32-NEXT:    s_endpgm
 ;
-; GFX13-LABEL: add_i32_varying_offset:
-; GFX13:       ; %bb.0: ; %entry
-; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 1
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    buffer_atomic_add_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
-; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_mov_b32_e32 v0, 0
-; GFX13-NEXT:    s_wait_loadcnt 0x0
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    global_store_b32 v0, v1, s[0:1]
-; GFX13-NEXT:    s_endpgm
+; GFX13W64-LABEL: add_i32_varying_offset:
+; GFX13W64:       ; %bb.0: ; %entry
+; GFX13W64-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W64-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13W64-NEXT:    v_mov_b32_e32 v1, 1
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    buffer_atomic_add_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
+; GFX13W64-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W64-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W64-NEXT:    s_wait_loadcnt 0x0
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W64-NEXT:    s_endpgm
+;
+; GFX13W32-LABEL: add_i32_varying_offset:
+; GFX13W32:       ; %bb.0: ; %entry
+; GFX13W32-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W32-NEXT:    s_mov_b32 s6, 0x3ff
+; GFX13W32-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13W32-NEXT:    v_dual_mov_b32 v1, 1 :: v_dual_bitop2_b32 v0, s6, v0 bitop3:0x40
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    buffer_atomic_add_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
+; GFX13W32-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W32-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W32-NEXT:    s_wait_loadcnt 0x0
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W32-NEXT:    s_endpgm
 entry:
   %lane = call i32 @llvm.amdgcn.workitem.id.x()
   %old = call i32 @llvm.amdgcn.raw.ptr.buffer.atomic.add(i32 1, ptr addrspace(8) %inout, i32 %lane, i32 0, i32 0)
@@ -3003,19 +3018,34 @@ define amdgpu_kernel void @sub_i32_varying_offset(ptr addrspace(1) %out, ptr add
 ; GFX12W32-NEXT:    global_store_b32 v0, v1, s[0:1]
 ; GFX12W32-NEXT:    s_endpgm
 ;
-; GFX13-LABEL: sub_i32_varying_offset:
-; GFX13:       ; %bb.0: ; %entry
-; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 1
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    buffer_atomic_sub_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
-; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_mov_b32_e32 v0, 0
-; GFX13-NEXT:    s_wait_loadcnt 0x0
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    global_store_b32 v0, v1, s[0:1]
-; GFX13-NEXT:    s_endpgm
+; GFX13W64-LABEL: sub_i32_varying_offset:
+; GFX13W64:       ; %bb.0: ; %entry
+; GFX13W64-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W64-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13W64-NEXT:    v_mov_b32_e32 v1, 1
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    buffer_atomic_sub_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
+; GFX13W64-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W64-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W64-NEXT:    s_wait_loadcnt 0x0
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W64-NEXT:    s_endpgm
+;
+; GFX13W32-LABEL: sub_i32_varying_offset:
+; GFX13W32:       ; %bb.0: ; %entry
+; GFX13W32-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W32-NEXT:    s_mov_b32 s6, 0x3ff
+; GFX13W32-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13W32-NEXT:    v_dual_mov_b32 v1, 1 :: v_dual_bitop2_b32 v0, s6, v0 bitop3:0x40
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    buffer_atomic_sub_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
+; GFX13W32-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W32-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W32-NEXT:    s_wait_loadcnt 0x0
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W32-NEXT:    s_endpgm
 entry:
   %lane = call i32 @llvm.amdgcn.workitem.id.x()
   %old = call i32 @llvm.amdgcn.raw.ptr.buffer.atomic.sub(i32 1, ptr addrspace(8) %inout, i32 %lane, i32 0, i32 0)
@@ -3025,3 +3055,4 @@ entry:
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; GFX11: {{.*}}
 ; GFX12: {{.*}}
+; GFX13: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
index debb51db298c1..68defb3528267 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_raw_buffer.ll
@@ -1246,19 +1246,34 @@ define amdgpu_kernel void @add_i32_varying_offset(ptr addrspace(1) %out, ptr add
 ; GFX12W32-NEXT:    global_store_b32 v0, v1, s[0:1]
 ; GFX12W32-NEXT:    s_endpgm
 ;
-; GFX13-LABEL: add_i32_varying_offset:
-; GFX13:       ; %bb.0: ; %entry
-; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 1
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    buffer_atomic_add_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
-; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_mov_b32_e32 v0, 0
-; GFX13-NEXT:    s_wait_loadcnt 0x0
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    global_store_b32 v0, v1, s[0:1]
-; GFX13-NEXT:    s_endpgm
+; GFX13W64-LABEL: add_i32_varying_offset:
+; GFX13W64:       ; %bb.0: ; %entry
+; GFX13W64-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W64-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13W64-NEXT:    v_mov_b32_e32 v1, 1
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    buffer_atomic_add_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
+; GFX13W64-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W64-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W64-NEXT:    s_wait_loadcnt 0x0
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W64-NEXT:    s_endpgm
+;
+; GFX13W32-LABEL: add_i32_varying_offset:
+; GFX13W32:       ; %bb.0: ; %entry
+; GFX13W32-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W32-NEXT:    s_mov_b32 s6, 0x3ff
+; GFX13W32-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13W32-NEXT:    v_dual_mov_b32 v1, 1 :: v_dual_bitop2_b32 v0, s6, v0 bitop3:0x40
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    buffer_atomic_add_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
+; GFX13W32-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W32-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W32-NEXT:    s_wait_loadcnt 0x0
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W32-NEXT:    s_endpgm
 entry:
   %lane = call i32 @llvm.amdgcn.workitem.id.x()
   %old = call i32 @llvm.amdgcn.raw.ptr.buffer.atomic.add(i32 1, ptr addrspace(8) %inout, i32 %lane, i32 0, i32 0)
@@ -2514,19 +2529,34 @@ define amdgpu_kernel void @sub_i32_varying_offset(ptr addrspace(1) %out, ptr add
 ; GFX12W32-NEXT:    global_store_b32 v0, v1, s[0:1]
 ; GFX12W32-NEXT:    s_endpgm
 ;
-; GFX13-LABEL: sub_i32_varying_offset:
-; GFX13:       ; %bb.0: ; %entry
-; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 1
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    buffer_atomic_sub_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
-; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_mov_b32_e32 v0, 0
-; GFX13-NEXT:    s_wait_loadcnt 0x0
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    global_store_b32 v0, v1, s[0:1]
-; GFX13-NEXT:    s_endpgm
+; GFX13W64-LABEL: sub_i32_varying_offset:
+; GFX13W64:       ; %bb.0: ; %entry
+; GFX13W64-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W64-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13W64-NEXT:    v_mov_b32_e32 v1, 1
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    buffer_atomic_sub_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
+; GFX13W64-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W64-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W64-NEXT:    s_wait_loadcnt 0x0
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W64-NEXT:    s_endpgm
+;
+; GFX13W32-LABEL: sub_i32_varying_offset:
+; GFX13W32:       ; %bb.0: ; %entry
+; GFX13W32-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W32-NEXT:    s_mov_b32 s6, 0x3ff
+; GFX13W32-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13W32-NEXT:    v_dual_mov_b32 v1, 1 :: v_dual_bitop2_b32 v0, s6, v0 bitop3:0x40
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    buffer_atomic_sub_u32 v1, v0, s[0:3], null offen th:TH_ATOMIC_RETURN
+; GFX13W32-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W32-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W32-NEXT:    s_wait_loadcnt 0x0
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W32-NEXT:    s_endpgm
 entry:
   %lane = call i32 @llvm.amdgcn.workitem.id.x()
   %old = call i32 @llvm.amdgcn.raw.ptr.buffer.atomic.sub(i32 1, ptr addrspace(8) %inout, i32 %lane, i32 0, i32 0)
@@ -2536,3 +2566,4 @@ entry:
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; GFX11: {{.*}}
 ; GFX12: {{.*}}
+; GFX13: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
index c140252b6003d..0a1e47d75d82d 100644
--- a/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
+++ b/llvm/test/CodeGen/AMDGPU/atomic_optimizations_struct_buffer.ll
@@ -1272,19 +1272,34 @@ define amdgpu_kernel void @add_i32_varying_vindex(ptr addrspace(1) %out, ptr add
 ; GFX12W32-NEXT:    global_store_b32 v0, v1, s[0:1]
 ; GFX12W32-NEXT:    s_endpgm
 ;
-; GFX13-LABEL: add_i32_varying_vindex:
-; GFX13:       ; %bb.0: ; %entry
-; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 1
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    buffer_atomic_add_u32 v1, v0, s[0:3], null idxen th:TH_ATOMIC_RETURN
-; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_mov_b32_e32 v0, 0
-; GFX13-NEXT:    s_wait_loadcnt 0x0
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    global_store_b32 v0, v1, s[0:1]
-; GFX13-NEXT:    s_endpgm
+; GFX13W64-LABEL: add_i32_varying_vindex:
+; GFX13W64:       ; %bb.0: ; %entry
+; GFX13W64-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W64-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13W64-NEXT:    v_mov_b32_e32 v1, 1
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    buffer_atomic_add_u32 v1, v0, s[0:3], null idxen th:TH_ATOMIC_RETURN
+; GFX13W64-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W64-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W64-NEXT:    s_wait_loadcnt 0x0
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W64-NEXT:    s_endpgm
+;
+; GFX13W32-LABEL: add_i32_varying_vindex:
+; GFX13W32:       ; %bb.0: ; %entry
+; GFX13W32-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W32-NEXT:    s_mov_b32 s6, 0x3ff
+; GFX13W32-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13W32-NEXT:    v_dual_mov_b32 v1, 1 :: v_dual_bitop2_b32 v0, s6, v0 bitop3:0x40
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    buffer_atomic_add_u32 v1, v0, s[0:3], null idxen th:TH_ATOMIC_RETURN
+; GFX13W32-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W32-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W32-NEXT:    s_wait_loadcnt 0x0
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W32-NEXT:    s_endpgm
 entry:
   %lane = call i32 @llvm.amdgcn.workitem.id.x()
   %old = call i32 @llvm.amdgcn.struct.ptr.buffer.atomic.add(i32 1, ptr addrspace(8) %inout, i32 %lane, i32 0, i32 0, i32 0)
@@ -2721,19 +2736,34 @@ define amdgpu_kernel void @sub_i32_varying_vindex(ptr addrspace(1) %out, ptr add
 ; GFX12W32-NEXT:    global_store_b32 v0, v1, s[0:1]
 ; GFX12W32-NEXT:    s_endpgm
 ;
-; GFX13-LABEL: sub_i32_varying_vindex:
-; GFX13:       ; %bb.0: ; %entry
-; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 1
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    buffer_atomic_sub_u32 v1, v0, s[0:3], null idxen th:TH_ATOMIC_RETURN
-; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_mov_b32_e32 v0, 0
-; GFX13-NEXT:    s_wait_loadcnt 0x0
-; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    global_store_b32 v0, v1, s[0:1]
-; GFX13-NEXT:    s_endpgm
+; GFX13W64-LABEL: sub_i32_varying_vindex:
+; GFX13W64:       ; %bb.0: ; %entry
+; GFX13W64-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W64-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13W64-NEXT:    v_mov_b32_e32 v1, 1
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    buffer_atomic_sub_u32 v1, v0, s[0:3], null idxen th:TH_ATOMIC_RETURN
+; GFX13W64-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W64-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W64-NEXT:    s_wait_loadcnt 0x0
+; GFX13W64-NEXT:    s_wait_kmcnt 0x0
+; GFX13W64-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W64-NEXT:    s_endpgm
+;
+; GFX13W32-LABEL: sub_i32_varying_vindex:
+; GFX13W32:       ; %bb.0: ; %entry
+; GFX13W32-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
+; GFX13W32-NEXT:    s_mov_b32 s6, 0x3ff
+; GFX13W32-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13W32-NEXT:    v_dual_mov_b32 v1, 1 :: v_dual_bitop2_b32 v0, s6, v0 bitop3:0x40
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    buffer_atomic_sub_u32 v1, v0, s[0:3], null idxen th:TH_ATOMIC_RETURN
+; GFX13W32-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13W32-NEXT:    v_mov_b32_e32 v0, 0
+; GFX13W32-NEXT:    s_wait_loadcnt 0x0
+; GFX13W32-NEXT:    s_wait_kmcnt 0x0
+; GFX13W32-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX13W32-NEXT:    s_endpgm
 entry:
   %lane = call i32 @llvm.amdgcn.workitem.id.x()
   %old = call i32 @llvm.amdgcn.struct.ptr.buffer.atomic.sub(i32 1, ptr addrspace(8) %inout, i32 %lane, i32 0, i32 0, i32 0)
@@ -2898,3 +2928,4 @@ entry:
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; GFX11: {{.*}}
 ; GFX12: {{.*}}
+; GFX13: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/bf16.ll b/llvm/test/CodeGen/AMDGPU/bf16.ll
index a15d21d8e6a99..04fe0d297b84d 100644
--- a/llvm/test/CodeGen/AMDGPU/bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/bf16.ll
@@ -5544,10 +5544,10 @@ define <2 x float> @global_extload_v2bf16_to_v2f32(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b32 v1, v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v1
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v1 :: v_dual_bitop2_b32 v1, s0, v1 bitop3:0x40
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %load = load <2 x bfloat>, ptr addrspace(1) %ptr
   %fpext = fpext <2 x bfloat> %load to <2 x float>
@@ -5638,10 +5638,10 @@ define <3 x float> @global_extload_v3bf16_to_v3f32(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b64 v[2:3], v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
 ; GFX1250-NEXT:    v_lshlrev_b32_e32 v2, 16, v3
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %load = load <3 x bfloat>, ptr addrspace(1) %ptr
@@ -5729,12 +5729,13 @@ define <4 x float> @global_extload_v4bf16_to_v4f32(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b64 v[2:3], v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v2, 16, v3
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v3
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v3 :: v_dual_bitop2_b32 v3, s0, v3 bitop3:0x40
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %load = load <4 x bfloat>, ptr addrspace(1) %ptr
   %fpext = fpext <4 x bfloat> %load to <4 x float>
@@ -5831,12 +5832,13 @@ define <5 x float> @global_extload_v5bf16_to_v5f32(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b128 v[2:5], v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v2, 16, v3
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v3
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v3 :: v_dual_bitop2_b32 v3, s0, v3 bitop3:0x40
 ; GFX1250-NEXT:    v_lshlrev_b32_e32 v4, 16, v4
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %load = load <5 x bfloat>, ptr addrspace(1) %ptr
@@ -5949,13 +5951,15 @@ define <6 x float> @global_extload_v6bf16_to_v6f32(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b96 v[4:6], v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v4 :: v_dual_lshlrev_b32 v2, 16, v5
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v4
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v5
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v4, 16, v6
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v6
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v4 :: v_dual_bitop2_b32 v1, s0, v4 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v5 :: v_dual_bitop2_b32 v3, s0, v5 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v6 :: v_dual_bitop2_b32 v5, s0, v6 bitop3:0x40
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %load = load <6 x bfloat>, ptr addrspace(1) %ptr
   %fpext = fpext <6 x bfloat> %load to <6 x float>
@@ -6066,15 +6070,18 @@ define <8 x float> @global_extload_v8bf16_to_v8f32(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b128 v[4:7], v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v4 :: v_dual_lshlrev_b32 v2, 16, v5
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v4
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v5
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v4, 16, v6
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v6
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v6, 16, v7
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v7
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v4 :: v_dual_bitop2_b32 v1, s0, v4 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v5 :: v_dual_bitop2_b32 v3, s0, v5 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v6 :: v_dual_bitop2_b32 v5, s0, v6 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v7 :: v_dual_bitop2_b32 v7, s0, v7 bitop3:0x40
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %load = load <8 x bfloat>, ptr addrspace(1) %ptr
   %fpext = fpext <8 x bfloat> %load to <8 x float>
@@ -6251,23 +6258,28 @@ define <16 x float> @global_extload_v16bf16_to_v16f32(ptr addrspace(1) %ptr) #0
 ; GFX1250-NEXT:    s_clause 0x1
 ; GFX1250-NEXT:    global_load_b128 v[4:7], v[0:1], off
 ; GFX1250-NEXT:    global_load_b128 v[12:15], v[0:1], off offset:16
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x1
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v4 :: v_dual_lshlrev_b32 v2, 16, v5
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v4
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v5
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v4, 16, v6
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v6
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v6, 16, v7
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v7
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v4 :: v_dual_bitop2_b32 v1, s0, v4 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v5 :: v_dual_bitop2_b32 v3, s0, v5 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v6 :: v_dual_bitop2_b32 v5, s0, v6 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v7 :: v_dual_bitop2_b32 v7, s0, v7 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v12 :: v_dual_lshlrev_b32 v10, 16, v13
-; GFX1250-NEXT:    v_and_b32_e32 v9, 0xffff0000, v12
-; GFX1250-NEXT:    v_and_b32_e32 v11, 0xffff0000, v13
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v12, 16, v14
-; GFX1250-NEXT:    v_and_b32_e32 v13, 0xffff0000, v14
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v14, 16, v15
-; GFX1250-NEXT:    v_and_b32_e32 v15, 0xffff0000, v15
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v12 :: v_dual_bitop2_b32 v9, s0, v12 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v10, 16, v13 :: v_dual_bitop2_b32 v11, s0, v13 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v12, 16, v14 :: v_dual_bitop2_b32 v13, s0, v14 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v14, 16, v15 :: v_dual_bitop2_b32 v15, s0, v15 bitop3:0x40
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %load = load <16 x bfloat>, ptr addrspace(1) %ptr
   %fpext = fpext <16 x bfloat> %load to <16 x float>
@@ -6570,39 +6582,49 @@ define <32 x float> @global_extload_v32bf16_to_v32f32(ptr addrspace(1) %ptr) #0
 ; GFX1250-NEXT:    global_load_b128 v[12:15], v[0:1], off offset:16
 ; GFX1250-NEXT:    global_load_b128 v[20:23], v[0:1], off offset:32
 ; GFX1250-NEXT:    global_load_b128 v[28:31], v[0:1], off offset:48
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x3
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v4 :: v_dual_lshlrev_b32 v2, 16, v5
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v4
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v5
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v4, 16, v6
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v6
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v6, 16, v7
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v7
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v4 :: v_dual_bitop2_b32 v1, s0, v4 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v5 :: v_dual_bitop2_b32 v3, s0, v5 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v6 :: v_dual_bitop2_b32 v5, s0, v6 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v7 :: v_dual_bitop2_b32 v7, s0, v7 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x2
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v12 :: v_dual_lshlrev_b32 v10, 16, v13
-; GFX1250-NEXT:    v_and_b32_e32 v9, 0xffff0000, v12
-; GFX1250-NEXT:    v_and_b32_e32 v11, 0xffff0000, v13
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v12, 16, v14
-; GFX1250-NEXT:    v_and_b32_e32 v13, 0xffff0000, v14
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v14, 16, v15
-; GFX1250-NEXT:    v_and_b32_e32 v15, 0xffff0000, v15
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v12 :: v_dual_bitop2_b32 v9, s0, v12 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v10, 16, v13 :: v_dual_bitop2_b32 v11, s0, v13 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v12, 16, v14 :: v_dual_bitop2_b32 v13, s0, v14 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v14, 16, v15 :: v_dual_bitop2_b32 v15, s0, v15 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x1
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v16, 16, v20 :: v_dual_lshlrev_b32 v18, 16, v21
-; GFX1250-NEXT:    v_and_b32_e32 v17, 0xffff0000, v20
-; GFX1250-NEXT:    v_and_b32_e32 v19, 0xffff0000, v21
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v20, 16, v22
-; GFX1250-NEXT:    v_and_b32_e32 v21, 0xffff0000, v22
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v22, 16, v23
-; GFX1250-NEXT:    v_and_b32_e32 v23, 0xffff0000, v23
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v16, 16, v20 :: v_dual_bitop2_b32 v17, s0, v20 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v18, 16, v21 :: v_dual_bitop2_b32 v19, s0, v21 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v20, 16, v22 :: v_dual_bitop2_b32 v21, s0, v22 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v22, 16, v23 :: v_dual_bitop2_b32 v23, s0, v23 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v24, 16, v28 :: v_dual_lshlrev_b32 v26, 16, v29
-; GFX1250-NEXT:    v_and_b32_e32 v25, 0xffff0000, v28
-; GFX1250-NEXT:    v_and_b32_e32 v27, 0xffff0000, v29
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v28, 16, v30
-; GFX1250-NEXT:    v_and_b32_e32 v29, 0xffff0000, v30
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v30, 16, v31
-; GFX1250-NEXT:    v_and_b32_e32 v31, 0xffff0000, v31
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v24, 16, v28 :: v_dual_bitop2_b32 v25, s0, v28 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v26, 16, v29 :: v_dual_bitop2_b32 v27, s0, v29 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v28, 16, v30 :: v_dual_bitop2_b32 v29, s0, v30 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v30, 16, v31 :: v_dual_bitop2_b32 v31, s0, v31 bitop3:0x40
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %load = load <32 x bfloat>, ptr addrspace(1) %ptr
   %fpext = fpext <32 x bfloat> %load to <32 x float>
@@ -6701,11 +6723,11 @@ define <2 x double> @global_extload_v2bf16_to_v2f64(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b32 v0, v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v1, 16, v0
-; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v1, 16, v0 :: v_dual_bitop2_b32 v2, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[0:1], v1
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[2:3], v2
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
@@ -6821,15 +6843,16 @@ define <3 x double> @global_extload_v3bf16_to_v3f64(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b64 v[0:1], v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_lshlrev_b32 v4, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v0
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v3, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v4, 16, v1
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[0:1], v2
-; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[4:5], v4
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3)
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[2:3], v3
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3)
+; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[4:5], v4
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %load = load <3 x bfloat>, ptr addrspace(1) %ptr
   %fpext = fpext <3 x bfloat> %load to <3 x double>
@@ -6942,16 +6965,18 @@ define <4 x double> @global_extload_v4bf16_to_v4f64(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b64 v[2:3], v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_lshlrev_b32 v4, 16, v3
-; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v2
-; GFX1250-NEXT:    v_and_b32_e32 v6, 0xffff0000, v3
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v2, s0, v2 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v3 :: v_dual_bitop2_b32 v6, s0, v3 bitop3:0x40
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
-; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[4:5], v4
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[2:3], v2
+; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[4:5], v4
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4)
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[6:7], v6
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %load = load <4 x bfloat>, ptr addrspace(1) %ptr
@@ -7079,16 +7104,18 @@ define <5 x double> @global_extload_v5bf16_to_v5f64(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b128 v[2:5], v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_lshlrev_b32 v5, 16, v3
-; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v2
-; GFX1250-NEXT:    v_and_b32_e32 v6, 0xffff0000, v3
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v2, s0, v2 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v5, 16, v3 :: v_dual_bitop2_b32 v6, s0, v3 bitop3:0x40
 ; GFX1250-NEXT:    v_lshlrev_b32_e32 v8, 16, v4
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4)
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
-; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[4:5], v5
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[2:3], v2
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4)
+; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[4:5], v5
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[6:7], v6
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[8:9], v8
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
@@ -7225,14 +7252,15 @@ define <6 x double> @global_extload_v6bf16_to_v6f64(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b96 v[4:6], v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 16, v4
-; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v4
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v4, 16, v5
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v5
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v8, 16, v6
-; GFX1250-NEXT:    v_and_b32_e32 v10, 0xffff0000, v6
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v4 :: v_dual_bitop2_b32 v2, s0, v4 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v5 :: v_dual_bitop2_b32 v7, s0, v5 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v6 :: v_dual_bitop2_b32 v10, s0, v6 bitop3:0x40
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[2:3], v2
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[4:5], v4
@@ -7397,14 +7425,18 @@ define <8 x double> @global_extload_v8bf16_to_v8f64(ptr addrspace(1) %ptr) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b128 v[8:11], v[0:1], off
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v8 :: v_dual_lshlrev_b32 v4, 16, v9
-; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v8
-; GFX1250-NEXT:    v_and_b32_e32 v6, 0xffff0000, v9
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v10 :: v_dual_lshlrev_b32 v12, 16, v11
-; GFX1250-NEXT:    v_and_b32_e32 v10, 0xffff0000, v10
-; GFX1250-NEXT:    v_and_b32_e32 v14, 0xffff0000, v11
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v8 :: v_dual_bitop2_b32 v2, s0, v8 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v9 :: v_dual_bitop2_b32 v6, s0, v9 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v10 :: v_dual_bitop2_b32 v10, s0, v10 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v12, 16, v11 :: v_dual_bitop2_b32 v14, s0, v11 bitop3:0x40
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[2:3], v2
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[4:5], v4
@@ -7685,21 +7717,28 @@ define <16 x double> @global_extload_v16bf16_to_v16f64(ptr addrspace(1) %ptr) #0
 ; GFX1250-NEXT:    s_clause 0x1
 ; GFX1250-NEXT:    global_load_b128 v[8:11], v[0:1], off
 ; GFX1250-NEXT:    global_load_b128 v[24:27], v[0:1], off offset:16
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x1
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v8 :: v_dual_lshlrev_b32 v4, 16, v9
-; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v8
-; GFX1250-NEXT:    v_and_b32_e32 v6, 0xffff0000, v9
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v10 :: v_dual_lshlrev_b32 v12, 16, v11
-; GFX1250-NEXT:    v_and_b32_e32 v10, 0xffff0000, v10
-; GFX1250-NEXT:    v_and_b32_e32 v14, 0xffff0000, v11
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v8 :: v_dual_bitop2_b32 v2, s0, v8 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v9 :: v_dual_bitop2_b32 v6, s0, v9 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v10 :: v_dual_bitop2_b32 v10, s0, v10 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v12, 16, v11 :: v_dual_bitop2_b32 v14, s0, v11 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v16, 16, v24 :: v_dual_lshlrev_b32 v20, 16, v25
-; GFX1250-NEXT:    v_and_b32_e32 v18, 0xffff0000, v24
-; GFX1250-NEXT:    v_and_b32_e32 v22, 0xffff0000, v25
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v24, 16, v26 :: v_dual_lshlrev_b32 v28, 16, v27
-; GFX1250-NEXT:    v_and_b32_e32 v26, 0xffff0000, v26
-; GFX1250-NEXT:    v_and_b32_e32 v30, 0xffff0000, v27
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v16, 16, v24 :: v_dual_bitop2_b32 v18, s0, v24 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v20, 16, v25 :: v_dual_bitop2_b32 v22, s0, v25 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v24, 16, v26 :: v_dual_bitop2_b32 v26, s0, v26 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v28, 16, v27 :: v_dual_bitop2_b32 v30, s0, v27 bitop3:0x40
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[2:3], v2
 ; GFX1250-NEXT:    v_cvt_f64_f32_e32 v[4:5], v4
@@ -32986,10 +33025,10 @@ define <3 x i16> @v_fptosi_v3bf16_to_v3i16(<3 x bfloat> %x) #0 {
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v0 :: v_dual_lshlrev_b32 v1, 16, v1
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v1, 16, v1 :: v_dual_lshlrev_b32 v0, 16, v0
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1250-NEXT:    v_cvt_pk_i16_f32 v0, v0, v2
 ; GFX1250-NEXT:    v_cvt_i32_f32_e32 v1, v1
+; GFX1250-NEXT:    v_cvt_pk_i16_f32 v0, v0, v2
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %op = fptosi <3 x bfloat> %x to <3 x i16>
   ret <3 x i16> %op
@@ -33114,10 +33153,10 @@ define <4 x i16> @v_fptosi_v4bf16_to_v4i16(<4 x bfloat> %x) #0 {
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v1
 ; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v0 :: v_dual_lshlrev_b32 v1, 16, v1
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v1, 16, v1 :: v_dual_lshlrev_b32 v0, 16, v0
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1250-NEXT:    v_cvt_pk_i16_f32 v0, v0, v3
 ; GFX1250-NEXT:    v_cvt_pk_i16_f32 v1, v1, v2
+; GFX1250-NEXT:    v_cvt_pk_i16_f32 v0, v0, v3
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %op = fptosi <4 x bfloat> %x to <4 x i16>
   ret <4 x i16> %op
@@ -33243,10 +33282,11 @@ define <2 x i32> @v_fptosi_v2bf16_to_v2i32(<2 x bfloat> %x) #0 {
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v1, 16, v0
-; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v1, 16, v0 :: v_dual_bitop2_b32 v2, s0, v0 bitop3:0x40
 ; GFX1250-NEXT:    v_cvt_i32_f32_e32 v0, v1
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cvt_i32_f32_e32 v1, v2
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %op = fptosi <2 x bfloat> %x to <2 x i32>
@@ -33334,13 +33374,14 @@ define <3 x i32> @v_fptosi_v3bf16_to_v3i32(<3 x bfloat> %x) #0 {
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_lshlrev_b32 v4, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v3, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v4, 16, v1
 ; GFX1250-NEXT:    v_cvt_i32_f32_e32 v0, v2
-; GFX1250-NEXT:    v_cvt_i32_f32_e32 v2, v4
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_cvt_i32_f32_e32 v1, v3
+; GFX1250-NEXT:    v_cvt_i32_f32_e32 v2, v4
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %op = fptosi <3 x bfloat> %x to <3 x i32>
   ret <3 x i32> %op
@@ -33439,14 +33480,16 @@ define <4 x i32> @v_fptosi_v4bf16_to_v4i32(<4 x bfloat> %x) #0 {
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_lshlrev_b32 v4, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v0
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v1
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v3, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v1 :: v_dual_bitop2_b32 v5, s0, v1 bitop3:0x40
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_cvt_i32_f32_e32 v0, v2
-; GFX1250-NEXT:    v_cvt_i32_f32_e32 v2, v4
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-NEXT:    v_cvt_i32_f32_e32 v1, v3
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    v_cvt_i32_f32_e32 v2, v4
 ; GFX1250-NEXT:    v_cvt_i32_f32_e32 v3, v5
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %op = fptosi <4 x bfloat> %x to <4 x i32>
@@ -33841,31 +33884,30 @@ define <2 x i64> @v_fptosi_v2bf16_to_v2i64(<2 x bfloat> %x) #0 {
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v1, 16, v0
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0xffff0000, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-NEXT:    v_trunc_f32_e32 v3, v0
-; GFX1250-NEXT:    v_mul_f32_e64 v2, 0x2f800000, |v3|
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_3)
-; GFX1250-NEXT:    v_floor_f32_e32 v5, v2
-; GFX1250-NEXT:    v_ashrrev_i32_e32 v2, 31, v3
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v1, 16, v0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX1250-NEXT:    v_trunc_f32_e32 v1, v1
-; GFX1250-NEXT:    v_fma_f32 v3, 0xcf800000, v5, |v3|
-; GFX1250-NEXT:    v_cvt_u32_f32_e32 v7, v5
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_trunc_f32_e32 v3, v0
 ; GFX1250-NEXT:    v_mul_f32_e64 v0, 0x2f800000, |v1|
-; GFX1250-NEXT:    v_cvt_u32_f32_e32 v8, v3
-; GFX1250-NEXT:    v_mov_b32_e32 v3, v2
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_mul_f32_e64 v2, 0x2f800000, |v3|
 ; GFX1250-NEXT:    v_floor_f32_e32 v4, v0
-; GFX1250-NEXT:    v_dual_ashrrev_i32 v0, 31, v1 :: v_dual_bitop2_b32 v7, v7, v2 bitop3:0x14
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1250-NEXT:    v_floor_f32_e32 v5, v2
+; GFX1250-NEXT:    v_dual_ashrrev_i32 v0, 31, v1 :: v_dual_ashrrev_i32 v2, 31, v3
 ; GFX1250-NEXT:    v_fma_f32 v6, 0xcf800000, v4, |v1|
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3)
+; GFX1250-NEXT:    v_fma_f32 v3, 0xcf800000, v5, |v3|
 ; GFX1250-NEXT:    v_cvt_u32_f32_e32 v4, v4
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1250-NEXT:    v_cvt_u32_f32_e32 v6, v6
+; GFX1250-NEXT:    v_cvt_u32_f32_e32 v7, v5
 ; GFX1250-NEXT:    v_mov_b32_e32 v1, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX1250-NEXT:    v_xor_b32_e32 v5, v4, v0
+; GFX1250-NEXT:    v_cvt_u32_f32_e32 v6, v6
+; GFX1250-NEXT:    v_cvt_u32_f32_e32 v8, v3
+; GFX1250-NEXT:    v_dual_mov_b32 v3, v2 :: v_dual_bitop2_b32 v5, v4, v0 bitop3:0x14
+; GFX1250-NEXT:    v_xor_b32_e32 v7, v7, v2
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-NEXT:    v_xor_b32_e32 v4, v6, v0
 ; GFX1250-NEXT:    v_xor_b32_e32 v6, v8, v2
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -34193,37 +34235,38 @@ define <3 x i64> @v_fptosi_v3bf16_to_v3i64(<3 x bfloat> %x) #0 {
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_lshlrev_b32 v1, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0xffff0000, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX1250-NEXT:    v_trunc_f32_e32 v6, v2
-; GFX1250-NEXT:    v_trunc_f32_e32 v8, v1
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_trunc_f32_e32 v7, v0
+; GFX1250-NEXT:    v_ashrrev_i32_e32 v0, 31, v6
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    v_trunc_f32_e32 v8, v1
 ; GFX1250-NEXT:    v_mul_f32_e64 v1, 0x2f800000, |v6|
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX1250-NEXT:    v_mul_f32_e64 v5, 0x2f800000, |v8|
 ; GFX1250-NEXT:    v_mul_f32_e64 v3, 0x2f800000, |v7|
-; GFX1250-NEXT:    v_dual_ashrrev_i32 v0, 31, v6 :: v_dual_ashrrev_i32 v2, 31, v7
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    v_dual_ashrrev_i32 v2, 31, v7 :: v_dual_ashrrev_i32 v4, 31, v8
+; GFX1250-NEXT:    v_mul_f32_e64 v5, 0x2f800000, |v8|
 ; GFX1250-NEXT:    v_floor_f32_e32 v9, v1
-; GFX1250-NEXT:    v_floor_f32_e32 v11, v5
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-NEXT:    v_floor_f32_e32 v10, v3
-; GFX1250-NEXT:    v_dual_mov_b32 v1, v0 :: v_dual_ashrrev_i32 v4, 31, v8
+; GFX1250-NEXT:    v_dual_mov_b32 v1, v0 :: v_dual_mov_b32 v3, v2
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    v_floor_f32_e32 v11, v5
 ; GFX1250-NEXT:    v_fma_f32 v6, 0xcf800000, v9, |v6|
-; GFX1250-NEXT:    v_fma_f32 v8, 0xcf800000, v11, |v8|
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4)
 ; GFX1250-NEXT:    v_fma_f32 v7, 0xcf800000, v10, |v7|
 ; GFX1250-NEXT:    v_cvt_u32_f32_e32 v9, v9
 ; GFX1250-NEXT:    v_cvt_u32_f32_e32 v10, v10
+; GFX1250-NEXT:    v_fma_f32 v8, 0xcf800000, v11, |v8|
 ; GFX1250-NEXT:    v_cvt_u32_f32_e32 v6, v6
 ; GFX1250-NEXT:    v_cvt_u32_f32_e32 v11, v11
 ; GFX1250-NEXT:    v_cvt_u32_f32_e32 v12, v7
+; GFX1250-NEXT:    v_dual_mov_b32 v5, v4 :: v_dual_bitop2_b32 v7, v9, v0 bitop3:0x14
 ; GFX1250-NEXT:    v_cvt_u32_f32_e32 v13, v8
-; GFX1250-NEXT:    v_dual_mov_b32 v3, v2 :: v_dual_mov_b32 v5, v4
-; GFX1250-NEXT:    v_xor_b32_e32 v7, v9, v0
 ; GFX1250-NEXT:    v_xor_b32_e32 v6, v6, v0
 ; GFX1250-NEXT:    v_xor_b32_e32 v9, v10, v2
 ; GFX1250-NEXT:    v_xor_b32_e32 v8, v12, v2
@@ -34638,24 +34681,25 @@ define <4 x i64> @v_fptosi_v4bf16_to_v4i64(<4 x bfloat> %x) #0 {
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_lshlrev_b32 v3, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0xffff0000, v0
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v1
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v3, 16, v1 :: v_dual_bitop2_b32 v1, s0, v1 bitop3:0x40
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_trunc_f32_e32 v7, v2
-; GFX1250-NEXT:    v_trunc_f32_e32 v9, v3
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-NEXT:    v_trunc_f32_e32 v8, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    v_trunc_f32_e32 v9, v3
 ; GFX1250-NEXT:    v_trunc_f32_e32 v10, v1
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-NEXT:    v_mul_f32_e64 v1, 0x2f800000, |v7|
-; GFX1250-NEXT:    v_mul_f32_e64 v5, 0x2f800000, |v9|
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-NEXT:    v_mul_f32_e64 v3, 0x2f800000, |v8|
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    v_mul_f32_e64 v5, 0x2f800000, |v9|
 ; GFX1250-NEXT:    v_mul_f32_e64 v11, 0x2f800000, |v10|
 ; GFX1250-NEXT:    v_dual_ashrrev_i32 v0, 31, v7 :: v_dual_ashrrev_i32 v2, 31, v8
 ; GFX1250-NEXT:    v_floor_f32_e32 v12, v1
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4)
 ; GFX1250-NEXT:    v_floor_f32_e32 v13, v3
 ; GFX1250-NEXT:    v_floor_f32_e32 v14, v5
 ; GFX1250-NEXT:    v_floor_f32_e32 v11, v11
@@ -38102,12 +38146,12 @@ define <2 x bfloat> @v_uitofp_v2i16_to_v2bf16(<2 x i16> %x) #0 {
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_lshrrev_b32_e32 v1, 16, v0
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_lshrrev_b32 v1, 16, v0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX1250-NEXT:    v_cvt_f32_u32_e32 v1, v1
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_cvt_f32_u32_e32 v0, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v1
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %op = uitofp <2 x i16> %x to <2 x bfloat>
@@ -38305,16 +38349,16 @@ define <3 x bfloat> @v_uitofp_v3i16_to_v3bf16(<3 x i16> %x) #0 {
 ; GFX1250TRUE16:       ; %bb.0:
 ; GFX1250TRUE16-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250TRUE16-NEXT:    s_wait_kmcnt 0x0
-; GFX1250TRUE16-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX1250TRUE16-NEXT:    v_lshrrev_b32_e32 v2, 16, v0
+; GFX1250TRUE16-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1250TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1250TRUE16-NEXT:    v_dual_lshrrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v1, s0, v1 bitop3:0x40
 ; GFX1250TRUE16-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX1250TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250TRUE16-NEXT:    v_cvt_f32_u32_e32 v1, v1
-; GFX1250TRUE16-NEXT:    v_cvt_f32_u32_e32 v2, v2
 ; GFX1250TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250TRUE16-NEXT:    v_cvt_f32_u32_e32 v2, v2
 ; GFX1250TRUE16-NEXT:    v_cvt_f32_u32_e32 v0, v0
+; GFX1250TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250TRUE16-NEXT:    v_cvt_pk_bf16_f32 v1, v1, s0
-; GFX1250TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250TRUE16-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v2
 ; GFX1250TRUE16-NEXT:    s_set_pc_i64 s[30:31]
 ;
@@ -38322,16 +38366,16 @@ define <3 x bfloat> @v_uitofp_v3i16_to_v3bf16(<3 x i16> %x) #0 {
 ; GFX1250FAKE16:       ; %bb.0:
 ; GFX1250FAKE16-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250FAKE16-NEXT:    s_wait_kmcnt 0x0
-; GFX1250FAKE16-NEXT:    v_lshrrev_b32_e32 v2, 16, v0
-; GFX1250FAKE16-NEXT:    v_and_b32_e32 v0, 0xffff, v0
+; GFX1250FAKE16-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1250FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1250FAKE16-NEXT:    v_dual_lshrrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX1250FAKE16-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX1250FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250FAKE16-NEXT:    v_cvt_f32_u32_e32 v2, v2
+; GFX1250FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250FAKE16-NEXT:    v_cvt_f32_u32_e32 v0, v0
-; GFX1250FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250FAKE16-NEXT:    v_cvt_f32_u32_e32 v1, v1
+; GFX1250FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250FAKE16-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v2
-; GFX1250FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250FAKE16-NEXT:    v_cvt_pk_bf16_f32 v1, v1, s0
 ; GFX1250FAKE16-NEXT:    s_set_pc_i64 s[30:31]
   %op = uitofp <3 x i16> %x to <3 x bfloat>
diff --git a/llvm/test/CodeGen/AMDGPU/branch-relaxation-gfx1250.ll b/llvm/test/CodeGen/AMDGPU/branch-relaxation-gfx1250.ll
index 63bcd96eb76ab..79e9f643074f2 100644
--- a/llvm/test/CodeGen/AMDGPU/branch-relaxation-gfx1250.ll
+++ b/llvm/test/CodeGen/AMDGPU/branch-relaxation-gfx1250.ll
@@ -329,13 +329,13 @@ define amdgpu_kernel void @min_long_forward_vbranch(ptr addrspace(1) %arg) #0 {
 ; GCN-NEXT:    v_nop
 ; GCN-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GCN-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GCN-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GCN-NEXT:    v_mov_b32_e32 v1, 0
+; GCN-NEXT:    s_mov_b32 s2, 0x3ff
+; GCN-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GCN-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GCN-NEXT:    s_wait_kmcnt 0x0
 ; GCN-NEXT:    global_load_b32 v2, v0, s[0:1] scale_offset scope:SCOPE_SYS
 ; GCN-NEXT:    s_wait_loadcnt 0x0
 ; GCN-NEXT:    v_lshlrev_b32_e32 v0, 2, v0
-; GCN-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GCN-NEXT:    v_add_nc_u64_e32 v[0:1], s[0:1], v[0:1]
 ; GCN-NEXT:    s_mov_b32 s0, exec_lo
 ; GCN-NEXT:    v_cmpx_ne_u32_e32 0, v2
@@ -367,13 +367,13 @@ define amdgpu_kernel void @min_long_forward_vbranch(ptr addrspace(1) %arg) #0 {
 ; GCN-ADD-PC64-NEXT:    v_nop
 ; GCN-ADD-PC64-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GCN-ADD-PC64-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GCN-ADD-PC64-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GCN-ADD-PC64-NEXT:    v_mov_b32_e32 v1, 0
+; GCN-ADD-PC64-NEXT:    s_mov_b32 s2, 0x3ff
+; GCN-ADD-PC64-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GCN-ADD-PC64-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GCN-ADD-PC64-NEXT:    s_wait_kmcnt 0x0
 ; GCN-ADD-PC64-NEXT:    global_load_b32 v2, v0, s[0:1] scale_offset scope:SCOPE_SYS
 ; GCN-ADD-PC64-NEXT:    s_wait_loadcnt 0x0
 ; GCN-ADD-PC64-NEXT:    v_lshlrev_b32_e32 v0, 2, v0
-; GCN-ADD-PC64-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GCN-ADD-PC64-NEXT:    v_add_nc_u64_e32 v[0:1], s[0:1], v[0:1]
 ; GCN-ADD-PC64-NEXT:    s_mov_b32 s0, exec_lo
 ; GCN-ADD-PC64-NEXT:    v_cmpx_ne_u32_e32 0, v2
diff --git a/llvm/test/CodeGen/AMDGPU/calling-conventions.ll b/llvm/test/CodeGen/AMDGPU/calling-conventions.ll
index 8a886acdbc82a..ab06a354631ee 100644
--- a/llvm/test/CodeGen/AMDGPU/calling-conventions.ll
+++ b/llvm/test/CodeGen/AMDGPU/calling-conventions.ll
@@ -1706,22 +1706,22 @@ define amdgpu_kernel void @amd_kernel_v4i8(<4 x i8> %arg0) {
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b32 s0, s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_mov_b32_e32 v0, 0x6050400
 ; GFX1250-NEXT:    v_mov_b32_e32 v1, 0xc0c0400
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    s_bfe_u32 s3, s0, 0x80008
 ; GFX1250-NEXT:    s_lshr_b32 s1, s0, 24
 ; GFX1250-NEXT:    s_lshr_b32 s2, s0, 16
-; GFX1250-NEXT:    s_add_co_i32 s1, s1, s1
-; GFX1250-NEXT:    s_add_co_i32 s2, s2, s2
-; GFX1250-NEXT:    s_bfe_u32 s3, s0, 0x80008
-; GFX1250-NEXT:    v_perm_b32 v0, s1, s2, v0
 ; GFX1250-NEXT:    s_add_co_i32 s0, s0, s0
 ; GFX1250-NEXT:    s_add_co_i32 s3, s3, s3
-; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    s_add_co_i32 s2, s2, s2
 ; GFX1250-NEXT:    v_perm_b32 v2, s3, s0, v1
+; GFX1250-NEXT:    v_mov_b32_e32 v0, 0x6050400
+; GFX1250-NEXT:    s_add_co_i32 s1, s1, s1
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_perm_b32 v0, s1, s2, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_lshlrev_b32_e32 v3, 16, v0
 ; GFX1250-NEXT:    v_mov_b64_e32 v[0:1], 0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250-NEXT:    v_or_b32_e32 v2, v2, v3
 ; GFX1250-NEXT:    global_store_b32 v[0:1], v2, off
 ; GFX1250-NEXT:    s_endpgm
@@ -1921,25 +1921,27 @@ define amdgpu_kernel void @amd_kernel_v5i8(<5 x i8> %arg0) {
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_mov_b32_e32 v0, 0x6050400
 ; GFX1250-NEXT:    v_mov_b32_e32 v1, 0xc0c0400
 ; GFX1250-NEXT:    v_mov_b64_e32 v[2:3], 0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    s_bfe_u32 s4, s0, 0x80008
 ; GFX1250-NEXT:    s_lshr_b32 s2, s0, 24
 ; GFX1250-NEXT:    s_lshr_b32 s3, s0, 16
-; GFX1250-NEXT:    s_add_co_i32 s2, s2, s2
-; GFX1250-NEXT:    s_add_co_i32 s3, s3, s3
-; GFX1250-NEXT:    s_bfe_u32 s4, s0, 0x80008
-; GFX1250-NEXT:    v_perm_b32 v0, s2, s3, v0
 ; GFX1250-NEXT:    s_add_co_i32 s0, s0, s0
 ; GFX1250-NEXT:    s_add_co_i32 s4, s4, s4
-; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    s_add_co_i32 s3, s3, s3
 ; GFX1250-NEXT:    v_perm_b32 v4, s4, s0, v1
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v5, 16, v0
 ; GFX1250-NEXT:    s_add_co_i32 s0, s1, s1
+; GFX1250-NEXT:    s_mov_b32 s1, 0x6050400
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v0, s1 :: v_dual_mov_b32 v6, s0
+; GFX1250-NEXT:    s_add_co_i32 s2, s2, s2
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_perm_b32 v0, s2, s3, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v5, 16, v0
 ; GFX1250-NEXT:    v_mov_b64_e32 v[0:1], 4
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2)
-; GFX1250-NEXT:    v_dual_mov_b32 v6, s0 :: v_dual_bitop2_b32 v4, v4, v5 bitop3:0x54
+; GFX1250-NEXT:    v_or_b32_e32 v4, v4, v5
 ; GFX1250-NEXT:    s_clause 0x1
 ; GFX1250-NEXT:    global_store_b8 v[0:1], v6, off
 ; GFX1250-NEXT:    global_store_b32 v[2:3], v4, off
@@ -2078,26 +2080,27 @@ define amdgpu_kernel void @amd_kernel_v8i8(<8 x i8> %arg0) {
 ; GFX1250-NEXT:    s_lshr_b32 s5, s1, 16
 ; GFX1250-NEXT:    s_add_co_i32 s4, s4, s4
 ; GFX1250-NEXT:    s_add_co_i32 s5, s5, s5
-; GFX1250-NEXT:    s_lshr_b32 s2, s0, 24
+; GFX1250-NEXT:    s_bfe_u32 s7, s1, 0x80008
 ; GFX1250-NEXT:    v_perm_b32 v2, s4, s5, v1
+; GFX1250-NEXT:    v_mov_b32_e32 v0, 0xc0c0400
+; GFX1250-NEXT:    s_add_co_i32 s1, s1, s1
+; GFX1250-NEXT:    s_add_co_i32 s7, s7, s7
+; GFX1250-NEXT:    s_lshr_b32 s2, s0, 24
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
+; GFX1250-NEXT:    v_perm_b32 v3, s7, s1, v0
 ; GFX1250-NEXT:    s_lshr_b32 s3, s0, 16
 ; GFX1250-NEXT:    s_add_co_i32 s2, s2, s2
 ; GFX1250-NEXT:    s_add_co_i32 s3, s3, s3
 ; GFX1250-NEXT:    s_bfe_u32 s6, s0, 0x80008
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
+; GFX1250-NEXT:    v_or_b32_e32 v3, v3, v2
 ; GFX1250-NEXT:    v_perm_b32 v1, s2, s3, v1
-; GFX1250-NEXT:    v_mov_b32_e32 v0, 0xc0c0400
-; GFX1250-NEXT:    s_bfe_u32 s7, s1, 0x80008
-; GFX1250-NEXT:    s_add_co_i32 s1, s1, s1
 ; GFX1250-NEXT:    s_add_co_i32 s0, s0, s0
-; GFX1250-NEXT:    s_add_co_i32 s7, s7, s7
 ; GFX1250-NEXT:    s_add_co_i32 s6, s6, s6
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v5, 16, v1
-; GFX1250-NEXT:    v_perm_b32 v3, s7, s1, v0
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_perm_b32 v4, s6, s0, v0
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v5, 16, v1
 ; GFX1250-NEXT:    v_mov_b64_e32 v[0:1], 0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX1250-NEXT:    v_or_b32_e32 v3, v3, v2
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250-NEXT:    v_or_b32_e32 v2, v4, v5
 ; GFX1250-NEXT:    global_store_b64 v[0:1], v[2:3], off
 ; GFX1250-NEXT:    s_endpgm
@@ -2314,31 +2317,30 @@ define amdgpu_kernel void @amd_kernel_v16i8(<16 x i8> %arg0) {
 ; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
 ; GFX1250-NEXT:    v_mov_b32_e32 v1, 0x6050400
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    s_lshr_b32 s10, s3, 24
-; GFX1250-NEXT:    s_lshr_b32 s11, s3, 16
-; GFX1250-NEXT:    s_add_co_i32 s10, s10, s10
-; GFX1250-NEXT:    s_add_co_i32 s11, s11, s11
 ; GFX1250-NEXT:    s_lshr_b32 s8, s2, 24
-; GFX1250-NEXT:    v_perm_b32 v2, s10, s11, v1
 ; GFX1250-NEXT:    s_lshr_b32 s9, s2, 16
 ; GFX1250-NEXT:    s_add_co_i32 s8, s8, s8
 ; GFX1250-NEXT:    s_add_co_i32 s9, s9, s9
-; GFX1250-NEXT:    s_lshr_b32 s4, s0, 24
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
+; GFX1250-NEXT:    s_lshr_b32 s10, s3, 24
 ; GFX1250-NEXT:    v_perm_b32 v3, s8, s9, v1
+; GFX1250-NEXT:    s_lshr_b32 s11, s3, 16
+; GFX1250-NEXT:    s_lshr_b32 s4, s0, 24
 ; GFX1250-NEXT:    s_lshr_b32 s5, s0, 16
 ; GFX1250-NEXT:    s_lshr_b32 s6, s1, 24
 ; GFX1250-NEXT:    s_lshr_b32 s7, s1, 16
-; GFX1250-NEXT:    s_add_co_i32 s6, s6, s6
+; GFX1250-NEXT:    s_add_co_i32 s11, s11, s11
+; GFX1250-NEXT:    s_add_co_i32 s10, s10, s10
 ; GFX1250-NEXT:    s_add_co_i32 s7, s7, s7
+; GFX1250-NEXT:    s_add_co_i32 s6, s6, s6
 ; GFX1250-NEXT:    s_add_co_i32 s5, s5, s5
 ; GFX1250-NEXT:    s_add_co_i32 s4, s4, s4
 ; GFX1250-NEXT:    v_lshlrev_b32_e32 v6, 16, v3
-; GFX1250-NEXT:    v_perm_b32 v3, s6, s7, v1
-; GFX1250-NEXT:    v_perm_b32 v1, s4, s5, v1
+; GFX1250-NEXT:    v_perm_b32 v2, s10, s11, v1
 ; GFX1250-NEXT:    v_mov_b32_e32 v0, 0xc0c0400
 ; GFX1250-NEXT:    s_bfe_u32 s14, s2, 0x80008
 ; GFX1250-NEXT:    s_bfe_u32 s15, s3, 0x80008
+; GFX1250-NEXT:    v_perm_b32 v3, s6, s7, v1
+; GFX1250-NEXT:    v_perm_b32 v1, s4, s5, v1
 ; GFX1250-NEXT:    s_bfe_u32 s12, s0, 0x80008
 ; GFX1250-NEXT:    s_bfe_u32 s13, s1, 0x80008
 ; GFX1250-NEXT:    s_add_co_i32 s3, s3, s3
@@ -2349,11 +2351,11 @@ define amdgpu_kernel void @amd_kernel_v16i8(<16 x i8> %arg0) {
 ; GFX1250-NEXT:    s_add_co_i32 s13, s13, s13
 ; GFX1250-NEXT:    s_add_co_i32 s0, s0, s0
 ; GFX1250-NEXT:    s_add_co_i32 s12, s12, s12
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v9, 16, v1
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
 ; GFX1250-NEXT:    v_perm_b32 v4, s15, s3, v0
 ; GFX1250-NEXT:    v_perm_b32 v5, s14, s2, v0
 ; GFX1250-NEXT:    v_perm_b32 v7, s13, s1, v0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v8, 16, v3
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v3 :: v_dual_lshlrev_b32 v9, 16, v1
 ; GFX1250-NEXT:    v_perm_b32 v0, s12, s0, v0
 ; GFX1250-NEXT:    v_or_b32_e32 v3, v4, v2
 ; GFX1250-NEXT:    v_or_b32_e32 v2, v5, v6
diff --git a/llvm/test/CodeGen/AMDGPU/carryout-selection.ll b/llvm/test/CodeGen/AMDGPU/carryout-selection.ll
index d3fa414cfda69..9a7687b72a779 100644
--- a/llvm/test/CodeGen/AMDGPU/carryout-selection.ll
+++ b/llvm/test/CodeGen/AMDGPU/carryout-selection.ll
@@ -404,10 +404,11 @@ define amdgpu_kernel void @vadd64rr(ptr addrspace(1) %out, i64 %a) {
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX1250-NEXT:    s_wait_xcnt 0x0
+; GFX1250-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[2:3], s[2:3], v[0:1]
 ; GFX1250-NEXT:    global_store_b64 v1, v[2:3], s[0:1]
 ; GFX1250-NEXT:    s_endpgm
@@ -415,11 +416,12 @@ define amdgpu_kernel void @vadd64rr(ptr addrspace(1) %out, i64 %a) {
 ; GFX13-LABEL: vadd64rr:
 ; GFX13:       ; %bb.0: ; %entry
 ; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX13-NEXT:    v_add_co_u32 v0, s2, s2, v0
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-NEXT:    v_add_co_ci_u32_e64 v1, null, s3, 0, s2
 ; GFX13-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
 ; GFX13-NEXT:    s_endpgm
@@ -539,9 +541,9 @@ define amdgpu_kernel void @vadd64ri(ptr addrspace(1) %out) {
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[2:3], 0x123456789876, v[0:1]
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_store_b64 v1, v[2:3], s[0:1]
@@ -550,10 +552,11 @@ define amdgpu_kernel void @vadd64ri(ptr addrspace(1) %out) {
 ; GFX13-LABEL: vadd64ri:
 ; GFX13:       ; %bb.0: ; %entry
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX13-NEXT:    v_add_co_u32 v0, s2, 0x56789876, v0
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-NEXT:    v_add_co_ci_u32_e64 v1, null, 0x1234, 0, s2
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
@@ -1257,13 +1260,14 @@ define amdgpu_kernel void @vuaddo64(ptr addrspace(1) %out, ptr addrspace(1) %car
 ; GFX1250-NEXT:    s_clause 0x1
 ; GFX1250-NEXT:    s_load_b64 s[6:7], s[4:5], 0x34 nv
 ; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    v_mov_b32_e32 v2, 0
+; GFX1250-NEXT:    s_wait_xcnt 0x0
+; GFX1250-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_co_u32 v0, s4, s6, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_co_ci_u32_e64 v1, s4, s7, 0, s4
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v3, 0, 1, s4
 ; GFX1250-NEXT:    s_clause 0x1
 ; GFX1250-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
@@ -1275,13 +1279,13 @@ define amdgpu_kernel void @vuaddo64(ptr addrspace(1) %out, ptr addrspace(1) %car
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[6:7], s[4:5], 0x34 nv
 ; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX13-NEXT:    v_add_co_u32 v0, s4, s6, v0
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX13-NEXT:    v_add_co_ci_u32_e64 v1, s4, s7, 0, s4
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-NEXT:    v_cndmask_b32_e64 v3, 0, 1, s4
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
@@ -1727,10 +1731,11 @@ define amdgpu_kernel void @vsub64rr(ptr addrspace(1) %out, i64 %a) {
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX1250-NEXT:    s_wait_xcnt 0x0
+; GFX1250-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_sub_nc_u64_e32 v[2:3], s[2:3], v[0:1]
 ; GFX1250-NEXT:    global_store_b64 v1, v[2:3], s[0:1]
 ; GFX1250-NEXT:    s_endpgm
@@ -1738,11 +1743,12 @@ define amdgpu_kernel void @vsub64rr(ptr addrspace(1) %out, i64 %a) {
 ; GFX13-LABEL: vsub64rr:
 ; GFX13:       ; %bb.0: ; %entry
 ; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX13-NEXT:    v_sub_co_u32 v0, s2, s2, v0
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-NEXT:    v_sub_co_ci_u32_e64 v1, null, s3, 0, s2
 ; GFX13-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
 ; GFX13-NEXT:    s_endpgm
@@ -1862,9 +1868,9 @@ define amdgpu_kernel void @vsub64ri(ptr addrspace(1) %out) {
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1250-NEXT:    v_sub_nc_u64_e32 v[2:3], 0x123456789876, v[0:1]
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_store_b64 v1, v[2:3], s[0:1]
@@ -1873,10 +1879,11 @@ define amdgpu_kernel void @vsub64ri(ptr addrspace(1) %out) {
 ; GFX13-LABEL: vsub64ri:
 ; GFX13:       ; %bb.0: ; %entry
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX13-NEXT:    v_sub_co_u32 v0, s2, 0x56789876, v0
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-NEXT:    v_sub_co_ci_u32_e64 v1, null, 0x1234, 0, s2
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
@@ -2581,13 +2588,14 @@ define amdgpu_kernel void @vusubo64(ptr addrspace(1) %out, ptr addrspace(1) %car
 ; GFX1250-NEXT:    s_clause 0x1
 ; GFX1250-NEXT:    s_load_b64 s[6:7], s[4:5], 0x34 nv
 ; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    v_mov_b32_e32 v2, 0
+; GFX1250-NEXT:    s_wait_xcnt 0x0
+; GFX1250-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_sub_co_u32 v0, s4, s6, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_sub_co_ci_u32_e64 v1, s4, s7, 0, s4
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v3, 0, 1, s4
 ; GFX1250-NEXT:    s_clause 0x1
 ; GFX1250-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
@@ -2599,13 +2607,13 @@ define amdgpu_kernel void @vusubo64(ptr addrspace(1) %out, ptr addrspace(1) %car
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[6:7], s[4:5], 0x34 nv
 ; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX13-NEXT:    v_sub_co_u32 v0, s4, s6, v0
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX13-NEXT:    v_sub_co_ci_u32_e64 v1, s4, s7, 0, s4
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-NEXT:    v_cndmask_b32_e64 v3, 0, 1, s4
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/ds_read2-gfx1250.ll b/llvm/test/CodeGen/AMDGPU/ds_read2-gfx1250.ll
index 33634dd0824d3..65a2a008aae68 100644
--- a/llvm/test/CodeGen/AMDGPU/ds_read2-gfx1250.ll
+++ b/llvm/test/CodeGen/AMDGPU/ds_read2-gfx1250.ll
@@ -805,13 +805,14 @@ define amdgpu_kernel void @sgemm_inner_loop_read2_sequence(ptr addrspace(1) %C,
 ; GFX1250-UNALIGNED-NEXT:    ds_load_2addr_b32 v[4:5], v4 offset1:1
 ; GFX1250-UNALIGNED-NEXT:    s_wait_dscnt 0x1
 ; GFX1250-UNALIGNED-NEXT:    v_dual_lshrrev_b32 v0, 8, v0 :: v_dual_add_f32 v2, v2, v3
-; GFX1250-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
-; GFX1250-UNALIGNED-NEXT:    v_and_b32_e32 v8, 0xffc, v0
+; GFX1250-UNALIGNED-NEXT:    s_wait_dscnt 0x0
+; GFX1250-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-UNALIGNED-NEXT:    v_add_f32_e32 v2, v2, v4
+; GFX1250-UNALIGNED-NEXT:    s_mov_b32 s2, 0xffc
+; GFX1250-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-UNALIGNED-NEXT:    v_dual_add_f32 v4, v2, v5 :: v_dual_bitop2_b32 v8, s2, v0 bitop3:0x40
 ; GFX1250-UNALIGNED-NEXT:    ds_load_2addr_b32 v[0:1], v8 offset1:1
 ; GFX1250-UNALIGNED-NEXT:    ds_load_2addr_b32 v[6:7], v8 offset0:32 offset1:33
-; GFX1250-UNALIGNED-NEXT:    s_wait_dscnt 0x2
-; GFX1250-UNALIGNED-NEXT:    v_add_f32_e32 v2, v2, v4
-; GFX1250-UNALIGNED-NEXT:    v_add_f32_e32 v4, v2, v5
 ; GFX1250-UNALIGNED-NEXT:    ds_load_2addr_b32 v[2:3], v8 offset0:64 offset1:65
 ; GFX1250-UNALIGNED-NEXT:    s_wait_dscnt 0x2
 ; GFX1250-UNALIGNED-NEXT:    v_add_f32_e32 v0, v4, v0
@@ -844,33 +845,32 @@ define amdgpu_kernel void @sgemm_inner_loop_read2_sequence(ptr addrspace(1) %C,
 ; GFX1250S-UNALIGNED-NEXT:    s_add_co_i32 s1, s1, s0
 ; GFX1250S-UNALIGNED-NEXT:    s_cmp_eq_u32 s2, 0
 ; GFX1250S-UNALIGNED-NEXT:    s_cselect_b32 s0, ttmp9, s1
-; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
-; GFX1250S-UNALIGNED-NEXT:    v_and_b32_e32 v8, 0xffc, v0
+; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
 ; GFX1250S-UNALIGNED-NEXT:    s_lshl_b32 s0, s0, 2
 ; GFX1250S-UNALIGNED-NEXT:    v_mov_b32_e32 v1, s0
+; GFX1250S-UNALIGNED-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
 ; GFX1250S-UNALIGNED-NEXT:    ds_load_b64 v[2:3], v1 offset:3104
 ; GFX1250S-UNALIGNED-NEXT:    ds_load_b64 v[4:5], v1 offset:3168
-; GFX1250S-UNALIGNED-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
+; GFX1250S-UNALIGNED-NEXT:    s_mov_b32 s2, 0xffc
+; GFX1250S-UNALIGNED-NEXT:    s_wait_dscnt 0x1
+; GFX1250S-UNALIGNED-NEXT:    v_dual_add_f32 v2, v2, v3 :: v_dual_bitop2_b32 v8, s2, v0 bitop3:0x40
 ; GFX1250S-UNALIGNED-NEXT:    ds_load_b64 v[0:1], v8
 ; GFX1250S-UNALIGNED-NEXT:    ds_load_b64 v[6:7], v8 offset:128
-; GFX1250S-UNALIGNED-NEXT:    s_wait_dscnt 0x3
-; GFX1250S-UNALIGNED-NEXT:    v_add_f32_e32 v2, v2, v3
 ; GFX1250S-UNALIGNED-NEXT:    s_wait_dscnt 0x2
-; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250S-UNALIGNED-NEXT:    v_add_f32_e32 v2, v2, v4
+; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
 ; GFX1250S-UNALIGNED-NEXT:    v_add_f32_e32 v4, v2, v5
 ; GFX1250S-UNALIGNED-NEXT:    ds_load_b64 v[2:3], v8 offset:256
 ; GFX1250S-UNALIGNED-NEXT:    s_wait_dscnt 0x2
 ; GFX1250S-UNALIGNED-NEXT:    v_add_f32_e32 v0, v4, v0
-; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1250S-UNALIGNED-NEXT:    v_dual_add_f32 v0, v0, v1 :: v_dual_mov_b32 v1, 0
 ; GFX1250S-UNALIGNED-NEXT:    s_wait_dscnt 0x1
+; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250S-UNALIGNED-NEXT:    v_add_f32_e32 v0, v0, v6
-; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1250S-UNALIGNED-NEXT:    v_add_f32_e32 v0, v0, v7
 ; GFX1250S-UNALIGNED-NEXT:    s_wait_dscnt 0x0
+; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250S-UNALIGNED-NEXT:    v_add_f32_e32 v0, v0, v2
-; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250S-UNALIGNED-NEXT:    v_add_f32_e32 v0, v0, v3
 ; GFX1250S-UNALIGNED-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250S-UNALIGNED-NEXT:    global_store_b32 v1, v0, s[0:1]
@@ -994,11 +994,11 @@ define amdgpu_kernel void @ds_read_diff_base_interleaving(
 ; GFX1250-UNALIGNED-NEXT:    s_load_b128 s[0:3], s[4:5], 0x8 nv
 ; GFX1250-UNALIGNED-NEXT:    v_dual_lshrrev_b32 v1, 6, v0 :: v_dual_lshlrev_b32 v0, 2, v0
 ; GFX1250-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1250-UNALIGNED-NEXT:    v_and_b32_e32 v1, 0x3ff0, v1
 ; GFX1250-UNALIGNED-NEXT:    v_and_b32_e32 v0, 0xffc, v0
+; GFX1250-UNALIGNED-NEXT:    v_and_b32_e32 v1, 0x3ff0, v1
 ; GFX1250-UNALIGNED-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX1250-UNALIGNED-NEXT:    v_dual_add_nc_u32 v2, s0, v1 :: v_dual_add_nc_u32 v3, s1, v0
+; GFX1250-UNALIGNED-NEXT:    v_dual_add_nc_u32 v3, s1, v0 :: v_dual_add_nc_u32 v2, s0, v1
 ; GFX1250-UNALIGNED-NEXT:    v_dual_add_nc_u32 v4, s2, v1 :: v_dual_add_nc_u32 v6, s3, v0
 ; GFX1250-UNALIGNED-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
 ; GFX1250-UNALIGNED-NEXT:    ds_load_2addr_b32 v[0:1], v2 offset1:1
@@ -1027,11 +1027,11 @@ define amdgpu_kernel void @ds_read_diff_base_interleaving(
 ; GFX1250S-UNALIGNED-NEXT:    s_load_b128 s[0:3], s[4:5], 0x8 nv
 ; GFX1250S-UNALIGNED-NEXT:    v_dual_lshrrev_b32 v1, 6, v0 :: v_dual_lshlrev_b32 v0, 2, v0
 ; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1250S-UNALIGNED-NEXT:    v_and_b32_e32 v1, 0x3ff0, v1
 ; GFX1250S-UNALIGNED-NEXT:    v_and_b32_e32 v0, 0xffc, v0
+; GFX1250S-UNALIGNED-NEXT:    v_and_b32_e32 v1, 0x3ff0, v1
 ; GFX1250S-UNALIGNED-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX1250S-UNALIGNED-NEXT:    v_dual_add_nc_u32 v2, s0, v1 :: v_dual_add_nc_u32 v3, s1, v0
+; GFX1250S-UNALIGNED-NEXT:    v_dual_add_nc_u32 v3, s1, v0 :: v_dual_add_nc_u32 v2, s0, v1
 ; GFX1250S-UNALIGNED-NEXT:    v_dual_add_nc_u32 v4, s2, v1 :: v_dual_add_nc_u32 v6, s3, v0
 ; GFX1250S-UNALIGNED-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
 ; GFX1250S-UNALIGNED-NEXT:    ds_load_b64 v[0:1], v2
diff --git a/llvm/test/CodeGen/AMDGPU/ds_write2.ll b/llvm/test/CodeGen/AMDGPU/ds_write2.ll
index 3a8b59bb3f900..d76e2445cbf48 100644
--- a/llvm/test/CodeGen/AMDGPU/ds_write2.ll
+++ b/llvm/test/CodeGen/AMDGPU/ds_write2.ll
@@ -1290,10 +1290,11 @@ define amdgpu_kernel void @write2_sgemm_sequence(ptr addrspace(1) %C, i32 %lda,
 ; GFX1250-UNALIGNED-NEXT:    s_add_co_i32 s2, s1, 0xc20
 ; GFX1250-UNALIGNED-NEXT:    v_dual_mov_b32 v1, s2 :: v_dual_lshrrev_b32 v0, 8, v0
 ; GFX1250-UNALIGNED-NEXT:    s_addk_co_i32 s1, 0xc60
+; GFX1250-UNALIGNED-NEXT:    s_mov_b32 s2, 0xffc
+; GFX1250-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-UNALIGNED-NEXT:    v_dual_mov_b32 v4, s1 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1250-UNALIGNED-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-UNALIGNED-NEXT:    v_dual_mov_b32 v4, s1 :: v_dual_mov_b32 v2, s0
-; GFX1250-UNALIGNED-NEXT:    v_mov_b32_e32 v3, s0
-; GFX1250-UNALIGNED-NEXT:    v_and_b32_e32 v0, 0xffc, v0
+; GFX1250-UNALIGNED-NEXT:    v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v3, s0
 ; GFX1250-UNALIGNED-NEXT:    ds_store_2addr_b32 v1, v2, v3 offset1:1
 ; GFX1250-UNALIGNED-NEXT:    ds_store_2addr_b32 v4, v2, v3 offset1:1
 ; GFX1250-UNALIGNED-NEXT:    ds_store_2addr_b32 v0, v2, v3 offset1:1
@@ -1429,14 +1430,15 @@ define amdgpu_kernel void @simple_write2_v4f32_superreg_align4(ptr addrspace(3)
 ; GFX1250-UNALIGNED-NEXT:    s_clause 0x1
 ; GFX1250-UNALIGNED-NEXT:    s_load_b64 s[6:7], s[4:5], 0x8 nv
 ; GFX1250-UNALIGNED-NEXT:    s_load_b32 s8, s[4:5], 0x0 nv
-; GFX1250-UNALIGNED-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
 ; GFX1250-UNALIGNED-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-UNALIGNED-NEXT:    s_load_b128 s[0:3], s[6:7], 0x0
+; GFX1250-UNALIGNED-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX1250-UNALIGNED-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-UNALIGNED-NEXT:    v_dual_mov_b32 v1, s2 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX1250-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-UNALIGNED-NEXT:    v_lshl_add_u32 v0, v0, 4, s8
-; GFX1250-UNALIGNED-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-UNALIGNED-NEXT:    v_dual_mov_b32 v1, s2 :: v_dual_mov_b32 v2, s3
-; GFX1250-UNALIGNED-NEXT:    v_dual_mov_b32 v3, s0 :: v_dual_mov_b32 v4, s1
+; GFX1250-UNALIGNED-NEXT:    v_dual_mov_b32 v2, s3 :: v_dual_mov_b32 v3, s0
+; GFX1250-UNALIGNED-NEXT:    v_mov_b32_e32 v4, s1
 ; GFX1250-UNALIGNED-NEXT:    ds_store_2addr_b32 v0, v1, v2 offset0:2 offset1:3
 ; GFX1250-UNALIGNED-NEXT:    ds_store_2addr_b32 v0, v3, v4 offset1:1
 ; GFX1250-UNALIGNED-NEXT:    s_endpgm
@@ -1450,14 +1452,15 @@ define amdgpu_kernel void @simple_write2_v4f32_superreg_align4(ptr addrspace(3)
 ; GFX1250S-UNALIGNED-NEXT:    s_clause 0x1
 ; GFX1250S-UNALIGNED-NEXT:    s_load_b64 s[6:7], s[4:5], 0x8 nv
 ; GFX1250S-UNALIGNED-NEXT:    s_load_b32 s8, s[4:5], 0x0 nv
-; GFX1250S-UNALIGNED-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
 ; GFX1250S-UNALIGNED-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250S-UNALIGNED-NEXT:    s_load_b128 s[0:3], s[6:7], 0x0
+; GFX1250S-UNALIGNED-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX1250S-UNALIGNED-NEXT:    s_wait_kmcnt 0x0
+; GFX1250S-UNALIGNED-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX1250S-UNALIGNED-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250S-UNALIGNED-NEXT:    v_lshl_add_u32 v4, v0, 4, s8
-; GFX1250S-UNALIGNED-NEXT:    s_wait_kmcnt 0x0
-; GFX1250S-UNALIGNED-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
-; GFX1250S-UNALIGNED-NEXT:    v_dual_mov_b32 v2, s0 :: v_dual_mov_b32 v3, s1
+; GFX1250S-UNALIGNED-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v2, s0
+; GFX1250S-UNALIGNED-NEXT:    v_mov_b32_e32 v3, s1
 ; GFX1250S-UNALIGNED-NEXT:    ds_store_b64 v4, v[0:1] offset:8
 ; GFX1250S-UNALIGNED-NEXT:    ds_store_b64 v4, v[2:3]
 ; GFX1250S-UNALIGNED-NEXT:    s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/fcanonicalize.bf16.ll b/llvm/test/CodeGen/AMDGPU/fcanonicalize.bf16.ll
index 99a0241f2d4a6..f559943070dd7 100644
--- a/llvm/test/CodeGen/AMDGPU/fcanonicalize.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/fcanonicalize.bf16.ll
@@ -721,8 +721,9 @@ define amdgpu_kernel void @v_test_canonicalize_var_v2bf16(ptr addrspace(1) %out)
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
+; GFX1250-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b32 v0, v0, s[0:1] scale_offset
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
@@ -745,13 +746,13 @@ define amdgpu_kernel void @v_test_canonicalize_fabs_var_v2bf16(ptr addrspace(1)
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
+; GFX1250-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b32 v0, v0, s[0:1] scale_offset
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    v_and_b32_e32 v0, 0x7fff7fff, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_pk_mul_bf16 v0, 1.0, v0 op_sel_hi:[0,1]
 ; GFX1250-NEXT:    global_store_b32 v1, v0, s[0:1]
 ; GFX1250-NEXT:    s_endpgm
@@ -772,13 +773,13 @@ define amdgpu_kernel void @v_test_canonicalize_fneg_fabs_var_v2bf16(ptr addrspac
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
+; GFX1250-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b32 v0, v0, s[0:1] scale_offset
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-NEXT:    v_and_b32_e32 v0, 0x7fff7fff, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_pk_mul_bf16 v0, 1.0, v0 op_sel_hi:[0,1] neg_lo:[0,1] neg_hi:[0,1]
 ; GFX1250-NEXT:    global_store_b32 v1, v0, s[0:1]
 ; GFX1250-NEXT:    s_endpgm
@@ -800,8 +801,9 @@ define amdgpu_kernel void @v_test_canonicalize_fneg_var_v2bf16(ptr addrspace(1)
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
+; GFX1250-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    global_load_b32 v0, v0, s[0:1] scale_offset
 ; GFX1250-NEXT:    s_wait_loadcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/fcanonicalize.ll b/llvm/test/CodeGen/AMDGPU/fcanonicalize.ll
index f40bc2700a76e..ff988c73a0dbc 100644
--- a/llvm/test/CodeGen/AMDGPU/fcanonicalize.ll
+++ b/llvm/test/CodeGen/AMDGPU/fcanonicalize.ll
@@ -7567,8 +7567,9 @@ define amdgpu_kernel void @v_test_canonicalize_var_v2f64(ptr addrspace(1) %out)
 ; GFX1251-NEXT:    v_nop
 ; GFX1251-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1251-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
-; GFX1251-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1251-NEXT:    v_mov_b32_e32 v4, 0
+; GFX1251-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1251-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1251-NEXT:    v_dual_mov_b32 v4, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1251-NEXT:    s_wait_kmcnt 0x0
 ; GFX1251-NEXT:    global_load_b128 v[0:3], v0, s[0:1] scale_offset
 ; GFX1251-NEXT:    s_wait_loadcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/flat-saddr-load.ll b/llvm/test/CodeGen/AMDGPU/flat-saddr-load.ll
index e7d4f25c5db83..5fc001c848575 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-saddr-load.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-saddr-load.ll
@@ -6071,8 +6071,9 @@ define amdgpu_ps float @flat_load_saddr_i8_offset_or_i64_imm_offset_4160(ptr add
 ; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX1250-NEXT:    v_or_b32_e32 v0, 0x1040, v0
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
+; GFX1250-NEXT:    s_mov_b32 s0, 0x1040
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x54
 ; GFX1250-NEXT:    flat_load_u8 v0, v[0:1]
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    ; return to shader part epilog
diff --git a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-f16-hw.ll b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-f16-hw.ll
index b9ac9dce0696c..f9561d1821dac 100644
--- a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-f16-hw.ll
+++ b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-f16-hw.ll
@@ -1099,12 +1099,12 @@ define <4 x i8> @to_fp8_v4f16(<4 x half> %x) {
 ; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v0, v0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v2, v1
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v1, 0xffff, v0
-; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v1, v2, 16, v1
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_bitop2_b32 v1, s0, v0 bitop3:0x40
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_lshrrev_b32 v1, 8, v1
-; GFX1250-FAKE16-NEXT:    v_lshrrev_b32_e32 v3, 24, v3
+; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v1, v2, 16, v1
+; GFX1250-FAKE16-NEXT:    v_dual_lshrrev_b32 v3, 24, v3 :: v_dual_lshrrev_b32 v1, 8, v1
 ; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
 ;
 ; GFX950-LABEL: to_fp8_v4f16:
@@ -1335,12 +1335,12 @@ define <4 x i8> @to_fp8_v4f16(<4 x half> %x) {
 ; GFX1310-NEXT:    s_wait_kmcnt 0x0
 ; GFX1310-NEXT:    v_cvt_pk_fp8_f16_e32 v0, v0
 ; GFX1310-NEXT:    v_cvt_pk_fp8_f16_e32 v2, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1310-NEXT:    v_and_b32_e32 v1, 0xffff, v0
-; GFX1310-NEXT:    v_lshl_or_b32 v1, v2, 16, v1
+; GFX1310-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_bitop2_b32 v1, s0, v0 bitop3:0x40
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1310-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_lshrrev_b32 v1, 8, v1
-; GFX1310-NEXT:    v_lshrrev_b32_e32 v3, 24, v3
+; GFX1310-NEXT:    v_lshl_or_b32 v1, v2, 16, v1
+; GFX1310-NEXT:    v_dual_lshrrev_b32 v3, 24, v3 :: v_dual_lshrrev_b32 v1, 8, v1
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
   %r = call <4 x i8> @llvm.convert.to.arbitrary.fp.v4i8.v4f16(<4 x half> %x, metadata !"Float8E4M3FN", metadata !"round.tonearest", i1 false)
   ret <4 x i8> %r
@@ -1367,12 +1367,12 @@ define <4 x i8> @to_bf8_v4f16(<4 x half> %x) {
 ; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_bf8_f16_e64 v0, v0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_bf8_f16_e64 v2, v1
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v1, 0xffff, v0
-; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v1, v2, 16, v1
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_bitop2_b32 v1, s0, v0 bitop3:0x40
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_lshrrev_b32 v1, 8, v1
-; GFX1250-FAKE16-NEXT:    v_lshrrev_b32_e32 v3, 24, v3
+; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v1, v2, 16, v1
+; GFX1250-FAKE16-NEXT:    v_dual_lshrrev_b32 v3, 24, v3 :: v_dual_lshrrev_b32 v1, 8, v1
 ; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
 ;
 ; GFX950-LABEL: to_bf8_v4f16:
@@ -1643,12 +1643,12 @@ define <4 x i8> @to_bf8_v4f16(<4 x half> %x) {
 ; GFX1310-NEXT:    s_wait_kmcnt 0x0
 ; GFX1310-NEXT:    v_cvt_pk_bf8_f16_e32 v0, v0
 ; GFX1310-NEXT:    v_cvt_pk_bf8_f16_e32 v2, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1310-NEXT:    v_and_b32_e32 v1, 0xffff, v0
-; GFX1310-NEXT:    v_lshl_or_b32 v1, v2, 16, v1
+; GFX1310-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_bitop2_b32 v1, s0, v0 bitop3:0x40
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1310-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_lshrrev_b32 v1, 8, v1
-; GFX1310-NEXT:    v_lshrrev_b32_e32 v3, 24, v3
+; GFX1310-NEXT:    v_lshl_or_b32 v1, v2, 16, v1
+; GFX1310-NEXT:    v_dual_lshrrev_b32 v3, 24, v3 :: v_dual_lshrrev_b32 v1, 8, v1
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
   %r = call <4 x i8> @llvm.convert.to.arbitrary.fp.v4i8.v4f16(<4 x half> %x, metadata !"Float8E5M2", metadata !"round.tonearest", i1 false)
   ret <4 x i8> %r
@@ -1718,25 +1718,25 @@ define i8 @to_fp8_f16_towardzero(half %x) {
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b16 v4, 8, v0
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v5, 0, 1, vcc_lo
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v2, v2, 0, vcc_lo
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 7, v1
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v4, 0x80, v4
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_3)
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v3, v3, v5
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v1, v1, 0, vcc_lo
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v5, 0, 8, vcc_lo
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v3, v3, 6
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v1, v4, v1, v5 bitop3:0xfe
 ; GFX1250-FAKE16-NEXT:    v_lshlrev_b16 v6, 3, v3
 ; GFX1250-FAKE16-NEXT:    v_cmp_gt_i16_e32 vcc_lo, 1, v3
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v4, 0x80, v4
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v1, v4, v1, v5 bitop3:0xfe
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v2, v4, v2, v6 bitop3:0xfe
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e32 v1, v2, v1, vcc_lo
 ; GFX1250-FAKE16-NEXT:    v_cmp_eq_f16_e32 vcc_lo, 0, v0
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e32 v1, v1, v4, vcc_lo
 ; GFX1250-FAKE16-NEXT:    v_cmp_o_f16_e32 vcc_lo, v0, v0
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e32 v0, 0x7f, v1, vcc_lo
 ; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
 ;
@@ -1802,25 +1802,25 @@ define i8 @to_fp8_f16_towardzero(half %x) {
 ; GFX1310-NEXT:    v_lshrrev_b16 v4, 8, v0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v5, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v2, v2, 0, vcc_lo
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 7, v1
-; GFX1310-NEXT:    v_and_b32_e32 v4, 0x80, v4
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_add_nc_u16 v3, v3, v5
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v1, v1, 0, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v5, 0, 8, vcc_lo
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_add_nc_u16 v3, v3, 6
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1310-NEXT:    v_bitop3_b16 v1, v4, v1, v5 bitop3:0xfe
 ; GFX1310-NEXT:    v_lshlrev_b16 v6, 3, v3
 ; GFX1310-NEXT:    v_cmp_gt_i16_e32 vcc_lo, 1, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-NEXT:    v_and_b32_e32 v4, 0x80, v4
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    v_bitop3_b16 v1, v4, v1, v5 bitop3:0xfe
 ; GFX1310-NEXT:    v_bitop3_b16 v2, v4, v2, v6 bitop3:0xfe
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, v2, v1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_eq_f16_e32 vcc_lo, 0, v0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, v1, v4, vcc_lo
 ; GFX1310-NEXT:    v_cmp_o_f16_e32 vcc_lo, v0, v0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v0, 0x7f, v1, vcc_lo
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
   %r = call i8 @llvm.convert.to.arbitrary.fp.i8.f16(half %x, metadata !"Float8E4M3FN", metadata !"round.towardzero", i1 false)
@@ -2001,19 +2001,26 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b16 v13, v14, v12
 ; GFX1250-FAKE16-NEXT:    s_movk_i32 s4, 0x7e
 ; GFX1250-FAKE16-NEXT:    v_cmp_ne_u16_e32 vcc_lo, 0, v7
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v17, 1, v20
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v9, v16, v9, v17 bitop3:0xe0
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v5, v5, v7, v21 bitop3:0xe0
-; GFX1250-FAKE16-NEXT:    v_dual_cndmask_b32 v5, 0, v5, s0 :: v_dual_bitop2_b32 v19, 1, v11 bitop3:0x40
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v19, 1, v11
+; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v9, v20, v9
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v5, 0, v5, s0
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v1, v8, v1, v19 bitop3:0xe0
 ; GFX1250-FAKE16-NEXT:    v_sub_nc_u16 v8, v14, 1 clamp
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e64 s0, 7, v9
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v5, v18, v5
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v1, v11, v1
 ; GFX1250-FAKE16-NEXT:    v_lshlrev_b16 v19, v8, 1
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v11, 1, v13
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b16 v8, v8, v12
+; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v9, v9, 0, s0
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 7, v1
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v7, v19, -1
@@ -2026,6 +2033,7 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v2, v2, v7
 ; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v7, 0x80, v3
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v3, v3, s4, 0x80 bitop3:0xec
+; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v11, 1, v13
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v6, 0, 1, vcc_lo
 ; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 7, v5
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v2, v2, 6
@@ -2034,51 +2042,42 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v5, v5, 0, vcc_lo
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v8, 0, 8, vcc_lo
 ; GFX1250-FAKE16-NEXT:    v_cmp_ne_u16_e32 vcc_lo, 0, v14
+; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v11, 0, 1, s0
 ; GFX1250-FAKE16-NEXT:    v_lshlrev_b16 v12, 3, v2
 ; GFX1250-FAKE16-NEXT:    v_cmp_eq_u16_e64 s1, 15, v2
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v5, v7, v5, v8 bitop3:0xfe
-; GFX1250-FAKE16-NEXT:    v_dual_cndmask_b32 v6, 0, v6, vcc_lo :: v_dual_bitop2_b32 v17, 1, v20 bitop3:0x40
+; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e32 v6, 0, v6, vcc_lo
+; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v8, v10, v11
+; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v10, v7, v1, v12 bitop3:0xfe
 ; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 6, v1
-; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v9, v16, v9, v17 bitop3:0xe0
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT:    v_lshrrev_b32_e32 v1, 31, v0
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v6, v13, v6
+; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v8, v8, 6
 ; GFX1250-FAKE16-NEXT:    s_and_b32 s3, s1, vcc_lo
-; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 15, v2
-; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v9, v20, v9
-; GFX1250-FAKE16-NEXT:    s_or_b32 vcc_lo, vcc_lo, s3
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e64 s0, 7, v9
-; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v11, 0, 1, s0
-; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v9, v9, 0, s0
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT:    v_lshlrev_b16 v1, 7, v1
 ; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e64 s0, 7, v6
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
-; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v8, v10, v11
-; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v10, v7, v1, v12 bitop3:0xfe
-; GFX1250-FAKE16-NEXT:    v_lshrrev_b32_e32 v1, 31, v0
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT:    v_lshlrev_b16 v12, 3, v8
+; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 15, v2
+; GFX1250-FAKE16-NEXT:    v_cmp_gt_i16_e64 s2, 1, v8
+; GFX1250-FAKE16-NEXT:    v_cmp_eq_u16_e64 s1, 15, v8
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v6, v6, 0, s0
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v11, 0, 8, s0
-; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v8, v8, 6
 ; GFX1250-FAKE16-NEXT:    v_cmp_gt_i16_e64 s0, 1, v2
-; GFX1250-FAKE16-NEXT:    v_lshlrev_b16 v1, 7, v1
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX1250-FAKE16-NEXT:    v_lshlrev_b16 v12, 3, v8
-; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v5, v10, v5, s0
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT:    s_or_b32 vcc_lo, vcc_lo, s3
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v6, v1, v6, v11 bitop3:0xfe
-; GFX1250-FAKE16-NEXT:    v_cmp_gt_i16_e64 s2, 1, v8
-; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e64 s0, 6, v9
+; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v5, v10, v5, s0
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v10, v1, v9, v12 bitop3:0xfe
-; GFX1250-FAKE16-NEXT:    v_cmp_eq_u16_e64 s1, 15, v8
-; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e32 v5, v5, v3, vcc_lo
-; GFX1250-FAKE16-NEXT:    v_cmp_eq_f16_e32 vcc_lo, 0, v4
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
-; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v2, v10, v6, s2
+; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e64 s0, 6, v9
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX1250-FAKE16-NEXT:    v_dual_cndmask_b32 v5, v5, v3, vcc_lo :: v_dual_cndmask_b32 v2, v10, v6, s2
 ; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e64 s2, 15, v8
 ; GFX1250-FAKE16-NEXT:    v_or_b32_e32 v6, 0x7e, v1
 ; GFX1250-FAKE16-NEXT:    s_and_b32 s0, s1, s0
+; GFX1250-FAKE16-NEXT:    v_cmp_eq_f16_e32 vcc_lo, 0, v4
 ; GFX1250-FAKE16-NEXT:    s_or_b32 s0, s2, s0
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v2, v2, v6, s0
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e32 v1, v2, v1, vcc_lo
@@ -2278,19 +2277,26 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
 ; GFX1310-NEXT:    v_lshrrev_b16 v13, v14, v12
 ; GFX1310-NEXT:    s_movk_i32 s4, 0x7e
 ; GFX1310-NEXT:    v_cmp_ne_u16_e32 vcc_lo, 0, v7
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_and_b32_e32 v17, 1, v20
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-NEXT:    v_bitop3_b16 v9, v16, v9, v17 bitop3:0xe0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_bitop3_b16 v5, v5, v7, v21 bitop3:0xe0
-; GFX1310-NEXT:    v_dual_cndmask_b32 v5, 0, v5, s0 :: v_dual_bitop2_b32 v19, 1, v11 bitop3:0x40
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_and_b32_e32 v19, 1, v11
+; GFX1310-NEXT:    v_add_nc_u16 v9, v20, v9
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_cndmask_b32_e64 v5, 0, v5, s0
 ; GFX1310-NEXT:    v_bitop3_b16 v1, v8, v1, v19 bitop3:0xe0
 ; GFX1310-NEXT:    v_sub_nc_u16 v8, v14, 1 clamp
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    v_cmp_lt_i16_e64 s0, 7, v9
 ; GFX1310-NEXT:    v_add_nc_u16 v5, v18, v5
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_add_nc_u16 v1, v11, v1
 ; GFX1310-NEXT:    v_lshlrev_b16 v19, v8, 1
-; GFX1310-NEXT:    v_and_b32_e32 v11, 1, v13
 ; GFX1310-NEXT:    v_lshrrev_b16 v8, v8, v12
+; GFX1310-NEXT:    v_cndmask_b32_e64 v9, v9, 0, s0
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 7, v1
 ; GFX1310-NEXT:    v_add_nc_u16 v7, v19, -1
@@ -2303,6 +2309,7 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
 ; GFX1310-NEXT:    v_add_nc_u16 v2, v2, v7
 ; GFX1310-NEXT:    v_and_b32_e32 v7, 0x80, v3
 ; GFX1310-NEXT:    v_bitop3_b16 v3, v3, s4, 0x80 bitop3:0xec
+; GFX1310-NEXT:    v_and_b32_e32 v11, 1, v13
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v6, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 7, v5
 ; GFX1310-NEXT:    v_add_nc_u16 v2, v2, 6
@@ -2311,51 +2318,42 @@ define <2 x i8> @to_fp8_v2f16_saturate(<2 x half> %x) {
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v5, v5, 0, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v8, 0, 8, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u16_e32 vcc_lo, 0, v14
+; GFX1310-NEXT:    v_cndmask_b32_e64 v11, 0, 1, s0
 ; GFX1310-NEXT:    v_lshlrev_b16 v12, 3, v2
 ; GFX1310-NEXT:    v_cmp_eq_u16_e64 s1, 15, v2
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bitop3_b16 v5, v7, v5, v8 bitop3:0xfe
-; GFX1310-NEXT:    v_dual_cndmask_b32 v6, 0, v6, vcc_lo :: v_dual_bitop2_b32 v17, 1, v20 bitop3:0x40
+; GFX1310-NEXT:    v_cndmask_b32_e32 v6, 0, v6, vcc_lo
+; GFX1310-NEXT:    v_add_nc_u16 v8, v10, v11
+; GFX1310-NEXT:    v_bitop3_b16 v10, v7, v1, v12 bitop3:0xfe
 ; GFX1310-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 6, v1
-; GFX1310-NEXT:    v_bitop3_b16 v9, v16, v9, v17 bitop3:0xe0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v1, 31, v0
 ; GFX1310-NEXT:    v_add_nc_u16 v6, v13, v6
+; GFX1310-NEXT:    v_add_nc_u16 v8, v8, 6
 ; GFX1310-NEXT:    s_and_b32 s3, s1, vcc_lo
-; GFX1310-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 15, v2
-; GFX1310-NEXT:    v_add_nc_u16 v9, v20, v9
-; GFX1310-NEXT:    s_or_b32 vcc_lo, vcc_lo, s3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1310-NEXT:    v_cmp_lt_i16_e64 s0, 7, v9
-; GFX1310-NEXT:    v_cndmask_b32_e64 v11, 0, 1, s0
-; GFX1310-NEXT:    v_cndmask_b32_e64 v9, v9, 0, s0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_lshlrev_b16 v1, 7, v1
 ; GFX1310-NEXT:    v_cmp_lt_i16_e64 s0, 7, v6
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
-; GFX1310-NEXT:    v_add_nc_u16 v8, v10, v11
-; GFX1310-NEXT:    v_bitop3_b16 v10, v7, v1, v12 bitop3:0xfe
-; GFX1310-NEXT:    v_lshrrev_b32_e32 v1, 31, v0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3)
+; GFX1310-NEXT:    v_lshlrev_b16 v12, 3, v8
+; GFX1310-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 15, v2
+; GFX1310-NEXT:    v_cmp_gt_i16_e64 s2, 1, v8
+; GFX1310-NEXT:    v_cmp_eq_u16_e64 s1, 15, v8
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v6, v6, 0, s0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v11, 0, 8, s0
-; GFX1310-NEXT:    v_add_nc_u16 v8, v8, 6
 ; GFX1310-NEXT:    v_cmp_gt_i16_e64 s0, 1, v2
-; GFX1310-NEXT:    v_lshlrev_b16 v1, 7, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX1310-NEXT:    v_lshlrev_b16 v12, 3, v8
-; GFX1310-NEXT:    v_cndmask_b32_e64 v5, v10, v5, s0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3)
+; GFX1310-NEXT:    s_or_b32 vcc_lo, vcc_lo, s3
 ; GFX1310-NEXT:    v_bitop3_b16 v6, v1, v6, v11 bitop3:0xfe
-; GFX1310-NEXT:    v_cmp_gt_i16_e64 s2, 1, v8
-; GFX1310-NEXT:    v_cmp_lt_i16_e64 s0, 6, v9
+; GFX1310-NEXT:    v_cndmask_b32_e64 v5, v10, v5, s0
 ; GFX1310-NEXT:    v_bitop3_b16 v10, v1, v9, v12 bitop3:0xfe
-; GFX1310-NEXT:    v_cmp_eq_u16_e64 s1, 15, v8
-; GFX1310-NEXT:    v_cndmask_b32_e32 v5, v5, v3, vcc_lo
-; GFX1310-NEXT:    v_cmp_eq_f16_e32 vcc_lo, 0, v4
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
-; GFX1310-NEXT:    v_cndmask_b32_e64 v2, v10, v6, s2
+; GFX1310-NEXT:    v_cmp_lt_i16_e64 s0, 6, v9
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX1310-NEXT:    v_dual_cndmask_b32 v5, v5, v3, vcc_lo :: v_dual_cndmask_b32 v2, v10, v6, s2
 ; GFX1310-NEXT:    v_cmp_lt_i16_e64 s2, 15, v8
 ; GFX1310-NEXT:    v_or_b32_e32 v6, 0x7e, v1
 ; GFX1310-NEXT:    s_and_b32 s0, s1, s0
+; GFX1310-NEXT:    v_cmp_eq_f16_e32 vcc_lo, 0, v4
 ; GFX1310-NEXT:    s_or_b32 s0, s2, s0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v2, v2, v6, s0
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, v2, v1, vcc_lo
diff --git a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-hw.ll b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-hw.ll
index cc038e9cb65ea..782879898bf40 100644
--- a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-hw.ll
+++ b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-fp8-hw.ll
@@ -620,53 +620,55 @@ define i8 @to_fp8_f32_saturate(float %x) {
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    v_frexp_exp_i32_f32_e32 v1, v0
 ; GFX1250-NEXT:    v_frexp_mant_f32_e32 v2, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
-; GFX1250-NEXT:    v_dual_sub_nc_u32 v3, 15, v1 :: v_dual_lshrrev_b32 v8, 19, v2
-; GFX1250-NEXT:    v_and_b32_e32 v4, 0x7fffff, v2
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0x7ffff, v2
-; GFX1250-NEXT:    v_bfe_u32 v2, v2, 20, 3
+; GFX1250-NEXT:    s_mov_b32 s0, 0x7fffff
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_sub_nc_u32 v3, 15, v1 :: v_dual_bitop2_b32 v4, s0, v2 bitop3:0x40
+; GFX1250-NEXT:    v_lshrrev_b32_e32 v8, 19, v2
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_min_u32_e32 v3, 31, v3
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-NEXT:    v_or_b32_e32 v5, 0x800000, v4
-; GFX1250-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v7
 ; GFX1250-NEXT:    v_bfe_u32 v4, v4, 20, 1
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1250-NEXT:    v_sub_nc_u32_e64 v6, v3, 1 clamp
+; GFX1250-NEXT:    v_and_b32_e32 v7, 0x7ffff, v2
 ; GFX1250-NEXT:    v_bfe_u32 v10, v5, v3, 1
-; GFX1250-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1250-NEXT:    v_bfe_u32 v2, v2, 20, 3
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_4)
 ; GFX1250-NEXT:    v_bfe_u32 v9, v5, 0, v6
-; GFX1250-NEXT:    v_dual_lshrrev_b32 v6, v6, v5 :: v_dual_lshrrev_b32 v5, v3, v5
-; GFX1250-NEXT:    v_bitop3_b32 v4, v8, v7, v4 bitop3:0xe0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_lshrrev_b32_e32 v6, v6, v5
+; GFX1250-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v7
+; GFX1250-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
 ; GFX1250-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v9
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
+; GFX1250-NEXT:    v_bitop3_b32 v4, v8, v7, v4 bitop3:0xe0
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v9, 0, 1, vcc_lo
 ; GFX1250-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v3
+; GFX1250-NEXT:    v_dual_lshrrev_b32 v5, v3, v5 :: v_dual_add_nc_u32 v2, v2, v4
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_bitop3_b32 v6, v6, v9, v10 bitop3:0xe0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-NEXT:    v_dual_cndmask_b32 v3, 0, v6 :: v_dual_add_nc_u32 v2, v2, v4
+; GFX1250-NEXT:    v_dual_cndmask_b32 v3, 0, v6 :: v_dual_lshrrev_b32 v4, 24, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3)
 ; GFX1250-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v2
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
-; GFX1250-NEXT:    v_dual_lshrrev_b32 v4, 24, v0 :: v_dual_add_nc_u32 v3, v5, v3
+; GFX1250-NEXT:    s_mov_b32 s0, 0x80
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_add_nc_u32 v3, v5, v3 :: v_dual_bitop2_b32 v4, s0, v4 bitop3:0x40
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v2, v2, 0, vcc_lo
 ; GFX1250-NEXT:    v_add_co_ci_u32_e64 v1, null, 6, v1, vcc_lo
-; GFX1250-NEXT:    v_and_b32_e32 v4, 0x80, v4
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v3
-; GFX1250-NEXT:    v_cmp_eq_u32_e64 s0, 15, v1
-; GFX1250-NEXT:    v_cmp_lt_i32_e64 s1, 15, v1
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v5, 3, v1
 ; GFX1250-NEXT:    v_cmp_gt_i32_e64 s2, 1, v1
+; GFX1250-NEXT:    v_cmp_eq_u32_e64 s0, 15, v1
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v3, v3, 0, vcc_lo
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v6, 0, 8, vcc_lo
+; GFX1250-NEXT:    v_or3_b32 v5, v4, v5, v2
 ; GFX1250-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 6, v2
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_cmp_lt_i32_e64 s1, 15, v1
+; GFX1250-NEXT:    v_or_b32_e32 v2, 0x7e, v4
 ; GFX1250-NEXT:    v_or3_b32 v3, v4, v6, v3
 ; GFX1250-NEXT:    s_and_b32 s0, s0, vcc_lo
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    s_or_b32 vcc_lo, s1, s0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v5, 3, v1
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
-; GFX1250-NEXT:    v_or3_b32 v5, v4, v5, v2
-; GFX1250-NEXT:    v_or_b32_e32 v2, 0x7e, v4
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v1, v5, v3, s2
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v1, v1, v2, vcc_lo
@@ -688,53 +690,55 @@ define i8 @to_fp8_f32_saturate(float %x) {
 ; GFX1310-NEXT:    s_wait_kmcnt 0x0
 ; GFX1310-NEXT:    v_frexp_exp_i32_f32_e32 v1, v0
 ; GFX1310-NEXT:    v_frexp_mant_f32_e32 v2, v0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
-; GFX1310-NEXT:    v_dual_sub_nc_u32 v3, 15, v1 :: v_dual_lshrrev_b32 v8, 19, v2
-; GFX1310-NEXT:    v_and_b32_e32 v4, 0x7fffff, v2
-; GFX1310-NEXT:    v_and_b32_e32 v7, 0x7ffff, v2
-; GFX1310-NEXT:    v_bfe_u32 v2, v2, 20, 3
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7fffff
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_sub_nc_u32 v3, 15, v1 :: v_dual_bitop2_b32 v4, s0, v2 bitop3:0x40
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v8, 19, v2
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_min_u32_e32 v3, 31, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_or_b32_e32 v5, 0x800000, v4
-; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v7
 ; GFX1310-NEXT:    v_bfe_u32 v4, v4, 20, 1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_sub_nc_u32_e64 v6, v3, 1 clamp
+; GFX1310-NEXT:    v_and_b32_e32 v7, 0x7ffff, v2
 ; GFX1310-NEXT:    v_bfe_u32 v10, v5, v3, 1
-; GFX1310-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_bfe_u32 v2, v2, 20, 3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_bfe_u32 v9, v5, 0, v6
-; GFX1310-NEXT:    v_dual_lshrrev_b32 v6, v6, v5 :: v_dual_lshrrev_b32 v5, v3, v5
-; GFX1310-NEXT:    v_bitop3_b32 v4, v8, v7, v4 bitop3:0xe0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v6, v6, v5
+; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v7
+; GFX1310-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v9
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_bitop3_b32 v4, v8, v7, v4 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v9, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v3
+; GFX1310-NEXT:    v_dual_lshrrev_b32 v5, v3, v5 :: v_dual_add_nc_u32 v2, v2, v4
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_bitop3_b32 v6, v6, v9, v10 bitop3:0xe0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1310-NEXT:    v_dual_cndmask_b32 v3, 0, v6 :: v_dual_add_nc_u32 v2, v2, v4
+; GFX1310-NEXT:    v_dual_cndmask_b32 v3, 0, v6 :: v_dual_lshrrev_b32 v4, 24, v0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3)
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v2
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
-; GFX1310-NEXT:    v_dual_lshrrev_b32 v4, 24, v0 :: v_dual_add_nc_u32 v3, v5, v3
+; GFX1310-NEXT:    s_mov_b32 s0, 0x80
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_add_nc_u32 v3, v5, v3 :: v_dual_bitop2_b32 v4, s0, v4 bitop3:0x40
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v2, v2, 0, vcc_lo
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v1, null, 6, v1, vcc_lo
-; GFX1310-NEXT:    v_and_b32_e32 v4, 0x80, v4
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v3
-; GFX1310-NEXT:    v_cmp_eq_u32_e64 s0, 15, v1
-; GFX1310-NEXT:    v_cmp_lt_i32_e64 s1, 15, v1
+; GFX1310-NEXT:    v_lshlrev_b32_e32 v5, 3, v1
 ; GFX1310-NEXT:    v_cmp_gt_i32_e64 s2, 1, v1
+; GFX1310-NEXT:    v_cmp_eq_u32_e64 s0, 15, v1
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v3, v3, 0, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v6, 0, 8, vcc_lo
+; GFX1310-NEXT:    v_or3_b32 v5, v4, v5, v2
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 6, v2
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_cmp_lt_i32_e64 s1, 15, v1
+; GFX1310-NEXT:    v_or_b32_e32 v2, 0x7e, v4
 ; GFX1310-NEXT:    v_or3_b32 v3, v4, v6, v3
 ; GFX1310-NEXT:    s_and_b32 s0, s0, vcc_lo
+; GFX1310-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    s_or_b32 vcc_lo, s1, s0
-; GFX1310-NEXT:    v_lshlrev_b32_e32 v5, 3, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
-; GFX1310-NEXT:    v_or3_b32 v5, v4, v5, v2
-; GFX1310-NEXT:    v_or_b32_e32 v2, 0x7e, v4
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v1, v5, v3, s2
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, v1, v2, vcc_lo
@@ -852,19 +856,19 @@ define i8 @to_fp8_f32_towardzero(float %x) {
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_dual_lshrrev_b32 v2, v3, v2 :: v_dual_lshrrev_b32 v3, 24, v0
 ; GFX1250-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v2
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0x80, v3
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v2, v2, 0, vcc_lo
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v6, 0, 8, vcc_lo
-; GFX1250-NEXT:    v_or3_b32 v4, v3, v5, v4
 ; GFX1250-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 1, v1
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_and_b32_e32 v3, 0x80, v3
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    v_or3_b32 v4, v3, v5, v4
 ; GFX1250-NEXT:    v_or3_b32 v2, v3, v6, v2
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v1, v4, v2, vcc_lo
 ; GFX1250-NEXT:    v_cmp_eq_f32_e32 vcc_lo, 0, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v1, v1, v3, vcc_lo
 ; GFX1250-NEXT:    v_cmp_o_f32_e32 vcc_lo, v0, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, 0x7f, v1, vcc_lo
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
 ;
@@ -891,19 +895,19 @@ define i8 @to_fp8_f32_towardzero(float %x) {
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_dual_lshrrev_b32 v2, v3, v2 :: v_dual_lshrrev_b32 v3, 24, v0
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v2
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
-; GFX1310-NEXT:    v_and_b32_e32 v3, 0x80, v3
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v2, v2, 0, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v6, 0, 8, vcc_lo
-; GFX1310-NEXT:    v_or3_b32 v4, v3, v5, v4
 ; GFX1310-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 1, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-NEXT:    v_and_b32_e32 v3, 0x80, v3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    v_or3_b32 v4, v3, v5, v4
 ; GFX1310-NEXT:    v_or3_b32 v2, v3, v6, v2
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, v4, v2, vcc_lo
 ; GFX1310-NEXT:    v_cmp_eq_f32_e32 vcc_lo, 0, v0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, v1, v3, vcc_lo
 ; GFX1310-NEXT:    v_cmp_o_f32_e32 vcc_lo, v0, v0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v0, 0x7f, v1, vcc_lo
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
   %r = call i8 @llvm.convert.to.arbitrary.fp.i8.f32(float %x, metadata !"Float8E4M3FN", metadata !"round.towardzero", i1 false)
@@ -1036,45 +1040,48 @@ define i4 @to_fp4_f32(float %x) {
 ; GFX1250-NEXT:    v_frexp_exp_i32_f32_e32 v1, v0
 ; GFX1250-NEXT:    v_frexp_mant_f32_e32 v3, v0
 ; GFX1250-NEXT:    s_mov_b32 s0, 0x7fffff
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_dual_sub_nc_u32 v2, 23, v1 :: v_dual_lshrrev_b32 v6, 21, v3
 ; GFX1250-NEXT:    v_and_or_b32 v4, v3, s0, 0x800000
-; GFX1250-NEXT:    v_and_b32_e32 v8, 0x1fffff, v3
-; GFX1250-NEXT:    v_bfe_u32 v3, v3, 22, 1
 ; GFX1250-NEXT:    v_min_u32_e32 v2, 31, v2
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1250-NEXT:    v_sub_nc_u32_e64 v5, v2, 1 clamp
+; GFX1250-NEXT:    v_and_b32_e32 v8, 0x1fffff, v3
 ; GFX1250-NEXT:    v_bfe_u32 v9, v4, v2, 1
+; GFX1250-NEXT:    v_bfe_u32 v3, v3, 22, 1
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_bfe_u32 v7, v4, 0, v5
-; GFX1250-NEXT:    v_dual_lshrrev_b32 v5, v5, v4 :: v_dual_lshrrev_b32 v4, v2, v4
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_lshrrev_b32_e32 v5, v5, v4
 ; GFX1250-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v7
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
 ; GFX1250-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v8
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_bitop3_b32 v5, v5, v7, v9 bitop3:0xe0
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v8, 0, 1, vcc_lo
 ; GFX1250-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v2
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_lshrrev_b32_e32 v4, v2, v4
 ; GFX1250-NEXT:    v_bitop3_b32 v6, v6, v8, v3 bitop3:0xe0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_dual_cndmask_b32 v2, 0, v5 :: v_dual_add_nc_u32 v3, v3, v6
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_dual_add_nc_u32 v2, v4, v2 :: v_dual_lshrrev_b32 v4, 28, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 1, v3
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX1250-NEXT:    v_and_b32_e32 v4, 8, v4
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_cmp_lt_i32_e64 s0, 1, v2
 ; GFX1250-NEXT:    v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v3, v3, 0, vcc_lo
-; GFX1250-NEXT:    v_and_b32_e32 v4, 8, v4
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v2, v2, 0, s0
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v5, 0, 2, s0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v6, 1, v1
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 1, v1
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_or3_b32 v2, v4, v5, v2
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v6, 1, v1
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_or3_b32 v3, v4, v6, v3
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v1, v3, v2, vcc_lo
 ; GFX1250-NEXT:    v_cmp_eq_f32_e32 vcc_lo, 0, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, v1, v4, vcc_lo
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
 ;
@@ -1088,45 +1095,48 @@ define i4 @to_fp4_f32(float %x) {
 ; GFX1310-NEXT:    v_frexp_exp_i32_f32_e32 v1, v0
 ; GFX1310-NEXT:    v_frexp_mant_f32_e32 v3, v0
 ; GFX1310-NEXT:    s_mov_b32 s0, 0x7fffff
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_dual_sub_nc_u32 v2, 23, v1 :: v_dual_lshrrev_b32 v6, 21, v3
 ; GFX1310-NEXT:    v_and_or_b32 v4, v3, s0, 0x800000
-; GFX1310-NEXT:    v_and_b32_e32 v8, 0x1fffff, v3
-; GFX1310-NEXT:    v_bfe_u32 v3, v3, 22, 1
 ; GFX1310-NEXT:    v_min_u32_e32 v2, 31, v2
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_sub_nc_u32_e64 v5, v2, 1 clamp
+; GFX1310-NEXT:    v_and_b32_e32 v8, 0x1fffff, v3
 ; GFX1310-NEXT:    v_bfe_u32 v9, v4, v2, 1
+; GFX1310-NEXT:    v_bfe_u32 v3, v3, 22, 1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bfe_u32 v7, v4, 0, v5
-; GFX1310-NEXT:    v_dual_lshrrev_b32 v5, v5, v4 :: v_dual_lshrrev_b32 v4, v2, v4
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v5, v5, v4
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v7
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v8
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_bitop3_b32 v5, v5, v7, v9 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v8, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v2
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v4, v2, v4
 ; GFX1310-NEXT:    v_bitop3_b32 v6, v6, v8, v3 bitop3:0xe0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_dual_cndmask_b32 v2, 0, v5 :: v_dual_add_nc_u32 v3, v3, v6
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_dual_add_nc_u32 v2, v4, v2 :: v_dual_lshrrev_b32 v4, 28, v0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 1, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    v_and_b32_e32 v4, 8, v4
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_cmp_lt_i32_e64 s0, 1, v2
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v1, null, 0, v1, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v3, v3, 0, vcc_lo
-; GFX1310-NEXT:    v_and_b32_e32 v4, 8, v4
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v2, v2, 0, s0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v5, 0, 2, s0
-; GFX1310-NEXT:    v_lshlrev_b32_e32 v6, 1, v1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 1, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_or3_b32 v2, v4, v5, v2
+; GFX1310-NEXT:    v_lshlrev_b32_e32 v6, 1, v1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_or3_b32 v3, v4, v6, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, v3, v2, vcc_lo
 ; GFX1310-NEXT:    v_cmp_eq_f32_e32 vcc_lo, 0, v0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v0, v1, v4, vcc_lo
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
   %r = call i4 @llvm.convert.to.arbitrary.fp.i4.f32(float %x, metadata !"Float4E2M1FN", metadata !"round.tonearest", i1 false)
@@ -1342,10 +1352,10 @@ define i8 @to_fp8_f64(double %x) {
 ; GFX1250-NEXT:    v_frexp_exp_i32_f64_e32 v2, v[0:1]
 ; GFX1250-NEXT:    v_frexp_mant_f64_e32 v[4:5], v[0:1]
 ; GFX1250-NEXT:    s_mov_b32 s0, 0
+; GFX1250-NEXT:    s_mov_b32 s1, 0xfffff
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_ashrrev_i32 v3, 31, v2 :: v_dual_bitop2_b32 v15, s1, v5 bitop3:0x40
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1250-NEXT:    v_ashrrev_i32_e32 v3, 31, v2
-; GFX1250-NEXT:    v_and_b32_e32 v15, 0xfffff, v5
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_mov_b32_e32 v10, v4
 ; GFX1250-NEXT:    v_sub_nc_u64_e32 v[6:7], 37, v[2:3]
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -1368,30 +1378,31 @@ define i8 @to_fp8_f64(double %x) {
 ; GFX1250-NEXT:    v_lshrrev_b64 v[10:11], v14, v[10:11]
 ; GFX1250-NEXT:    v_lshrrev_b32_e32 v14, 17, v5
 ; GFX1250-NEXT:    v_cmp_ne_u64_e32 vcc_lo, 0, v[8:9]
-; GFX1250-NEXT:    v_and_b32_e32 v9, 0x1ffff, v5
 ; GFX1250-NEXT:    v_mov_b32_e32 v8, v4
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v17, 0, 1, vcc_lo
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_or_b32_e32 v4, v17, v16
+; GFX1250-NEXT:    v_and_b32_e32 v4, v10, v4
+; GFX1250-NEXT:    v_and_b32_e32 v9, 0x1ffff, v5
+; GFX1250-NEXT:    v_bfe_u32 v10, v15, 18, 1
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_cmp_ne_u64_e32 vcc_lo, 0, v[8:9]
-; GFX1250-NEXT:    v_dual_mov_b32 v9, 0 :: v_dual_bitop2_b32 v4, v17, v16 bitop3:0x54
+; GFX1250-NEXT:    v_mov_b32_e32 v9, 0
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v11, 0, 1, vcc_lo
 ; GFX1250-NEXT:    v_cmp_ne_u64_e32 vcc_lo, 0, v[6:7]
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
-; GFX1250-NEXT:    v_dual_mov_b32 v7, v9 :: v_dual_bitop2_b32 v4, v10, v4 bitop3:0x40
-; GFX1250-NEXT:    v_bfe_u32 v10, v15, 18, 1
+; GFX1250-NEXT:    v_mov_b32_e32 v7, v9
 ; GFX1250-NEXT:    v_bfe_u32 v6, v5, 18, 2
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_dual_cndmask_b32 v8, 0, v4, vcc_lo :: v_dual_bitop2_b32 v10, v11, v10 bitop3:0x54
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[4:5], v[12:13], v[8:9]
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_dual_lshrrev_b32 v10, 24, v1 :: v_dual_bitop2_b32 v8, v14, v10 bitop3:0x40
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[6:7], v[6:7], v[8:9]
 ; GFX1250-NEXT:    v_mov_b32_e32 v9, s0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_cmp_lt_i64_e64 s0, 3, v[4:5]
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0x80, v10
 ; GFX1250-NEXT:    v_cmp_lt_i64_e32 vcc_lo, 3, v[6:7]
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v4, v4, 0, s0
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v8, 0, 1, vcc_lo
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v7, v7, 0, vcc_lo
@@ -1406,23 +1417,24 @@ define i8 @to_fp8_f64(double %x) {
 ; GFX1250-NEXT:    v_cmp_eq_u64_e64 s0, 30, v[2:3]
 ; GFX1250-NEXT:    v_cmp_lt_i64_e64 s1, 30, v[2:3]
 ; GFX1250-NEXT:    v_cmp_gt_i64_e64 s2, 1, v[2:3]
-; GFX1250-NEXT:    v_or_b32_e32 v3, 0x7c, v5
-; GFX1250-NEXT:    v_or3_b32 v4, v5, v9, v4
 ; GFX1250-NEXT:    s_and_b32 s0, s0, vcc_lo
 ; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    s_or_b32 vcc_lo, s1, s0
+; GFX1250-NEXT:    v_and_b32_e32 v5, 0x80, v10
 ; GFX1250-NEXT:    v_or_b32_e32 v8, v8, v5
+; GFX1250-NEXT:    v_or3_b32 v4, v5, v9, v4
+; GFX1250-NEXT:    v_or_b32_e32 v3, 0x7c, v5
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_or_b32_e32 v8, v8, v6
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_cndmask_b32_e64 v2, v8, v4, s2
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v3, vcc_lo
 ; GFX1250-NEXT:    v_cmp_eq_f64_e32 vcc_lo, 0, v[0:1]
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v5, vcc_lo
 ; GFX1250-NEXT:    v_cmp_class_f64_e64 vcc_lo, v[0:1], 0x204
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v2, v2, v3, vcc_lo
 ; GFX1250-NEXT:    v_cmp_o_f64_e32 vcc_lo, v[0:1], v[0:1]
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250-NEXT:    v_cndmask_b32_e32 v0, 0x7e, v2, vcc_lo
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
 ;
@@ -1438,68 +1450,68 @@ define i8 @to_fp8_f64(double %x) {
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_ashrrev_i32_e32 v13, 31, v12
 ; GFX1310-NEXT:    v_sub_co_u32 v4, vcc_lo, 37, v12
+; GFX1310-NEXT:    v_bfe_u32 v17, v3, 18, 2
 ; GFX1310-NEXT:    v_and_b32_e32 v7, 0x1ffff, v3
-; GFX1310-NEXT:    v_and_b32_e32 v9, 0xfffff, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_sub_co_ci_u32_e64 v5, null, 0, v13, vcc_lo
-; GFX1310-NEXT:    v_bfe_u32 v17, v3, 18, 2
-; GFX1310-NEXT:    v_mov_b32_e32 v6, v2
-; GFX1310-NEXT:    v_bfe_u32 v11, v9, 18, 1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX1310-NEXT:    v_dual_mov_b32 v6, v2 :: v_dual_lshrrev_b32 v15, 17, v3
 ; GFX1310-NEXT:    v_cmp_gt_u64_e32 vcc_lo, 63, v[4:5]
-; GFX1310-NEXT:    v_or_b32_e32 v9, 0x100000, v9
-; GFX1310-NEXT:    v_dual_lshrrev_b32 v15, 17, v3 :: v_dual_cndmask_b32 v5, 0, v5, vcc_lo
-; GFX1310-NEXT:    v_cndmask_b32_e32 v4, 63, v4, vcc_lo
+; GFX1310-NEXT:    v_dual_cndmask_b32 v5, 0, v5 :: v_dual_cndmask_b32 v4, 63, v4
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
 ; GFX1310-NEXT:    v_cmp_lt_u64_e32 vcc_lo, 1, v[4:5]
-; GFX1310-NEXT:    v_cndmask_b32_e32 v8, 1, v4, vcc_lo
+; GFX1310-NEXT:    s_mov_b32 s0, 0xfffff
+; GFX1310-NEXT:    v_dual_cndmask_b32 v8, 1, v4, vcc_lo :: v_dual_bitop2_b32 v9, s0, v3 bitop3:0x40
 ; GFX1310-NEXT:    v_cmp_ne_u64_e32 vcc_lo, 0, v[6:7]
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
-; GFX1310-NEXT:    v_dual_add_nc_u32 v14, -1, v8 :: v_dual_mov_b32 v8, v2
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_add_nc_u32_e32 v14, -1, v8
+; GFX1310-NEXT:    v_bfe_u32 v11, v9, 18, 1
+; GFX1310-NEXT:    s_mov_b32 s0, 0x100000
+; GFX1310-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_dual_mov_b32 v8, v2 :: v_dual_bitop2_b32 v9, s0, v9 bitop3:0x54
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v10, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_lshlrev_b64_e64 v[6:7], v14, 1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_or_b32_e32 v10, v10, v11
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_add_co_u32 v11, vcc_lo, v6, -1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v16, null, -1, v7, vcc_lo
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_and_b32_e32 v10, v15, v10
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_and_b32_e32 v2, v2, v11
 ; GFX1310-NEXT:    v_lshrrev_b64 v[6:7], v4, v[8:9]
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_and_b32_e32 v3, v9, v16
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_add_co_u32 v10, s0, v17, v10
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v11, null, 0, 0, s0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cmp_ne_u64_e32 vcc_lo, 0, v[2:3]
 ; GFX1310-NEXT:    v_and_b32_e32 v15, 1, v6
 ; GFX1310-NEXT:    v_lshrrev_b64 v[2:3], v14, v[8:9]
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v16, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_lt_i64_e32 vcc_lo, 3, v[10:11]
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_or_b32_e32 v8, v16, v15
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v9, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v3, v11, 0, vcc_lo
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_dual_lshrrev_b32 v11, 24, v1 :: v_dual_bitop2_b32 v2, v2, v8 bitop3:0x40
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_add_co_u32 v8, s0, v12, v9
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v9, null, 0, v13, s0
 ; GFX1310-NEXT:    v_cmp_ne_u64_e64 s0, 0, v[4:5]
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v2, 0, v2, s0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_add_co_u32 v4, s0, v8, 14
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v5, null, 0, v9, s0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_add_co_u32 v6, s0, v6, v2
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v7, null, 0, v7, s0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_lshlrev_b64_e32 v[8:9], 2, v[4:5]
 ; GFX1310-NEXT:    v_and_b32_e32 v9, 0x80, v11
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v2, v10, 0, vcc_lo
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cmp_lt_i64_e64 s0, 3, v[6:7]
 ; GFX1310-NEXT:    v_cmp_gt_i64_e64 s2, 1, v[4:5]
 ; GFX1310-NEXT:    v_cmp_lt_i64_e64 s1, 30, v[4:5]
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cmp_lt_i64_e32 vcc_lo, 3, v[2:3]
 ; GFX1310-NEXT:    v_or_b32_e32 v3, 0x7c, v9
 ; GFX1310-NEXT:    v_or_b32_e32 v7, v8, v9
@@ -1773,53 +1785,53 @@ define i8 @to_fp8_bf16(bfloat %x) {
 ; GFX1250-FAKE16-NEXT:    s_movk_i32 s0, 0x80
 ; GFX1250-FAKE16-NEXT:    v_sub_nc_u16 v5, v3, 1 clamp
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2)
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v6, 0x7f, v4
+; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v8, 7, v4
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v9, v4, s0, 0x7f bitop3:0xec
 ; GFX1250-FAKE16-NEXT:    v_cmp_ne_u16_e64 s0, 0, v3
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v8, 7, v4
+; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v6, 0x7f, v4
 ; GFX1250-FAKE16-NEXT:    v_lshlrev_b16 v7, v5, 1
-; GFX1250-FAKE16-NEXT:    v_lshrrev_b16 v10, 4, v6
+; GFX1250-FAKE16-NEXT:    v_cmp_ne_u16_e32 vcc_lo, 0, v8
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b16 v4, 3, v4
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b16 v12, v3, v9
-; GFX1250-FAKE16-NEXT:    v_cmp_ne_u16_e32 vcc_lo, 0, v8
+; GFX1250-FAKE16-NEXT:    v_lshrrev_b16 v10, 4, v6
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v7, v7, -1
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v11, 1, v10
-; GFX1250-FAKE16-NEXT:    v_lshrrev_b16 v5, v5, v9
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v8, 0, 1, vcc_lo
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT:    v_lshrrev_b16 v5, v5, v9
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v11, 1, v10
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v6, v6, v7, 0x80 bitop3:0xc8
 ; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v7, 1, v12
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v4, v4, v8, v11 bitop3:0xe0
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_cmp_ne_u16_e32 vcc_lo, 0, v6
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v4, v10, v4
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v6, 0, 1, vcc_lo
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 7, v4
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v5, v5, v6, v7 bitop3:0xe0
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v4, v4, 0, vcc_lo
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v3, 0, v5, s0
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v5, 0, 1, vcc_lo
-; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v3, v12, v3
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v3, v12, v3
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v2, v2, v5
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 7, v3
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
 ; GFX1250-FAKE16-NEXT:    v_add_nc_u16 v2, v2, 6
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v3, v3, 0, vcc_lo
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e64 v5, 0, 8, vcc_lo
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
 ; GFX1250-FAKE16-NEXT:    v_lshlrev_b16 v6, 3, v2
 ; GFX1250-FAKE16-NEXT:    v_cmp_gt_i16_e32 vcc_lo, 1, v2
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v3, v0, v3, v5 bitop3:0xfe
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-FAKE16-NEXT:    v_bitop3_b16 v4, v0, v4, v6 bitop3:0xfe
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e32 v2, v4, v3, vcc_lo
 ; GFX1250-FAKE16-NEXT:    v_cmp_eq_f32_e32 vcc_lo, 0, v1
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e32 v0, v2, v0, vcc_lo
 ; GFX1250-FAKE16-NEXT:    v_cmp_o_f32_e32 vcc_lo, v1, v1
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_cndmask_b32_e32 v0, 0x7f, v0, vcc_lo
 ; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
 ;
@@ -1844,53 +1856,53 @@ define i8 @to_fp8_bf16(bfloat %x) {
 ; GFX1310-NEXT:    s_movk_i32 s0, 0x80
 ; GFX1310-NEXT:    v_sub_nc_u16 v5, v3, 1 clamp
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2)
-; GFX1310-NEXT:    v_and_b32_e32 v6, 0x7f, v4
+; GFX1310-NEXT:    v_and_b32_e32 v8, 7, v4
 ; GFX1310-NEXT:    v_bitop3_b16 v9, v4, s0, 0x7f bitop3:0xec
 ; GFX1310-NEXT:    v_cmp_ne_u16_e64 s0, 0, v3
-; GFX1310-NEXT:    v_and_b32_e32 v8, 7, v4
+; GFX1310-NEXT:    v_and_b32_e32 v6, 0x7f, v4
 ; GFX1310-NEXT:    v_lshlrev_b16 v7, v5, 1
-; GFX1310-NEXT:    v_lshrrev_b16 v10, 4, v6
+; GFX1310-NEXT:    v_cmp_ne_u16_e32 vcc_lo, 0, v8
 ; GFX1310-NEXT:    v_lshrrev_b16 v4, 3, v4
 ; GFX1310-NEXT:    v_lshrrev_b16 v12, v3, v9
-; GFX1310-NEXT:    v_cmp_ne_u16_e32 vcc_lo, 0, v8
+; GFX1310-NEXT:    v_lshrrev_b16 v10, 4, v6
 ; GFX1310-NEXT:    v_add_nc_u16 v7, v7, -1
-; GFX1310-NEXT:    v_and_b32_e32 v11, 1, v10
-; GFX1310-NEXT:    v_lshrrev_b16 v5, v5, v9
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v8, 0, 1, vcc_lo
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_lshrrev_b16 v5, v5, v9
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    v_and_b32_e32 v11, 1, v10
 ; GFX1310-NEXT:    v_bitop3_b16 v6, v6, v7, 0x80 bitop3:0xc8
 ; GFX1310-NEXT:    v_and_b32_e32 v7, 1, v12
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_bitop3_b16 v4, v4, v8, v11 bitop3:0xe0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cmp_ne_u16_e32 vcc_lo, 0, v6
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_add_nc_u16 v4, v10, v4
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v6, 0, 1, vcc_lo
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 7, v4
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bitop3_b16 v5, v5, v6, v7 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v4, v4, 0, vcc_lo
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v3, 0, v5, s0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v5, 0, 1, vcc_lo
-; GFX1310-NEXT:    v_add_nc_u16 v3, v12, v3
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_add_nc_u16 v3, v12, v3
 ; GFX1310-NEXT:    v_add_nc_u16 v2, v2, v5
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cmp_lt_i16_e32 vcc_lo, 7, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_add_nc_u16 v2, v2, 6
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v3, v3, 0, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v5, 0, 8, vcc_lo
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_lshlrev_b16 v6, 3, v2
 ; GFX1310-NEXT:    v_cmp_gt_i16_e32 vcc_lo, 1, v2
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_bitop3_b16 v3, v0, v3, v5 bitop3:0xfe
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_bitop3_b16 v4, v0, v4, v6 bitop3:0xfe
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v2, v4, v3, vcc_lo
 ; GFX1310-NEXT:    v_cmp_eq_f32_e32 vcc_lo, 0, v1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v0, v2, v0, vcc_lo
 ; GFX1310-NEXT:    v_cmp_o_f32_e32 vcc_lo, v1, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v0, 0x7f, v0, vcc_lo
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
   %r = call i8 @llvm.convert.to.arbitrary.fp.i8.bf16(bfloat %x, metadata !"Float8E4M3FN", metadata !"round.tonearest", i1 false)
@@ -2040,49 +2052,51 @@ define i8 @to_e5m3fnu_f32(float %x) {
 ; GFX1310-NEXT:    s_wait_kmcnt 0x0
 ; GFX1310-NEXT:    v_frexp_exp_i32_f32_e32 v1, v0
 ; GFX1310-NEXT:    v_frexp_mant_f32_e32 v2, v0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
-; GFX1310-NEXT:    v_dual_sub_nc_u32 v3, 7, v1 :: v_dual_lshrrev_b32 v10, 19, v2
-; GFX1310-NEXT:    v_and_b32_e32 v4, 0x7fffff, v2
-; GFX1310-NEXT:    v_and_b32_e32 v9, 0x7ffff, v2
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7fffff
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_sub_nc_u32 v3, 7, v1 :: v_dual_bitop2_b32 v4, s0, v2 bitop3:0x40
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7ffff
+; GFX1310-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_dual_lshrrev_b32 v10, 19, v2 :: v_dual_bitop2_b32 v9, s0, v2 bitop3:0x40
 ; GFX1310-NEXT:    v_bfe_u32 v2, v2, 20, 3
 ; GFX1310-NEXT:    v_min_u32_e32 v3, 31, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_or_b32_e32 v5, 0x800000, v4
 ; GFX1310-NEXT:    v_bfe_u32 v4, v4, 20, 1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_sub_nc_u32_e64 v6, v3, 1 clamp
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bfe_u32 v8, v5, v3, 1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bfe_u32 v7, v5, 0, v6
-; GFX1310-NEXT:    v_dual_lshrrev_b32 v6, v6, v5 :: v_dual_lshrrev_b32 v5, v3, v5
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v6, v6, v5
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v7
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v9
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_bitop3_b32 v6, v6, v7, v8 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v5, v3, v5
 ; GFX1310-NEXT:    v_bitop3_b32 v4, v10, v7, v4 bitop3:0xe0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_dual_cndmask_b32 v3, 0, v6 :: v_dual_add_nc_u32 v2, v2, v4
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_add_nc_u32_e32 v3, v5, v3
-; GFX1310-NEXT:    v_cmp_lt_i32_e64 s0, 7, v2
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_cmp_lt_i32_e64 s0, 7, v2
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v2, v2, 0, s0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v3, v3, 0, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v4, 0, 8, vcc_lo
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v1, null, 14, v1, s0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_or_b32_e32 v3, v4, v3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_lshl_or_b32 v2, v1, 3, v2
 ; GFX1310-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 1, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, v2, v3, vcc_lo
 ; GFX1310-NEXT:    v_cmp_neq_f32_e32 vcc_lo, 0, v0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, 0, v1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_o_f32_e32 vcc_lo, v0, v0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v0, 0xff, v1, vcc_lo
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
   %r = call i8 @llvm.convert.to.arbitrary.fp.i8.f32(float %x, metadata !"Float8E5M3FNU", metadata !"round.tonearest", i1 false)
@@ -2326,95 +2340,96 @@ define <2 x i8> @to_e5m3fnu_v2f32(<2 x float> %x) {
 ; GFX1310-NEXT:    s_wait_bvhcnt 0x0
 ; GFX1310-NEXT:    s_wait_kmcnt 0x0
 ; GFX1310-NEXT:    v_frexp_exp_i32_f32_e32 v2, v1
-; GFX1310-NEXT:    v_frexp_mant_f32_e32 v3, v1
 ; GFX1310-NEXT:    v_frexp_exp_i32_f32_e32 v4, v0
+; GFX1310-NEXT:    v_frexp_mant_f32_e32 v3, v1
 ; GFX1310-NEXT:    v_frexp_mant_f32_e32 v5, v0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
-; GFX1310-NEXT:    v_dual_sub_nc_u32 v6, 7, v2 :: v_dual_lshrrev_b32 v8, 19, v3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_dual_sub_nc_u32 v6, 7, v2 :: v_dual_sub_nc_u32 v10, 7, v4
 ; GFX1310-NEXT:    v_and_b32_e32 v7, 0x7fffff, v3
-; GFX1310-NEXT:    v_dual_sub_nc_u32 v10, 7, v4 :: v_dual_lshrrev_b32 v12, 19, v5
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7fffff
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_lshrrev_b32 v8, 19, v3 :: v_dual_bitop2_b32 v11, s0, v5 bitop3:0x40
+; GFX1310-NEXT:    v_and_b32_e32 v9, 0x7ffff, v3
 ; GFX1310-NEXT:    v_min_u32_e32 v6, 31, v6
-; GFX1310-NEXT:    v_and_b32_e32 v11, 0x7fffff, v5
-; GFX1310-NEXT:    v_or_b32_e32 v14, 0x800000, v7
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4)
 ; GFX1310-NEXT:    v_min_u32_e32 v10, 31, v10
-; GFX1310-NEXT:    v_and_b32_e32 v9, 0x7ffff, v3
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7ffff
+; GFX1310-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_lshrrev_b32 v12, 19, v5 :: v_dual_bitop2_b32 v13, s0, v5 bitop3:0x40
+; GFX1310-NEXT:    v_bfe_u32 v3, v3, 20, 3
 ; GFX1310-NEXT:    v_sub_nc_u32_e64 v15, v6, 1 clamp
 ; GFX1310-NEXT:    v_or_b32_e32 v16, 0x800000, v11
-; GFX1310-NEXT:    v_bfe_u32 v18, v14, v6, 1
 ; GFX1310-NEXT:    v_sub_nc_u32_e64 v17, v10, 1 clamp
-; GFX1310-NEXT:    v_and_b32_e32 v13, 0x7ffff, v5
-; GFX1310-NEXT:    v_bfe_u32 v19, v14, 0, v15
-; GFX1310-NEXT:    v_lshrrev_b32_e32 v15, v15, v14
+; GFX1310-NEXT:    v_or_b32_e32 v14, 0x800000, v7
 ; GFX1310-NEXT:    v_bfe_u32 v7, v7, 20, 1
+; GFX1310-NEXT:    v_bfe_u32 v11, v11, 20, 1
+; GFX1310-NEXT:    v_bfe_u32 v5, v5, 20, 3
 ; GFX1310-NEXT:    v_bfe_u32 v21, v16, 0, v17
 ; GFX1310-NEXT:    v_lshrrev_b32_e32 v17, v17, v16
-; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v19
+; GFX1310-NEXT:    v_bfe_u32 v19, v14, 0, v15
+; GFX1310-NEXT:    v_bfe_u32 v18, v14, v6, 1
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v15, v15, v14
 ; GFX1310-NEXT:    v_bfe_u32 v20, v16, v10, 1
-; GFX1310-NEXT:    v_bfe_u32 v3, v3, 20, 3
-; GFX1310-NEXT:    v_bfe_u32 v11, v11, 20, 1
-; GFX1310-NEXT:    v_bfe_u32 v5, v5, 20, 3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v19
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v19, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v21
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bitop3_b32 v15, v15, v19, v18 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v21, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v9
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_bitop3_b32 v17, v17, v21, v20 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v9, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v6
 ; GFX1310-NEXT:    v_lshrrev_b32_e32 v14, v6, v14
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bitop3_b32 v7, v8, v9, v7 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v6, 0, v15, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v13
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_dual_add_nc_u32 v3, v3, v7 :: v_dual_add_nc_u32 v6, v14, v6
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v13, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v10
 ; GFX1310-NEXT:    v_lshrrev_b32_e32 v10, v10, v16
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cmp_lt_i32_e64 s0, 7, v3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_bitop3_b32 v9, v12, v13, v11 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v8, 0, v17, vcc_lo
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v6
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v3, v3, 0, s0
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v2, null, 14, v2, s0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_dual_add_nc_u32 v5, v5, v9 :: v_dual_add_nc_u32 v7, v10, v8
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v6, v6, 0, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v8, 0, 8, vcc_lo
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_lshl_or_b32 v3, v2, 3, v3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cmp_lt_i32_e64 s0, 7, v5
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v7
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v5, v5, 0, s0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v7, v7, 0, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v9, 0, 8, vcc_lo
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v4, null, 14, v4, s0
 ; GFX1310-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 1, v2
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_or_b32_e32 v7, v9, v7
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_lshl_or_b32 v5, v4, 3, v5
 ; GFX1310-NEXT:    v_or_b32_e32 v6, v8, v6
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v2, v3, v6, vcc_lo
 ; GFX1310-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 1, v4
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v3, v5, v7, vcc_lo
 ; GFX1310-NEXT:    v_cmp_neq_f32_e32 vcc_lo, 0, v1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v2, 0, v2, vcc_lo
 ; GFX1310-NEXT:    v_cmp_neq_f32_e32 vcc_lo, 0, v0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v3, 0, v3, vcc_lo
 ; GFX1310-NEXT:    v_cmp_o_f32_e32 vcc_lo, v1, v1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, 0xff, v2, vcc_lo
 ; GFX1310-NEXT:    v_cmp_o_f32_e32 vcc_lo, v0, v0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v0, 0xff, v3, vcc_lo
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_lshlrev_b16 v2, 8, v1
 ; GFX1310-NEXT:    v_and_b32_e32 v1, 0xff, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bitop3_b16 v0, v0, v2, 0xff bitop3:0xec
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
   %r = call <2 x i8> @llvm.convert.to.arbitrary.fp.v2i8.v2f32(<2 x float> %x, metadata !"Float8E5M3FNU", metadata !"round.tonearest", i1 false)
@@ -2870,74 +2885,76 @@ define <4 x i8> @to_e5m3fnu_v4f32(<4 x float> %x) {
 ; GFX1310-NEXT:    v_frexp_exp_i32_f32_e32 v6, v0
 ; GFX1310-NEXT:    v_frexp_mant_f32_e32 v8, v0
 ; GFX1310-NEXT:    v_frexp_exp_i32_f32_e32 v9, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_4)
-; GFX1310-NEXT:    v_dual_sub_nc_u32 v7, 7, v4 :: v_dual_lshrrev_b32 v12, 19, v5
-; GFX1310-NEXT:    v_and_b32_e32 v11, 0x7fffff, v5
-; GFX1310-NEXT:    v_and_b32_e32 v13, 0x7ffff, v5
-; GFX1310-NEXT:    v_dual_sub_nc_u32 v16, 7, v6 :: v_dual_sub_nc_u32 v18, 7, v9
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7fffff
+; GFX1310-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_sub_nc_u32 v7, 7, v4 :: v_dual_bitop2_b32 v11, s0, v5 bitop3:0x40
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7ffff
+; GFX1310-NEXT:    v_dual_lshrrev_b32 v12, 19, v5 :: v_dual_bitop2_b32 v13, s0, v5 bitop3:0x40
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7fffff
+; GFX1310-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_sub_nc_u32 v18, 7, v9 :: v_dual_bitop2_b32 v17, s0, v8 bitop3:0x40
 ; GFX1310-NEXT:    v_min_u32_e32 v7, 31, v7
-; GFX1310-NEXT:    v_or_b32_e32 v14, 0x800000, v11
+; GFX1310-NEXT:    s_mov_b32 s0, 0x800000
+; GFX1310-NEXT:    v_dual_sub_nc_u32 v16, 7, v6 :: v_dual_bitop2_b32 v14, s0, v11 bitop3:0x54
+; GFX1310-NEXT:    v_frexp_mant_f32_e32 v10, v3
 ; GFX1310-NEXT:    v_bfe_u32 v11, v11, 20, 1
-; GFX1310-NEXT:    v_bfe_u32 v5, v5, 20, 3
-; GFX1310-NEXT:    v_and_b32_e32 v17, 0x7fffff, v8
 ; GFX1310-NEXT:    v_sub_nc_u32_e64 v15, v7, 1 clamp
-; GFX1310-NEXT:    v_min_u32_e32 v16, 31, v16
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4)
 ; GFX1310-NEXT:    v_bfe_u32 v21, v14, v7, 1
-; GFX1310-NEXT:    v_frexp_mant_f32_e32 v10, v3
+; GFX1310-NEXT:    v_min_u32_e32 v16, 31, v16
+; GFX1310-NEXT:    v_bfe_u32 v5, v5, 20, 3
 ; GFX1310-NEXT:    v_or_b32_e32 v22, 0x800000, v17
 ; GFX1310-NEXT:    v_bfe_u32 v20, v14, 0, v15
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v15, v15, v14
 ; GFX1310-NEXT:    v_min_u32_e32 v18, 31, v18
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX1310-NEXT:    v_and_b32_e32 v19, 0x7fffff, v10
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v20
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v20, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v13
+; GFX1310-NEXT:    v_bitop3_b32 v15, v15, v20, v21 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v13, 0, 1, vcc_lo
+; GFX1310-NEXT:    v_sub_nc_u32_e64 v20, v16, 1 clamp
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v7
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_and_b32_e32 v19, 0x7fffff, v10
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_bitop3_b32 v11, v12, v13, v11 bitop3:0xe0
-; GFX1310-NEXT:    v_lshrrev_b32_e32 v15, v15, v14
-; GFX1310-NEXT:    v_bfe_u32 v13, v22, v16, 1
-; GFX1310-NEXT:    v_add_nc_u32_e32 v5, v5, v11
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_2)
-; GFX1310-NEXT:    v_bitop3_b32 v15, v15, v20, v21 bitop3:0xe0
-; GFX1310-NEXT:    v_sub_nc_u32_e64 v20, v16, 1 clamp
 ; GFX1310-NEXT:    v_dual_cndmask_b32 v7, 0, v15 :: v_dual_lshrrev_b32 v14, v7, v14
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_bfe_u32 v12, v22, 0, v20
-; GFX1310-NEXT:    v_lshrrev_b32_e32 v11, v20, v22
-; GFX1310-NEXT:    v_sub_nc_u32_e64 v15, v18, 1 clamp
+; GFX1310-NEXT:    v_bfe_u32 v13, v22, v16, 1
+; GFX1310-NEXT:    v_dual_add_nc_u32 v5, v5, v11 :: v_dual_lshrrev_b32 v11, v20, v22
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_add_nc_u32_e32 v7, v14, v7
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v12
+; GFX1310-NEXT:    v_sub_nc_u32_e64 v15, v18, 1 clamp
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_cmp_lt_i32_e64 s0, 7, v7
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v12, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v5
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v7, v7, 0, s0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3)
 ; GFX1310-NEXT:    v_bitop3_b32 v11, v11, v12, v13 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v5, v5, 0, vcc_lo
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v4, null, 14, v4, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v12, 0, 8, s0
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v16
 ; GFX1310-NEXT:    v_lshrrev_b32_e32 v14, v16, v22
-; GFX1310-NEXT:    v_or_b32_e32 v13, 0x800000, v19
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_lshl_or_b32 v5, v4, 3, v5
+; GFX1310-NEXT:    v_cmp_neq_f32_e64 s0, 0, v1
 ; GFX1310-NEXT:    v_or_b32_e32 v7, v12, v7
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v11, 0, v11, vcc_lo
 ; GFX1310-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 1, v4
-; GFX1310-NEXT:    v_cmp_neq_f32_e64 s0, 0, v1
-; GFX1310-NEXT:    v_bfe_u32 v12, v13, v18, 1
-; GFX1310-NEXT:    v_bfe_u32 v19, v19, 20, 1
-; GFX1310-NEXT:    v_dual_add_nc_u32 v11, v14, v11 :: v_dual_lshrrev_b32 v14, v15, v13
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v4, v5, v7, vcc_lo
+; GFX1310-NEXT:    v_or_b32_e32 v13, 0x800000, v19
+; GFX1310-NEXT:    v_bfe_u32 v19, v19, 20, 1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_dual_cndmask_b32 v4, 0, v4, s0 :: v_dual_add_nc_u32 v11, v14, v11
 ; GFX1310-NEXT:    v_bfe_u32 v5, v13, 0, v15
-; GFX1310-NEXT:    v_lshrrev_b32_e32 v13, v18, v13
+; GFX1310-NEXT:    v_bfe_u32 v12, v13, v18, 1
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v14, v15, v13
 ; GFX1310-NEXT:    v_frexp_mant_f32_e32 v15, v2
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4)
-; GFX1310-NEXT:    v_cndmask_b32_e64 v4, 0, v4, s0
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v11
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v13, v18, v13
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v7, v11, 0, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v11, 0, 8, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v5
@@ -2945,107 +2962,106 @@ define <4 x i8> @to_e5m3fnu_v4f32(<4 x float> %x) {
 ; GFX1310-NEXT:    v_cmp_o_f32_e32 vcc_lo, v1, v1
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bitop3_b32 v5, v14, v5, v12 bitop3:0xe0
-; GFX1310-NEXT:    v_and_b32_e32 v12, 0x7ffff, v8
-; GFX1310-NEXT:    v_frexp_exp_i32_f32_e32 v14, v2
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, 0xff, v4, vcc_lo
+; GFX1310-NEXT:    v_frexp_exp_i32_f32_e32 v14, v2
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v18
 ; GFX1310-NEXT:    v_or_b32_e32 v4, v11, v7
+; GFX1310-NEXT:    v_and_b32_e32 v12, 0x7ffff, v8
 ; GFX1310-NEXT:    v_bfe_u32 v11, v17, 20, 1
-; GFX1310-NEXT:    v_and_b32_e32 v17, 0x7ffff, v10
-; GFX1310-NEXT:    v_dual_lshrrev_b32 v7, 19, v8 :: v_dual_cndmask_b32 v5, 0, v5, vcc_lo
-; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v12
+; GFX1310-NEXT:    v_dual_cndmask_b32 v5, 0, v5 :: v_dual_lshrrev_b32 v18, 19, v10
 ; GFX1310-NEXT:    v_sub_nc_u32_e32 v16, 7, v14
-; GFX1310-NEXT:    v_lshrrev_b32_e32 v18, 19, v10
-; GFX1310-NEXT:    v_bfe_u32 v10, v10, 20, 3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v12
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v7, 19, v8
 ; GFX1310-NEXT:    v_bfe_u32 v8, v8, 20, 3
-; GFX1310-NEXT:    v_cndmask_b32_e64 v12, 0, 1, vcc_lo
-; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v17
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7fffff
+; GFX1310-NEXT:    v_dual_add_nc_u32 v5, v13, v5 :: v_dual_bitop2_b32 v13, s0, v15 bitop3:0x40
 ; GFX1310-NEXT:    v_min_u32_e32 v16, 31, v16
+; GFX1310-NEXT:    v_cndmask_b32_e64 v12, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_lshlrev_b16 v1, 8, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    v_or_b32_e32 v20, 0x800000, v13
+; GFX1310-NEXT:    v_sub_nc_u32_e64 v21, v16, 1 clamp
+; GFX1310-NEXT:    v_and_b32_e32 v17, 0x7ffff, v10
 ; GFX1310-NEXT:    v_bitop3_b32 v7, v7, v12, v11 bitop3:0xe0
+; GFX1310-NEXT:    v_bfe_u32 v10, v10, 20, 3
+; GFX1310-NEXT:    v_bfe_u32 v13, v13, 20, 1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v17
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v17, 0, 1, vcc_lo
-; GFX1310-NEXT:    v_sub_nc_u32_e64 v21, v16, 1 clamp
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
-; GFX1310-NEXT:    v_bitop3_b32 v12, v18, v17, v19 bitop3:0xe0
-; GFX1310-NEXT:    v_add_nc_u32_e32 v5, v13, v5
-; GFX1310-NEXT:    v_and_b32_e32 v13, 0x7fffff, v15
-; GFX1310-NEXT:    v_add_nc_u32_e32 v10, v10, v12
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v5
-; GFX1310-NEXT:    v_or_b32_e32 v20, 0x800000, v13
-; GFX1310-NEXT:    v_lshrrev_b32_e32 v12, 19, v15
-; GFX1310-NEXT:    v_bfe_u32 v13, v13, 20, 1
+; GFX1310-NEXT:    v_bitop3_b32 v12, v18, v17, v19 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v5, v5, 0, vcc_lo
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4)
+; GFX1310-NEXT:    v_cndmask_b32_e64 v19, 0, 8, vcc_lo
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_add_nc_u32 v7, v8, v7 :: v_dual_add_nc_u32 v10, v10, v12
 ; GFX1310-NEXT:    v_bfe_u32 v11, v20, 0, v21
-; GFX1310-NEXT:    v_lshrrev_b32_e32 v18, v21, v20
-; GFX1310-NEXT:    v_and_b32_e32 v21, 0x7ffff, v15
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7ffff
+; GFX1310-NEXT:    v_dual_lshrrev_b32 v18, v21, v20 :: v_dual_bitop2_b32 v21, s0, v15 bitop3:0x40
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v12, 19, v15
 ; GFX1310-NEXT:    v_bfe_u32 v17, v20, v16, 1
-; GFX1310-NEXT:    v_cndmask_b32_e64 v19, 0, 8, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e64 s0, 0, v11
-; GFX1310-NEXT:    v_add_nc_u32_e32 v7, v8, v7
-; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v21
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_or_b32_e32 v5, v19, v5
+; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v21
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v11, 0, 1, s0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_bitop3_b32 v11, v18, v11, v17 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v17, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v16
 ; GFX1310-NEXT:    v_lshrrev_b32_e32 v16, v16, v20
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bitop3_b32 v12, v12, v17, v13 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v11, 0, v11, vcc_lo
 ; GFX1310-NEXT:    v_bfe_u32 v13, v15, 20, 3
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v10
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_dual_add_nc_u32 v11, v16, v11 :: v_dual_add_nc_u32 v12, v13, v12
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v10, v10, 0, vcc_lo
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v9, null, 14, v9, vcc_lo
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v11
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_cmp_lt_i32_e64 s0, 7, v12
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_lshl_or_b32 v8, v9, 3, v10
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v10, v11, 0, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v11, 0, 8, vcc_lo
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v12, v12, 0, s0
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v13, null, 14, v14, s0
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v7
 ; GFX1310-NEXT:    v_cmp_gt_i32_e64 s0, 1, v9
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_lshl_or_b32 v9, v13, 3, v12
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v7, v7, 0, vcc_lo
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_dual_cndmask_b32 v5, v8, v5, s0 :: v_dual_bitop2_b32 v8, v11, v10 bitop3:0x54
 ; GFX1310-NEXT:    v_add_co_ci_u32_e64 v6, null, 14, v6, vcc_lo
 ; GFX1310-NEXT:    v_cmp_neq_f32_e32 vcc_lo, 0, v3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_lshl_or_b32 v7, v6, 3, v7
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_3) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v5, 0, v5, vcc_lo
 ; GFX1310-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 1, v13
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v8, v9, v8, vcc_lo
 ; GFX1310-NEXT:    v_cmp_o_f32_e32 vcc_lo, v3, v3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v3, 0xff, v5, vcc_lo
 ; GFX1310-NEXT:    v_cmp_neq_f32_e32 vcc_lo, 0, v2
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v5, 0, v8, vcc_lo
 ; GFX1310-NEXT:    v_cmp_gt_i32_e32 vcc_lo, 1, v6
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_4) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_lshlrev_b16 v3, 8, v3
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v4, v7, v4, vcc_lo
 ; GFX1310-NEXT:    v_cmp_o_f32_e32 vcc_lo, v2, v2
-; GFX1310-NEXT:    v_lshlrev_b16 v3, 8, v3
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v2, 0xff, v5, vcc_lo
 ; GFX1310-NEXT:    v_cmp_neq_f32_e32 vcc_lo, 0, v0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bitop3_b16 v2, v2, v3, 0xff bitop3:0xec
-; GFX1310-NEXT:    v_cndmask_b32_e32 v4, 0, v4, vcc_lo
-; GFX1310-NEXT:    v_and_b32_e32 v3, 0xffff, v1
+; GFX1310-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1310-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_dual_cndmask_b32 v4, 0, v4, vcc_lo :: v_dual_bitop2_b32 v3, s0, v1 bitop3:0x40
 ; GFX1310-NEXT:    v_cmp_o_f32_e32 vcc_lo, v0, v0
 ; GFX1310-NEXT:    v_lshl_or_b32 v3, v2, 16, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_2)
-; GFX1310-NEXT:    v_cndmask_b32_e32 v0, 0xff, v4, vcc_lo
-; GFX1310-NEXT:    v_lshlrev_b32_e32 v4, 16, v2
+; GFX1310-NEXT:    s_mov_b32 s0, 0xff
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_cndmask_b32 v0, s0, v4, vcc_lo :: v_dual_lshlrev_b32 v4, 16, v2
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bitop3_b16 v0, v0, v1, 0xff bitop3:0xec
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1310-NEXT:    v_dual_lshrrev_b32 v1, 8, v3 :: v_dual_lshrrev_b32 v3, 24, v4
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
   %r = call <4 x i8> @llvm.convert.to.arbitrary.fp.v4i8.v4f32(<4 x float> %x, metadata !"Float8E5M3FNU", metadata !"round.tonearest", i1 false)
@@ -3222,52 +3238,54 @@ define i8 @to_e5m3fnu_f32_sat(float %x) {
 ; GFX1310-NEXT:    s_wait_kmcnt 0x0
 ; GFX1310-NEXT:    v_frexp_exp_i32_f32_e32 v1, v0
 ; GFX1310-NEXT:    v_frexp_mant_f32_e32 v2, v0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_4)
-; GFX1310-NEXT:    v_dual_sub_nc_u32 v3, 7, v1 :: v_dual_lshrrev_b32 v10, 19, v2
-; GFX1310-NEXT:    v_and_b32_e32 v4, 0x7fffff, v2
-; GFX1310-NEXT:    v_and_b32_e32 v9, 0x7ffff, v2
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7fffff
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_sub_nc_u32 v3, 7, v1 :: v_dual_bitop2_b32 v4, s0, v2 bitop3:0x40
+; GFX1310-NEXT:    s_mov_b32 s0, 0x7ffff
+; GFX1310-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_dual_lshrrev_b32 v10, 19, v2 :: v_dual_bitop2_b32 v9, s0, v2 bitop3:0x40
 ; GFX1310-NEXT:    v_bfe_u32 v2, v2, 20, 3
 ; GFX1310-NEXT:    v_min_u32_e32 v3, 31, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(SKIP_1) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_or_b32_e32 v5, 0x800000, v4
 ; GFX1310-NEXT:    v_bfe_u32 v4, v4, 20, 1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_sub_nc_u32_e64 v6, v3, 1 clamp
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bfe_u32 v8, v5, v3, 1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_bfe_u32 v7, v5, 0, v6
-; GFX1310-NEXT:    v_dual_lshrrev_b32 v6, v6, v5 :: v_dual_lshrrev_b32 v5, v3, v5
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v6, v6, v5
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v7
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v9
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_bitop3_b32 v6, v6, v7, v8 bitop3:0xe0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v7, 0, 1, vcc_lo
 ; GFX1310-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-NEXT:    v_lshrrev_b32_e32 v5, v3, v5
 ; GFX1310-NEXT:    v_bitop3_b32 v4, v10, v7, v4 bitop3:0xe0
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_dual_cndmask_b32 v3, 0, v6 :: v_dual_add_nc_u32 v2, v2, v4
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_add_nc_u32_e32 v3, v5, v3
-; GFX1310-NEXT:    v_cmp_lt_i32_e64 s0, 7, v2
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1310-NEXT:    v_cmp_lt_i32_e64 s0, 7, v2
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 7, v3
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_3) | instid1(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v2, v2, 0, s0
+; GFX1310-NEXT:    v_add_co_ci_u32_e64 v1, null, 14, v1, s0
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v3, v3, 0, vcc_lo
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v4, 0, 8, vcc_lo
-; GFX1310-NEXT:    v_add_co_ci_u32_e64 v1, null, 14, v1, s0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1310-NEXT:    v_cmp_lt_i32_e32 vcc_lo, 6, v2
-; GFX1310-NEXT:    v_or_b32_e32 v3, v4, v3
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_3) | instid1(VALU_DEP_3)
-; GFX1310-NEXT:    v_lshl_or_b32 v4, v1, 3, v2
-; GFX1310-NEXT:    v_cmp_gt_i32_e64 s2, 1, v1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4)
 ; GFX1310-NEXT:    v_cmp_eq_u32_e64 s0, 31, v1
+; GFX1310-NEXT:    v_cmp_gt_i32_e64 s2, 1, v1
 ; GFX1310-NEXT:    v_cmp_lt_i32_e64 s1, 31, v1
-; GFX1310-NEXT:    v_cndmask_b32_e64 v1, v4, v3, s2
 ; GFX1310-NEXT:    s_and_b32 s0, s0, vcc_lo
 ; GFX1310-NEXT:    v_cmp_neq_f32_e32 vcc_lo, 0, v0
+; GFX1310-NEXT:    v_or_b32_e32 v3, v4, v3
+; GFX1310-NEXT:    v_lshl_or_b32 v4, v1, 3, v2
 ; GFX1310-NEXT:    s_or_b32 s0, s1, s0
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1310-NEXT:    v_cndmask_b32_e64 v1, v4, v3, s2
 ; GFX1310-NEXT:    v_cndmask_b32_e64 v1, v1, 0xfe, s0
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_cndmask_b32_e32 v1, 0, v1, vcc_lo
diff --git a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-widen.ll b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-widen.ll
index fa961455f3900..aa37d09b6ea96 100644
--- a/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-widen.ll
+++ b/llvm/test/CodeGen/AMDGPU/float-to-arbitrary-fp-widen.ll
@@ -1016,20 +1016,20 @@ define <5 x i8> @to_fp8_v5f16(<5 x half> %x) {
 ; GFX1250-FAKE16-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v4, v2
-; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v0, v0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v3, s0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v2, v1
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v1, 0xffff, v4
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v5, 0xffff, v0
+; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v0, v0
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v6, 16, v2 :: v_dual_bitop2_b32 v1, s0, v4 bitop3:0x40
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v5, 0xffff, v0
 ; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v7, v3, 16, v1
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v1, v2, 16, v5
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v6, 16, v2 :: v_dual_lshrrev_b32 v1, 8, v1
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b64 v[6:7], 24, v[6:7]
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v3, v6
+; GFX1250-FAKE16-NEXT:    v_dual_lshrrev_b32 v1, 8, v1 :: v_dual_mov_b32 v3, v6
 ; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
 ;
 ; GFX1310-LABEL: to_fp8_v5f16:
@@ -1043,16 +1043,17 @@ define <5 x i8> @to_fp8_v5f16(<5 x half> %x) {
 ; GFX1310-NEXT:    v_cvt_pk_fp8_f16_e32 v0, v0
 ; GFX1310-NEXT:    v_cvt_pk_fp8_f16_e32 v4, s0
 ; GFX1310-NEXT:    v_cvt_pk_fp8_f16_e32 v2, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4)
 ; GFX1310-NEXT:    v_and_b32_e32 v1, 0xffff, v5
-; GFX1310-NEXT:    v_and_b32_e32 v6, 0xffff, v0
+; GFX1310-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_bitop2_b32 v6, s0, v0 bitop3:0x40
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_lshl_or_b32 v4, v4, 16, v1
 ; GFX1310-NEXT:    v_lshl_or_b32 v1, v2, 16, v6
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1310-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_lshrrev_b32 v1, 8, v1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_lshrrev_b64 v[3:4], 24, v[3:4]
-; GFX1310-NEXT:    v_mov_b32_e32 v4, v5
+; GFX1310-NEXT:    v_dual_mov_b32 v4, v5 :: v_dual_lshrrev_b32 v1, 8, v1
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
   %r = call <5 x i8> @llvm.convert.to.arbitrary.fp.v5i8.v5f16(<5 x half> %x, metadata !"Float8E4M3FN", metadata !"round.tonearest", i1 false)
   ret <5 x i8> %r
@@ -1798,17 +1799,18 @@ define <6 x i8> @to_fp8_v6f16(<6 x half> %x) {
 ; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v0, v0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v4, v2
-; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v3, s0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v2, v1
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v1, 0xffff, v0
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v5, 0xffff, v4
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v3, s0
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v6, 16, v2 :: v_dual_bitop2_b32 v1, s0, v0 bitop3:0x40
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v1, v2, 16, v1
-; GFX1250-FAKE16-NEXT:    v_lshlrev_b32_e32 v6, 16, v2
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshrrev_b32 v1, 8, v1 :: v_dual_bitop2_b32 v5, s0, v4 bitop3:0x40
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v7, v3, 16, v5
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1250-FAKE16-NEXT:    v_lshrrev_b32_e32 v1, 8, v1
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b64 v[8:9], 24, v[6:7]
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-FAKE16-NEXT:    v_dual_lshrrev_b32 v5, 8, v7 :: v_dual_mov_b32 v3, v8
@@ -1823,17 +1825,18 @@ define <6 x i8> @to_fp8_v6f16(<6 x half> %x) {
 ; GFX1310-NEXT:    s_wait_kmcnt 0x0
 ; GFX1310-NEXT:    v_cvt_pk_fp8_f16_e32 v0, v0
 ; GFX1310-NEXT:    v_cvt_pk_fp8_f16_e32 v6, v2
-; GFX1310-NEXT:    v_cvt_pk_fp8_f16_e32 v3, s0
 ; GFX1310-NEXT:    v_cvt_pk_fp8_f16_e32 v2, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1310-NEXT:    v_and_b32_e32 v1, 0xffff, v0
-; GFX1310-NEXT:    v_and_b32_e32 v5, 0xffff, v6
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1310-NEXT:    v_cvt_pk_fp8_f16_e32 v3, s0
+; GFX1310-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_lshlrev_b32 v4, 16, v2 :: v_dual_bitop2_b32 v1, s0, v0 bitop3:0x40
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1310-NEXT:    v_lshl_or_b32 v1, v2, 16, v1
-; GFX1310-NEXT:    v_lshlrev_b32_e32 v4, 16, v2
+; GFX1310-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_lshrrev_b32 v1, 8, v1 :: v_dual_bitop2_b32 v5, s0, v6 bitop3:0x40
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1310-NEXT:    v_lshl_or_b32 v5, v3, 16, v5
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1310-NEXT:    v_lshrrev_b32_e32 v1, 8, v1
 ; GFX1310-NEXT:    v_lshrrev_b64 v[3:4], 24, v[4:5]
 ; GFX1310-NEXT:    v_dual_mov_b32 v4, v6 :: v_dual_lshrrev_b32 v5, 8, v5
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
@@ -2805,16 +2808,17 @@ define <7 x i8> @to_fp8_v7f16(<7 x half> %x) {
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v6, v3
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v2, v1
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_fp8_f16_e64 v0, v0
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v1, 0xffff, v4
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v3, 0xffff, v0
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v8, 16, v2 :: v_dual_bitop2_b32 v1, s0, v4 bitop3:0x40
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v9, v6, 16, v1
-; GFX1250-FAKE16-NEXT:    v_lshlrev_b32_e32 v8, 16, v2
-; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v1, v2, 16, v3
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX1250-FAKE16-NEXT:    v_lshrrev_b32_e32 v5, 8, v9
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshrrev_b32 v5, 8, v9 :: v_dual_bitop2_b32 v3, s0, v0 bitop3:0x40
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b64 v[8:9], 24, v[8:9]
+; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v1, v2, 16, v3
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-FAKE16-NEXT:    v_dual_lshrrev_b32 v1, 8, v1 :: v_dual_mov_b32 v3, v8
 ; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
@@ -3694,20 +3698,20 @@ define <5 x i8> @to_bf8_v5f16(<5 x half> %x) {
 ; GFX1250-FAKE16-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_bf8_f16_e64 v4, v2
-; GFX1250-FAKE16-NEXT:    v_cvt_pk_bf8_f16_e64 v0, v0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_bf8_f16_e64 v3, s0
 ; GFX1250-FAKE16-NEXT:    v_cvt_pk_bf8_f16_e64 v2, v1
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v1, 0xffff, v4
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v5, 0xffff, v0
+; GFX1250-FAKE16-NEXT:    v_cvt_pk_bf8_f16_e64 v0, v0
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v6, 16, v2 :: v_dual_bitop2_b32 v1, s0, v4 bitop3:0x40
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v5, 0xffff, v0
 ; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v7, v3, 16, v1
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_lshl_or_b32 v1, v2, 16, v5
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v6, 16, v2 :: v_dual_lshrrev_b32 v1, 8, v1
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b64 v[6:7], 24, v[6:7]
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v3, v6
+; GFX1250-FAKE16-NEXT:    v_dual_lshrrev_b32 v1, 8, v1 :: v_dual_mov_b32 v3, v6
 ; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
 ;
 ; GFX1310-LABEL: to_bf8_v5f16:
@@ -3721,16 +3725,17 @@ define <5 x i8> @to_bf8_v5f16(<5 x half> %x) {
 ; GFX1310-NEXT:    v_cvt_pk_bf8_f16_e32 v0, v0
 ; GFX1310-NEXT:    v_cvt_pk_bf8_f16_e32 v4, s0
 ; GFX1310-NEXT:    v_cvt_pk_bf8_f16_e32 v2, v1
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_4)
 ; GFX1310-NEXT:    v_and_b32_e32 v1, 0xffff, v5
-; GFX1310-NEXT:    v_and_b32_e32 v6, 0xffff, v0
+; GFX1310-NEXT:    s_mov_b32 s0, 0xffff
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instid1(SALU_CYCLE_1)
+; GFX1310-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_bitop2_b32 v6, s0, v0 bitop3:0x40
 ; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_lshl_or_b32 v4, v4, 16, v1
 ; GFX1310-NEXT:    v_lshl_or_b32 v1, v2, 16, v6
-; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1310-NEXT:    v_dual_lshlrev_b32 v3, 16, v2 :: v_dual_lshrrev_b32 v1, 8, v1
+; GFX1310-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1310-NEXT:    v_lshrrev_b64 v[3:4], 24, v[3:4]
-; GFX1310-NEXT:    v_mov_b32_e32 v4, v5
+; GFX1310-NEXT:    v_dual_mov_b32 v4, v5 :: v_dual_lshrrev_b32 v1, 8, v1
 ; GFX1310-NEXT:    s_set_pc_i64 s[30:31]
   %r = call <5 x i8> @llvm.convert.to.arbitrary.fp.v5i8.v5f16(<5 x half> %x, metadata !"Float8E5M2", metadata !"round.tonearest", i1 false)
   ret <5 x i8> %r
diff --git a/llvm/test/CodeGen/AMDGPU/integer-mad-patterns.ll b/llvm/test/CodeGen/AMDGPU/integer-mad-patterns.ll
index d57229a23b14f..3b0f6308771fb 100644
--- a/llvm/test/CodeGen/AMDGPU/integer-mad-patterns.ll
+++ b/llvm/test/CodeGen/AMDGPU/integer-mad-patterns.ll
@@ -1187,16 +1187,16 @@ define <3 x i16> @clpeak_imad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
 ; GFX1250-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-SDAG-NEXT:    v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX1250-SDAG-NEXT:    v_pk_add_u16 v1, v1, 1
+; GFX1250-SDAG-NEXT:    v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
 ; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v4, v0, v2, v0
 ; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v5, v1, v3, v1
 ; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v1, v1, v3, 1
+; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
 ; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-SDAG-NEXT:    v_pk_mul_lo_u16 v6, v4, v2
 ; GFX1250-SDAG-NEXT:    v_pk_mul_lo_u16 v7, v5, v3
-; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v3, v5, v3, 1
+; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v3, v5, v3, 1 op_sel_hi:[1,1,0]
 ; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v2, v4, v2, 1 op_sel_hi:[1,1,0]
 ; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-SDAG-NEXT:    v_pk_mul_lo_u16 v0, v6, v0
@@ -2711,16 +2711,16 @@ define <3 x i16> @clpeak_umad_pat_v3i16(<3 x i16> %x, <3 x i16> %y) {
 ; GFX1250-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-SDAG-NEXT:    v_pk_add_u16 v0, v0, 1 op_sel_hi:[1,0]
-; GFX1250-SDAG-NEXT:    v_pk_add_u16 v1, v1, 1
+; GFX1250-SDAG-NEXT:    v_pk_add_u16 v1, v1, 1 op_sel_hi:[1,0]
 ; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v4, v0, v2, v0
 ; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v5, v1, v3, v1
 ; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v0, v0, v2, 1 op_sel_hi:[1,1,0]
-; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v1, v1, v3, 1
+; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v1, v1, v3, 1 op_sel_hi:[1,1,0]
 ; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-SDAG-NEXT:    v_pk_mul_lo_u16 v6, v4, v2
 ; GFX1250-SDAG-NEXT:    v_pk_mul_lo_u16 v7, v5, v3
-; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v3, v5, v3, 1
+; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v3, v5, v3, 1 op_sel_hi:[1,1,0]
 ; GFX1250-SDAG-NEXT:    v_pk_mad_u16 v2, v4, v2, 1 op_sel_hi:[1,1,0]
 ; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_4) | instskip(NEXT) | instid1(VALU_DEP_4)
 ; GFX1250-SDAG-NEXT:    v_pk_mul_lo_u16 v0, v6, v0
@@ -5512,15 +5512,16 @@ define i32 @clpeak_imad_pat_u24(i32 %x, i32 %y) {
 ; GFX1250-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-SDAG-NEXT:    v_and_b32_e32 v0, 0xffffff, v0
-; GFX1250-SDAG-NEXT:    v_and_b32_e32 v1, 0xffffff, v1
-; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-SDAG-NEXT:    v_add_nc_u32_e32 v0, 1, v0
-; GFX1250-SDAG-NEXT:    v_mul_lo_u32 v2, v1, v0
+; GFX1250-SDAG-NEXT:    s_mov_b32 s0, 0xffffff
+; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT:    v_dual_add_nc_u32 v0, 1, v0 :: v_dual_bitop2_b32 v1, s0, v1 bitop3:0x40
 ; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-SDAG-NEXT:    v_mul_lo_u32 v2, v1, v0
 ; GFX1250-SDAG-NEXT:    v_add_nc_u32_e32 v0, v2, v0
-; GFX1250-SDAG-NEXT:    v_mul_lo_u32 v0, v0, v1
 ; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-SDAG-NEXT:    v_mul_lo_u32 v0, v0, v1
 ; GFX1250-SDAG-NEXT:    v_mad_u32 v1, v0, v2, v0
+; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-SDAG-NEXT:    v_mad_u32 v0, v1, v0, v1
 ; GFX1250-SDAG-NEXT:    s_set_pc_i64 s[30:31]
 ;
@@ -5529,17 +5530,18 @@ define i32 @clpeak_imad_pat_u24(i32 %x, i32 %y) {
 ; GFX1250-GISEL-FAKE16-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-GISEL-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xffffff, v0
-; GFX1250-GISEL-FAKE16-NEXT:    v_and_b32_e32 v1, 0xffffff, v1
-; GFX1250-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-GISEL-FAKE16-NEXT:    v_add_nc_u32_e32 v0, 1, v0
-; GFX1250-GISEL-FAKE16-NEXT:    v_mul_lo_u32 v2, v1, v0
+; GFX1250-GISEL-FAKE16-NEXT:    s_mov_b32 s0, 0xffffff
+; GFX1250-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-GISEL-FAKE16-NEXT:    v_dual_add_nc_u32 v0, 1, v0 :: v_dual_bitop2_b32 v1, s0, v1 bitop3:0x40
 ; GFX1250-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-FAKE16-NEXT:    v_mul_lo_u32 v2, v1, v0
 ; GFX1250-GISEL-FAKE16-NEXT:    v_add_nc_u32_e32 v0, v2, v0
+; GFX1250-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1250-GISEL-FAKE16-NEXT:    v_mul_lo_u32 v0, v0, v1
 ; GFX1250-GISEL-FAKE16-NEXT:    v_add_nc_u32_e32 v1, 1, v2
-; GFX1250-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1250-GISEL-FAKE16-NEXT:    v_mul_lo_u32 v1, v0, v1
 ; GFX1250-GISEL-FAKE16-NEXT:    v_add_nc_u32_e32 v0, 1, v0
+; GFX1250-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-GISEL-FAKE16-NEXT:    v_mul_lo_u32 v0, v1, v0
 ; GFX1250-GISEL-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
 ;
@@ -5548,17 +5550,18 @@ define i32 @clpeak_imad_pat_u24(i32 %x, i32 %y) {
 ; GFX1250-GISEL-REAL16-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-GISEL-REAL16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-GISEL-REAL16-NEXT:    v_and_b32_e32 v0, 0xffffff, v0
-; GFX1250-GISEL-REAL16-NEXT:    v_and_b32_e32 v1, 0xffffff, v1
-; GFX1250-GISEL-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX1250-GISEL-REAL16-NEXT:    v_add_nc_u32_e32 v0, 1, v0
-; GFX1250-GISEL-REAL16-NEXT:    v_mul_lo_u32 v2, v1, v0
+; GFX1250-GISEL-REAL16-NEXT:    s_mov_b32 s0, 0xffffff
+; GFX1250-GISEL-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-GISEL-REAL16-NEXT:    v_dual_add_nc_u32 v0, 1, v0 :: v_dual_bitop2_b32 v1, s0, v1 bitop3:0x40
 ; GFX1250-GISEL-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-GISEL-REAL16-NEXT:    v_mul_lo_u32 v2, v1, v0
 ; GFX1250-GISEL-REAL16-NEXT:    v_add_nc_u32_e32 v0, v2, v0
+; GFX1250-GISEL-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1250-GISEL-REAL16-NEXT:    v_mul_lo_u32 v0, v0, v1
 ; GFX1250-GISEL-REAL16-NEXT:    v_add_nc_u32_e32 v1, 1, v2
-; GFX1250-GISEL-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1250-GISEL-REAL16-NEXT:    v_mul_lo_u32 v1, v0, v1
 ; GFX1250-GISEL-REAL16-NEXT:    v_add_nc_u32_e32 v0, 1, v0
+; GFX1250-GISEL-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-GISEL-REAL16-NEXT:    v_mul_lo_u32 v0, v1, v0
 ; GFX1250-GISEL-REAL16-NEXT:    s_set_pc_i64 s[30:31]
 ; GFX1250-GISEL-LABEL: clpeak_imad_pat_u24:
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.load.b128.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.load.b128.ll
index ec1c395f306a6..0413dfa292d11 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.load.b128.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.av.load.b128.ll
@@ -7678,8 +7678,9 @@ define <4 x float> @global_load_i8_offset_or_i64_imm_offset_4160(ptr addrspace(6
 ; GFX1250-SDAG:       ; %bb.0:
 ; GFX1250-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-SDAG-NEXT:    v_mov_b32_e32 v3, 0
-; GFX1250-SDAG-NEXT:    v_or_b32_e32 v2, 0x1040, v1
+; GFX1250-SDAG-NEXT:    s_mov_b32 s0, 0x1040
+; GFX1250-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT:    v_dual_mov_b32 v3, 0 :: v_dual_bitop2_b32 v2, s0, v1 bitop3:0x54
 ; GFX1250-SDAG-NEXT:    global_load_b128 v[0:3], v[2:3], off scope:SCOPE_SYS
 ; GFX1250-SDAG-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-SDAG-NEXT:    s_set_pc_i64 s[30:31]
@@ -7691,8 +7692,9 @@ define <4 x float> @global_load_i8_offset_or_i64_imm_offset_4160(ptr addrspace(6
 ; GFX1310-SDAG-NEXT:    s_wait_samplecnt 0x0
 ; GFX1310-SDAG-NEXT:    s_wait_bvhcnt 0x0
 ; GFX1310-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX1310-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX1310-SDAG-NEXT:    v_or_b32_e32 v1, 0x1040, v1
+; GFX1310-SDAG-NEXT:    s_mov_b32 s0, 0x1040
+; GFX1310-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1310-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v1, s0, v1 bitop3:0x54
 ; GFX1310-SDAG-NEXT:    global_load_b128 v[0:3], v[1:2], off scope:SCOPE_SYS
 ; GFX1310-SDAG-NEXT:    s_wait_loadcnt 0x0
 ; GFX1310-SDAG-NEXT:    s_set_pc_i64 s[30:31]
@@ -7737,8 +7739,9 @@ define <4 x float> @global_load_i8_offset_or_i64_imm_offset_4160(ptr addrspace(6
 ; GFX1250-ISEL:       ; %bb.0:
 ; GFX1250-ISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-ISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-ISEL-NEXT:    v_mov_b32_e32 v3, 0
-; GFX1250-ISEL-NEXT:    v_or_b32_e32 v2, 0x1040, v1
+; GFX1250-ISEL-NEXT:    s_mov_b32 s0, 0x1040
+; GFX1250-ISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-ISEL-NEXT:    v_dual_mov_b32 v3, 0 :: v_dual_bitop2_b32 v2, s0, v1 bitop3:0x54
 ; GFX1250-ISEL-NEXT:    global_load_b128 v[0:3], v[2:3], off scope:SCOPE_SYS
 ; GFX1250-ISEL-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-ISEL-NEXT:    s_set_pc_i64 s[30:31]
@@ -7750,8 +7753,9 @@ define <4 x float> @global_load_i8_offset_or_i64_imm_offset_4160(ptr addrspace(6
 ; GFX1310-ISEL-NEXT:    s_wait_samplecnt 0x0
 ; GFX1310-ISEL-NEXT:    s_wait_bvhcnt 0x0
 ; GFX1310-ISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX1310-ISEL-NEXT:    v_mov_b32_e32 v2, 0
-; GFX1310-ISEL-NEXT:    v_or_b32_e32 v1, 0x1040, v1
+; GFX1310-ISEL-NEXT:    s_mov_b32 s0, 0x1040
+; GFX1310-ISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1310-ISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v1, s0, v1 bitop3:0x54
 ; GFX1310-ISEL-NEXT:    global_load_b128 v[0:3], v[1:2], off scope:SCOPE_SYS
 ; GFX1310-ISEL-NEXT:    s_wait_loadcnt 0x0
 ; GFX1310-ISEL-NEXT:    s_set_pc_i64 s[30:31]
@@ -16348,8 +16352,9 @@ define <4 x float> @global_load_saddr_i8_offset_or_i64_imm_offset_4160(ptr addrs
 ; GFX1250-SDAG:       ; %bb.0:
 ; GFX1250-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-SDAG-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1250-SDAG-NEXT:    v_or_b32_e32 v0, 0x1040, v0
+; GFX1250-SDAG-NEXT:    s_mov_b32 s0, 0x1040
+; GFX1250-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x54
 ; GFX1250-SDAG-NEXT:    global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
 ; GFX1250-SDAG-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-SDAG-NEXT:    s_set_pc_i64 s[30:31]
@@ -16361,8 +16366,9 @@ define <4 x float> @global_load_saddr_i8_offset_or_i64_imm_offset_4160(ptr addrs
 ; GFX1310-SDAG-NEXT:    s_wait_samplecnt 0x0
 ; GFX1310-SDAG-NEXT:    s_wait_bvhcnt 0x0
 ; GFX1310-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX1310-SDAG-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1310-SDAG-NEXT:    v_or_b32_e32 v0, 0x1040, v0
+; GFX1310-SDAG-NEXT:    s_mov_b32 s0, 0x1040
+; GFX1310-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1310-SDAG-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x54
 ; GFX1310-SDAG-NEXT:    global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
 ; GFX1310-SDAG-NEXT:    s_wait_loadcnt 0x0
 ; GFX1310-SDAG-NEXT:    s_set_pc_i64 s[30:31]
@@ -16407,8 +16413,9 @@ define <4 x float> @global_load_saddr_i8_offset_or_i64_imm_offset_4160(ptr addrs
 ; GFX1250-ISEL:       ; %bb.0:
 ; GFX1250-ISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-ISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-ISEL-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1250-ISEL-NEXT:    v_or_b32_e32 v0, 0x1040, v0
+; GFX1250-ISEL-NEXT:    s_mov_b32 s0, 0x1040
+; GFX1250-ISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-ISEL-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x54
 ; GFX1250-ISEL-NEXT:    global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
 ; GFX1250-ISEL-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-ISEL-NEXT:    s_set_pc_i64 s[30:31]
@@ -16420,8 +16427,9 @@ define <4 x float> @global_load_saddr_i8_offset_or_i64_imm_offset_4160(ptr addrs
 ; GFX1310-ISEL-NEXT:    s_wait_samplecnt 0x0
 ; GFX1310-ISEL-NEXT:    s_wait_bvhcnt 0x0
 ; GFX1310-ISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX1310-ISEL-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1310-ISEL-NEXT:    v_or_b32_e32 v0, 0x1040, v0
+; GFX1310-ISEL-NEXT:    s_mov_b32 s0, 0x1040
+; GFX1310-ISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1310-ISEL-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x54
 ; GFX1310-ISEL-NEXT:    global_load_b128 v[0:3], v[0:1], off scope:SCOPE_DEV
 ; GFX1310-ISEL-NEXT:    s_wait_loadcnt 0x0
 ; GFX1310-ISEL-NEXT:    s_set_pc_i64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.permlane.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.permlane.ll
index 99240b29d2a9e..374f96cc6d687 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.permlane.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.permlane.ll
@@ -1132,12 +1132,12 @@ define amdgpu_kernel void @v_permlane16_b32_vvv_i32(ptr addrspace(1) %out, i32 %
 ; GFX13-GISEL-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
 ; GFX13-GISEL-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
 ; GFX13-GISEL-NEXT:    v_bfe_u32 v0, v0, 10, 10
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v0, s2
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v1, 0
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v0, s3, s4
 ; GFX13-GISEL-NEXT:    global_store_b32 v1, v0, s[0:1]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -1215,39 +1215,23 @@ define amdgpu_kernel void @v_permlane16_b32_vvv_i64(ptr addrspace(1) %out, i64 %
 ; GFX12-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
 ; GFX12-NEXT:    s_endpgm
 ;
-; GFX13-SDAG-LABEL: v_permlane16_b32_vvv_i64:
-; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_bfe_u32 v0, v0, 10, 10
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s4, v1
-; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s5, v0
-; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v0, s2
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v1, s4, s5
-; GFX13-SDAG-NEXT:    v_permlane16_b32 v0, v0, s4, s5
-; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
-; GFX13-SDAG-NEXT:    s_endpgm
-;
-; GFX13-GISEL-LABEL: v_permlane16_b32_vvv_i64:
-; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_bfe_u32 v0, v0, 10, 10
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v1
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s5, v0
-; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v0, s4, s5
-; GFX13-GISEL-NEXT:    v_permlane16_b32 v1, v1, s4, s5
-; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
-; GFX13-GISEL-NEXT:    s_endpgm
+; GFX13-LABEL: v_permlane16_b32_vvv_i64:
+; GFX13:       ; %bb.0:
+; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-NEXT:    v_bfe_u32 v0, v0, 10, 10
+; GFX13-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
+; GFX13-NEXT:    v_readfirstlane_b32 s5, v0
+; GFX13-NEXT:    s_wait_kmcnt 0x0
+; GFX13-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX13-NEXT:    v_mov_b32_e32 v1, s3
+; GFX13-NEXT:    v_permlane16_b32 v0, v0, s4, s5
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX13-NEXT:    v_permlane16_b32 v1, v1, s4, s5
+; GFX13-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
+; GFX13-NEXT:    s_endpgm
   %tidx = call i32 @llvm.amdgcn.workitem.id.x()
   %tidy = call i32 @llvm.amdgcn.workitem.id.y()
   %v = call i64 @llvm.amdgcn.permlane16.i64(i64 %src0, i64 %src0, i32 %tidx, i32 %tidy, i1 false, i1 false)
@@ -1375,12 +1359,12 @@ define amdgpu_kernel void @v_permlane16_b32_vvv_f32(ptr addrspace(1) %out, float
 ; GFX13-GISEL-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
 ; GFX13-GISEL-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
 ; GFX13-GISEL-NEXT:    v_bfe_u32 v0, v0, 10, 10
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v0, s2
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v1, 0
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v0, s3, s4
 ; GFX13-GISEL-NEXT:    global_store_b32 v1, v0, s[0:1]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -1504,14 +1488,15 @@ define amdgpu_kernel void @v_permlane16_b32_vvv_f64(ptr addrspace(1) %out, doubl
 ; GFX13-SDAG-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
 ; GFX13-SDAG-NEXT:    v_bfe_u32 v0, v0, 10, 10
 ; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s5, v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v0, s2
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v1, s4, s5
+; GFX13-SDAG-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX13-SDAG-NEXT:    v_mov_b32_e32 v1, s3
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v0, v0, s4, s5
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v1, s4, s5
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
 ; GFX13-SDAG-NEXT:    s_endpgm
 ;
@@ -1520,12 +1505,13 @@ define amdgpu_kernel void @v_permlane16_b32_vvv_f64(ptr addrspace(1) %out, doubl
 ; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
 ; GFX13-GISEL-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
 ; GFX13-GISEL-NEXT:    v_bfe_u32 v0, v0, 10, 10
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s5, v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v2, s2
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v3, s3
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, s2 :: v_dual_mov_b32 v1, s3
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v3, s3
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v2, s4, s5
 ; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
@@ -1615,10 +1601,10 @@ define amdgpu_kernel void @v_permlane16_b32_vvs_i32(ptr addrspace(1) %out, i32 %
 ; GFX13-SDAG-LABEL: v_permlane16_b32_vvs_i32:
 ; GFX13-SDAG:       ; %bb.0:
 ; GFX13-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v1, s2
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s2 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s2, v0
 ; GFX13-SDAG-NEXT:    v_mov_b32_e32 v0, 0
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v1, s2, s3
@@ -1628,12 +1614,13 @@ define amdgpu_kernel void @v_permlane16_b32_vvs_i32(ptr addrspace(1) %out, i32 %
 ; GFX13-GISEL-LABEL: v_permlane16_b32_vvs_i32:
 ; GFX13-GISEL:       ; %bb.0:
 ; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v1, 0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v0, s4, s3
 ; GFX13-GISEL-NEXT:    global_store_b32 v1, v0, s[0:1]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -1710,39 +1697,22 @@ define amdgpu_kernel void @v_permlane16_b32_vvs_i64(ptr addrspace(1) %out, i64 %
 ; GFX12-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
 ; GFX12-NEXT:    s_endpgm
 ;
-; GFX13-SDAG-LABEL: v_permlane16_b32_vvs_i64:
-; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    s_clause 0x1
-; GFX13-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    s_load_b32 s4, s[4:5], 0x34 nv
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
-; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s5, v0
-; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v0, s2
-; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v1, s5, s4
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
-; GFX13-SDAG-NEXT:    v_permlane16_b32 v0, v0, s5, s4
-; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
-; GFX13-SDAG-NEXT:    s_endpgm
-;
-; GFX13-GISEL-LABEL: v_permlane16_b32_vvs_i64:
-; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    s_clause 0x1
-; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    s_load_b32 s4, s[4:5], 0x34 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s5, v0
-; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
-; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v0, s5, s4
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_permlane16_b32 v1, v1, s5, s4
-; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
-; GFX13-GISEL-NEXT:    s_endpgm
+; GFX13-LABEL: v_permlane16_b32_vvs_i64:
+; GFX13:       ; %bb.0:
+; GFX13-NEXT:    s_clause 0x1
+; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-NEXT:    s_load_b32 s4, s[4:5], 0x34 nv
+; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-NEXT:    s_wait_kmcnt 0x0
+; GFX13-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s3
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX13-NEXT:    v_readfirstlane_b32 s5, v0
+; GFX13-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-NEXT:    v_permlane16_b32 v1, v1, s5, s4
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX13-NEXT:    v_permlane16_b32 v0, v0, s5, s4
+; GFX13-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
+; GFX13-NEXT:    s_endpgm
   %tidx = call i32 @llvm.amdgcn.workitem.id.x()
   %v = call i64 @llvm.amdgcn.permlane16.i64(i64 %src0, i64 %src0, i32 %tidx, i32 %src2, i1 false, i1 false)
   store i64 %v, ptr addrspace(1) %out
@@ -1825,10 +1795,10 @@ define amdgpu_kernel void @v_permlane16_b32_vvs_f32(ptr addrspace(1) %out, float
 ; GFX13-SDAG-LABEL: v_permlane16_b32_vvs_f32:
 ; GFX13-SDAG:       ; %bb.0:
 ; GFX13-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v1, s2
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s2 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s2, v0
 ; GFX13-SDAG-NEXT:    v_mov_b32_e32 v0, 0
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v1, s2, s3
@@ -1838,12 +1808,13 @@ define amdgpu_kernel void @v_permlane16_b32_vvs_f32(ptr addrspace(1) %out, float
 ; GFX13-GISEL-LABEL: v_permlane16_b32_vvs_f32:
 ; GFX13-GISEL:       ; %bb.0:
 ; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v1, 0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v0, s4, s3
 ; GFX13-GISEL-NEXT:    global_store_b32 v1, v0, s[0:1]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -1964,11 +1935,11 @@ define amdgpu_kernel void @v_permlane16_b32_vvs_f64(ptr addrspace(1) %out, doubl
 ; GFX13-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
 ; GFX13-SDAG-NEXT:    s_load_b32 s4, s[4:5], 0x34 nv
 ; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
-; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s5, v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v0, s2
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s3
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s5, v0
+; GFX13-SDAG-NEXT:    v_mov_b32_e32 v0, s2
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v1, s5, s4
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v0, v0, s5, s4
@@ -1980,12 +1951,13 @@ define amdgpu_kernel void @v_permlane16_b32_vvs_f64(ptr addrspace(1) %out, doubl
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
 ; GFX13-GISEL-NEXT:    s_load_b32 s4, s[4:5], 0x34 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s5, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s5, 0x3ff
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v2, s2
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v3, s3
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, s2 :: v_dual_bitop2_b32 v0, s5, v0 bitop3:0x40
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s5, v0
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v3, s3
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v2, s5, s4
 ; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_3)
@@ -4648,12 +4620,12 @@ define amdgpu_kernel void @v_permlanex16_b32_vvv_i32(ptr addrspace(1) %out, i32
 ; GFX13-GISEL-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
 ; GFX13-GISEL-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
 ; GFX13-GISEL-NEXT:    v_bfe_u32 v0, v0, 10, 10
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v0, s2
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v1, 0
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v0, s3, s4
 ; GFX13-GISEL-NEXT:    global_store_b32 v1, v0, s[0:1]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -4784,12 +4756,12 @@ define amdgpu_kernel void @v_permlanex16_b32_vvv_f32(ptr addrspace(1) %out, floa
 ; GFX13-GISEL-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
 ; GFX13-GISEL-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
 ; GFX13-GISEL-NEXT:    v_bfe_u32 v0, v0, 10, 10
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v0, s2
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s3, v1
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v1, 0
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v0, s3, s4
 ; GFX13-GISEL-NEXT:    global_store_b32 v1, v0, s[0:1]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -4867,39 +4839,23 @@ define amdgpu_kernel void @v_permlanex16_b32_vvv_i64(ptr addrspace(1) %out, i64
 ; GFX12-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
 ; GFX12-NEXT:    s_endpgm
 ;
-; GFX13-SDAG-LABEL: v_permlanex16_b32_vvv_i64:
-; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_bfe_u32 v0, v0, 10, 10
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s4, v1
-; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s5, v0
-; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v0, s2
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v1, s4, s5
-; GFX13-SDAG-NEXT:    v_permlanex16_b32 v0, v0, s4, s5
-; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
-; GFX13-SDAG-NEXT:    s_endpgm
-;
-; GFX13-GISEL-LABEL: v_permlanex16_b32_vvv_i64:
-; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_bfe_u32 v0, v0, 10, 10
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v1
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s5, v0
-; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v0, s4, s5
-; GFX13-GISEL-NEXT:    v_permlanex16_b32 v1, v1, s4, s5
-; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
-; GFX13-GISEL-NEXT:    s_endpgm
+; GFX13-LABEL: v_permlanex16_b32_vvv_i64:
+; GFX13:       ; %bb.0:
+; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
+; GFX13-NEXT:    v_bfe_u32 v0, v0, 10, 10
+; GFX13-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
+; GFX13-NEXT:    v_readfirstlane_b32 s5, v0
+; GFX13-NEXT:    s_wait_kmcnt 0x0
+; GFX13-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX13-NEXT:    v_mov_b32_e32 v1, s3
+; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s4, s5
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX13-NEXT:    v_permlanex16_b32 v1, v1, s4, s5
+; GFX13-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
+; GFX13-NEXT:    s_endpgm
   %tidx = call i32 @llvm.amdgcn.workitem.id.x()
   %tidy = call i32 @llvm.amdgcn.workitem.id.y()
   %v = call i64 @llvm.amdgcn.permlanex16.i64(i64 %src0, i64 %src0, i32 %tidx, i32 %tidy, i1 false, i1 false)
@@ -5020,14 +4976,15 @@ define amdgpu_kernel void @v_permlanex16_b32_vvv_f64(ptr addrspace(1) %out, doub
 ; GFX13-SDAG-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
 ; GFX13-SDAG-NEXT:    v_bfe_u32 v0, v0, 10, 10
 ; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
-; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_4) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s5, v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v0, s2
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v1, s4, s5
+; GFX13-SDAG-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX13-SDAG-NEXT:    v_mov_b32_e32 v1, s3
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v0, v0, s4, s5
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v1, s4, s5
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
 ; GFX13-SDAG-NEXT:    s_endpgm
 ;
@@ -5036,12 +4993,13 @@ define amdgpu_kernel void @v_permlanex16_b32_vvv_f64(ptr addrspace(1) %out, doub
 ; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
 ; GFX13-GISEL-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
 ; GFX13-GISEL-NEXT:    v_bfe_u32 v0, v0, 10, 10
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s5, v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v2, s2
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v3, s3
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v1
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, s2 :: v_dual_mov_b32 v1, s3
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v3, s3
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_3)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v2, s4, s5
 ; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
@@ -5131,10 +5089,10 @@ define amdgpu_kernel void @v_permlanex16_b32_vvs_i32(ptr addrspace(1) %out, i32
 ; GFX13-SDAG-LABEL: v_permlanex16_b32_vvs_i32:
 ; GFX13-SDAG:       ; %bb.0:
 ; GFX13-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v1, s2
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s2 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s2, v0
 ; GFX13-SDAG-NEXT:    v_mov_b32_e32 v0, 0
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v1, s2, s3
@@ -5144,12 +5102,13 @@ define amdgpu_kernel void @v_permlanex16_b32_vvs_i32(ptr addrspace(1) %out, i32
 ; GFX13-GISEL-LABEL: v_permlanex16_b32_vvs_i32:
 ; GFX13-GISEL:       ; %bb.0:
 ; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v1, 0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v0, s4, s3
 ; GFX13-GISEL-NEXT:    global_store_b32 v1, v0, s[0:1]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -5235,10 +5194,10 @@ define amdgpu_kernel void @v_permlanex16_b32_vvs_f32(ptr addrspace(1) %out, floa
 ; GFX13-SDAG-LABEL: v_permlanex16_b32_vvs_f32:
 ; GFX13-SDAG:       ; %bb.0:
 ; GFX13-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v1, s2
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s2 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s2, v0
 ; GFX13-SDAG-NEXT:    v_mov_b32_e32 v0, 0
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v1, s2, s3
@@ -5248,12 +5207,13 @@ define amdgpu_kernel void @v_permlanex16_b32_vvs_f32(ptr addrspace(1) %out, floa
 ; GFX13-GISEL-LABEL: v_permlanex16_b32_vvs_f32:
 ; GFX13-GISEL:       ; %bb.0:
 ; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v1, 0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s4, v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v0, s4, s3
 ; GFX13-GISEL-NEXT:    global_store_b32 v1, v0, s[0:1]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -5330,39 +5290,22 @@ define amdgpu_kernel void @v_permlanex16_b32_vvs_i64(ptr addrspace(1) %out, i64
 ; GFX12-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
 ; GFX12-NEXT:    s_endpgm
 ;
-; GFX13-SDAG-LABEL: v_permlanex16_b32_vvs_i64:
-; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    s_clause 0x1
-; GFX13-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    s_load_b32 s4, s[4:5], 0x34 nv
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
-; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s5, v0
-; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v0, s2
-; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v1, s5, s4
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
-; GFX13-SDAG-NEXT:    v_permlanex16_b32 v0, v0, s5, s4
-; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
-; GFX13-SDAG-NEXT:    s_endpgm
-;
-; GFX13-GISEL-LABEL: v_permlanex16_b32_vvs_i64:
-; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    s_clause 0x1
-; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    s_load_b32 s4, s[4:5], 0x34 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s5, v0
-; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
-; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v0, s5, s4
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_permlanex16_b32 v1, v1, s5, s4
-; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
-; GFX13-GISEL-NEXT:    s_endpgm
+; GFX13-LABEL: v_permlanex16_b32_vvs_i64:
+; GFX13:       ; %bb.0:
+; GFX13-NEXT:    s_clause 0x1
+; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX13-NEXT:    s_load_b32 s4, s[4:5], 0x34 nv
+; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-NEXT:    s_wait_kmcnt 0x0
+; GFX13-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s3
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX13-NEXT:    v_readfirstlane_b32 s5, v0
+; GFX13-NEXT:    v_mov_b32_e32 v0, s2
+; GFX13-NEXT:    v_permlanex16_b32 v1, v1, s5, s4
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
+; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s5, s4
+; GFX13-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
+; GFX13-NEXT:    s_endpgm
   %tidx = call i32 @llvm.amdgcn.workitem.id.x()
   %v = call i64 @llvm.amdgcn.permlanex16.i64(i64 %src0, i64 %src0, i32 %tidx, i32 %src2, i1 false, i1 false)
   store i64 %v, ptr addrspace(1) %out
@@ -5480,11 +5423,11 @@ define amdgpu_kernel void @v_permlanex16_b32_vvs_f64(ptr addrspace(1) %out, doub
 ; GFX13-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
 ; GFX13-SDAG-NEXT:    s_load_b32 s4, s[4:5], 0x34 nv
 ; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_2) | instid1(VALU_DEP_1)
-; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s5, v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v0, s2
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, s3
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_readfirstlane_b32 s5, v0
+; GFX13-SDAG-NEXT:    v_mov_b32_e32 v0, s2
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v1, s5, s4
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v0, v0, s5, s4
@@ -5496,12 +5439,13 @@ define amdgpu_kernel void @v_permlanex16_b32_vvs_f64(ptr addrspace(1) %out, doub
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
 ; GFX13-GISEL-NEXT:    s_load_b32 s4, s[4:5], 0x34 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_2)
-; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s5, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s5, 0x3ff
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v2, s2
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, s3 :: v_dual_mov_b32 v3, s3
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, s2 :: v_dual_bitop2_b32 v0, s5, v0 bitop3:0x40
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX13-GISEL-NEXT:    v_readfirstlane_b32 s5, v0
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX13-GISEL-NEXT:    v_mov_b32_e32 v3, s3
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v2, s5, s4
 ; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_3)
@@ -7091,10 +7035,10 @@ define amdgpu_kernel void @v_permlane16_b32_tid_tid_i32(ptr addrspace(1) %out, i
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlane16_b32 v0, v0, s0, s1
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -7145,10 +7089,10 @@ define amdgpu_kernel void @v_permlane16_b32_tid_tid_f32(ptr addrspace(1) %out, i
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlane16_b32 v0, v0, s0, s1
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -7247,11 +7191,13 @@ define amdgpu_kernel void @v_permlane16_b32_tid_tid_i64(ptr addrspace(1) %out, i
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v2, 0
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
+; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v1, s0, s1
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_3)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v0, v0, s0, s1
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -7365,11 +7311,12 @@ define amdgpu_kernel void @v_permlane16_b32_tid_tid_f64(ptr addrspace(1) %out, f
 ;
 ; GFX13-SDAG-LABEL: v_permlane16_b32_tid_tid_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -7380,11 +7327,12 @@ define amdgpu_kernel void @v_permlane16_b32_tid_tid_f64(ptr addrspace(1) %out, f
 ;
 ; GFX13-GISEL-LABEL: v_permlane16_b32_tid_tid_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -7441,10 +7389,10 @@ define amdgpu_kernel void @v_permlane16_b32_undef_tid_i32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlane16_b32 v0, v0, s0, s1
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -7496,10 +7444,10 @@ define amdgpu_kernel void @v_permlane16_b32_undef_tid_f32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlane16_b32 v0, v0, s0, s1
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -7597,11 +7545,12 @@ define amdgpu_kernel void @v_permlane16_b32_undef_tid_i64(ptr addrspace(1) %out,
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v2, s0, s1
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v0, v0, s0, s1
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -7611,11 +7560,12 @@ define amdgpu_kernel void @v_permlane16_b32_undef_tid_i64(ptr addrspace(1) %out,
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v0, s0, s1
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v1, v2, s0, s1
 ; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -7716,11 +7666,12 @@ define amdgpu_kernel void @v_permlane16_b32_undef_tid_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-SDAG-LABEL: v_permlane16_b32_undef_tid_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -7731,11 +7682,12 @@ define amdgpu_kernel void @v_permlane16_b32_undef_tid_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-GISEL-LABEL: v_permlane16_b32_undef_tid_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -7837,7 +7789,7 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_i32(ptr addrspace(1) %out, i32
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
 ; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, 0x3039 :: v_dual_mov_b32 v2, 0
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, 0x3039
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v0, s0, s1
@@ -7948,7 +7900,7 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_f32(ptr addrspace(1) %out, i32
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
 ; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, 0x449a5000 :: v_dual_mov_b32 v2, 0
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, 0x449a5000
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v0, s0, s1
@@ -8066,11 +8018,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_i64(ptr addrspace(1) %out, i32
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
 ; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, 0x3039
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v3, 0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v3, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v2, v2, s0, s1
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v0, s0, s1
 ; GFX13-SDAG-NEXT:    global_store_b64 v3, v[1:2], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -8081,10 +8034,10 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_i64(ptr addrspace(1) %out, i32
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
 ; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, 0x3039 :: v_dual_mov_b32 v2, 0
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v3, 0
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v3, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v1, v0, s0, s1
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v2, v2, s0, s1
 ; GFX13-GISEL-NEXT:    global_store_b64 v3, v[1:2], s[2:3]
@@ -8193,12 +8146,13 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_f64(ptr addrspace(1) %out, i32
 ;
 ; GFX13-SDAG-LABEL: v_permlane16_b32_i_tid_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v3, 0x40934a00 :: v_dual_mov_b32 v2, 0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v4, 0
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v3, 0x40934a00 :: v_dual_mov_b32 v4, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -8209,12 +8163,13 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_f64(ptr addrspace(1) %out, i32
 ;
 ; GFX13-GISEL-LABEL: v_permlane16_b32_i_tid_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v3, 0x40934a00
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v4, 0
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v3, 0x40934a00 :: v_dual_mov_b32 v4, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -8271,10 +8226,10 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_i32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[1,0]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -8326,10 +8281,10 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_f32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[1,0]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -8427,11 +8382,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_i64(ptr addrspace(1) %out,
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v2, s0, s1 op_sel:[1,0]
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[1,0]
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -8441,11 +8397,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_i64(ptr addrspace(1) %out,
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[1,0]
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v1, v2, s0, s1 op_sel:[1,0]
 ; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -8546,11 +8503,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-SDAG-LABEL: v_permlane16_b32_i_tid_fi_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -8561,11 +8519,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-GISEL-LABEL: v_permlane16_b32_i_tid_fi_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -8623,10 +8582,10 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_bc_i32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[0,1]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -8678,10 +8637,10 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_bc_f32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[0,1]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -8779,11 +8738,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_bc_i64(ptr addrspace(1) %out,
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v2, s0, s1 op_sel:[0,1]
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[0,1]
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -8793,11 +8753,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_bc_i64(ptr addrspace(1) %out,
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[0,1]
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v1, v2, s0, s1 op_sel:[0,1]
 ; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -8898,11 +8859,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_bc_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-SDAG-LABEL: v_permlane16_b32_i_tid_bc_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -8913,11 +8875,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_bc_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-GISEL-LABEL: v_permlane16_b32_i_tid_bc_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -8975,10 +8938,10 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_bc_i32(ptr addrspace(1) %ou
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[1,1]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -9030,10 +8993,10 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_bc_f32(ptr addrspace(1) %ou
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[1,1]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -9131,11 +9094,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_bc_i64(ptr addrspace(1) %ou
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v1, v2, s0, s1 op_sel:[1,1]
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[1,1]
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -9145,11 +9109,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_bc_i64(ptr addrspace(1) %ou
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v0, v0, s0, s1 op_sel:[1,1]
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlane16_b32 v1, v2, s0, s1 op_sel:[1,1]
 ; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -9250,11 +9215,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_bc_f64(ptr addrspace(1) %ou
 ;
 ; GFX13-SDAG-LABEL: v_permlane16_b32_i_tid_fi_bc_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -9265,11 +9231,12 @@ define amdgpu_kernel void @v_permlane16_b32_i_tid_fi_bc_f64(ptr addrspace(1) %ou
 ;
 ; GFX13-GISEL-LABEL: v_permlane16_b32_i_tid_fi_bc_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -9327,10 +9294,10 @@ define amdgpu_kernel void @v_permlanex16_b32_tid_tid_i32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s0, s1
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -9381,10 +9348,10 @@ define amdgpu_kernel void @v_permlanex16_b32_tid_tid_f32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s0, s1
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -9483,11 +9450,13 @@ define amdgpu_kernel void @v_permlanex16_b32_tid_tid_i64(ptr addrspace(1) %out,
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v2, 0
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
+; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v1, s0, s1
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_3)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v0, v0, s0, s1
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -9601,11 +9570,12 @@ define amdgpu_kernel void @v_permlanex16_b32_tid_tid_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-SDAG-LABEL: v_permlanex16_b32_tid_tid_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -9616,11 +9586,12 @@ define amdgpu_kernel void @v_permlanex16_b32_tid_tid_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-GISEL-LABEL: v_permlanex16_b32_tid_tid_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -9677,10 +9648,10 @@ define amdgpu_kernel void @v_permlanex16_b32_undef_tid_i32(ptr addrspace(1) %out
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s0, s1
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -9732,10 +9703,10 @@ define amdgpu_kernel void @v_permlanex16_b32_undef_tid_f32(ptr addrspace(1) %out
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s0, s1
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -9833,11 +9804,12 @@ define amdgpu_kernel void @v_permlanex16_b32_undef_tid_i64(ptr addrspace(1) %out
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v2, s0, s1
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v0, v0, s0, s1
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -9847,11 +9819,12 @@ define amdgpu_kernel void @v_permlanex16_b32_undef_tid_i64(ptr addrspace(1) %out
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v0, s0, s1
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v1, v2, s0, s1
 ; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -9952,11 +9925,12 @@ define amdgpu_kernel void @v_permlanex16_b32_undef_tid_f64(ptr addrspace(1) %out
 ;
 ; GFX13-SDAG-LABEL: v_permlanex16_b32_undef_tid_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -9967,11 +9941,12 @@ define amdgpu_kernel void @v_permlanex16_b32_undef_tid_f64(ptr addrspace(1) %out
 ;
 ; GFX13-GISEL-LABEL: v_permlanex16_b32_undef_tid_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -10073,7 +10048,7 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_i32(ptr addrspace(1) %out, i3
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
 ; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, 0x3039 :: v_dual_mov_b32 v2, 0
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, 0x3039
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v0, s0, s1
@@ -10184,7 +10159,7 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_f32(ptr addrspace(1) %out, i3
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
 ; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v1, 0x449a5000 :: v_dual_mov_b32 v2, 0
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, 0x449a5000
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v0, s0, s1
@@ -10302,11 +10277,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_i64(ptr addrspace(1) %out, i3
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
 ; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, 0x3039
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v3, 0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v3, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v2, v2, s0, s1
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v0, s0, s1
 ; GFX13-SDAG-NEXT:    global_store_b64 v3, v[1:2], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -10317,10 +10293,10 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_i64(ptr addrspace(1) %out, i3
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
 ; GFX13-GISEL-NEXT:    v_dual_mov_b32 v1, 0x3039 :: v_dual_mov_b32 v2, 0
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v3, 0
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v3, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v1, v0, s0, s1
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v2, v2, s0, s1
 ; GFX13-GISEL-NEXT:    global_store_b64 v3, v[1:2], s[2:3]
@@ -10429,12 +10405,13 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_f64(ptr addrspace(1) %out, i3
 ;
 ; GFX13-SDAG-LABEL: v_permlanex16_b32_i_tid_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_dual_mov_b32 v3, 0x40934a00 :: v_dual_mov_b32 v2, 0
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v4, 0
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v3, 0x40934a00 :: v_dual_mov_b32 v4, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -10445,12 +10422,13 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_f64(ptr addrspace(1) %out, i3
 ;
 ; GFX13-GISEL-LABEL: v_permlanex16_b32_i_tid_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v3, 0x40934a00
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v4, 0
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v3, 0x40934a00 :: v_dual_mov_b32 v4, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -10507,10 +10485,10 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_i32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[1,0]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -10562,10 +10540,10 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_f32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[1,0]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -10663,11 +10641,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_i64(ptr addrspace(1) %out,
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v2, s0, s1 op_sel:[1,0]
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[1,0]
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -10677,11 +10656,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_i64(ptr addrspace(1) %out,
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[1,0]
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v1, v2, s0, s1 op_sel:[1,0]
 ; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -10782,11 +10762,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-SDAG-LABEL: v_permlanex16_b32_i_tid_fi_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -10797,11 +10778,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-GISEL-LABEL: v_permlanex16_b32_i_tid_fi_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -10859,10 +10841,10 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_bc_i32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[0,1]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -10914,10 +10896,10 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_bc_f32(ptr addrspace(1) %out,
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[0,1]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -11015,11 +10997,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_bc_i64(ptr addrspace(1) %out,
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v2, s0, s1 op_sel:[0,1]
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[0,1]
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -11029,11 +11012,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_bc_i64(ptr addrspace(1) %out,
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[0,1]
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v1, v2, s0, s1 op_sel:[0,1]
 ; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -11134,11 +11118,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_bc_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-SDAG-LABEL: v_permlanex16_b32_i_tid_bc_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -11149,11 +11134,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_bc_f64(ptr addrspace(1) %out,
 ;
 ; GFX13-GISEL-LABEL: v_permlanex16_b32_i_tid_bc_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -11211,10 +11197,10 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_bc_i32(ptr addrspace(1) %o
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[1,1]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -11266,10 +11252,10 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_bc_f32(ptr addrspace(1) %o
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[1,1]
 ; GFX13-NEXT:    global_store_b32 v1, v0, s[2:3]
 ; GFX13-NEXT:    s_endpgm
@@ -11367,11 +11353,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_bc_i64(ptr addrspace(1) %o
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
-; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v1, v2, s0, s1 op_sel:[1,1]
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-SDAG-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[1,1]
 ; GFX13-SDAG-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-SDAG-NEXT:    s_endpgm
@@ -11381,11 +11368,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_bc_i64(ptr addrspace(1) %o
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-GISEL-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v0, v0, s0, s1 op_sel:[1,1]
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX13-GISEL-NEXT:    v_permlanex16_b32 v1, v2, s0, s1 op_sel:[1,1]
 ; GFX13-GISEL-NEXT:    global_store_b64 v2, v[0:1], s[2:3]
 ; GFX13-GISEL-NEXT:    s_endpgm
@@ -11486,11 +11474,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_bc_f64(ptr addrspace(1) %o
 ;
 ; GFX13-SDAG-LABEL: v_permlanex16_b32_i_tid_fi_bc_f64:
 ; GFX13-SDAG:       ; %bb.0:
-; GFX13-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-SDAG-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-SDAG-NEXT:    s_clause 0x1
 ; GFX13-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-SDAG-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-SDAG-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-SDAG-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -11501,11 +11490,12 @@ define amdgpu_kernel void @v_permlanex16_b32_i_tid_fi_bc_f64(ptr addrspace(1) %o
 ;
 ; GFX13-GISEL-LABEL: v_permlanex16_b32_i_tid_fi_bc_f64:
 ; GFX13-GISEL:       ; %bb.0:
-; GFX13-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-GISEL-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX13-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-GISEL-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX13-GISEL-NEXT:    s_clause 0x1
 ; GFX13-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x30 nv
 ; GFX13-GISEL-NEXT:    s_load_b64 s[2:3], s[4:5], 0x24 nv
-; GFX13-GISEL-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX13-GISEL-NEXT:    v_cvt_f64_f32_e32 v[0:1], v0
 ; GFX13-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
@@ -14205,11 +14195,11 @@ define amdgpu_kernel void @v_permlanex16_convergent(ptr addrspace(1) %out, i32 %
 ; GFX13-LABEL: v_permlanex16_convergent:
 ; GFX13:       ; %bb.0:
 ; GFX13-NEXT:    s_load_b96 s[0:2], s[4:5], 0x2c nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-NEXT:    s_mov_b32 s3, 0x3ff
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
-; GFX13-NEXT:    v_mov_b32_e32 v1, s0
+; GFX13-NEXT:    v_dual_mov_b32 v1, s0 :: v_dual_bitop2_b32 v0, s3, v0 bitop3:0x40
 ; GFX13-NEXT:    s_mov_b32 s0, exec_lo
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX13-NEXT:    v_permlanex16_b32 v1, v1, s1, s2
 ; GFX13-NEXT:    v_cmpx_eq_u32_e32 0, v0
 ; GFX13-NEXT:    s_cbranch_execz .LBB142_2
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ptr.s.buffer.load-gfx12.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ptr.s.buffer.load-gfx12.ll
index 1a05c8ff127f5..3a8273ed28431 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ptr.s.buffer.load-gfx12.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.ptr.s.buffer.load-gfx12.ll
@@ -167,8 +167,9 @@ define amdgpu_kernel void @ptr_s_buffer_load_i8_divergent_offset(ptr addrspace(1
 ; GFX1250-SDAG-NEXT:    v_nop
 ; GFX1250-SDAG-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x10 nv
-; GFX1250-SDAG-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-SDAG-NEXT:    v_mov_b32_e32 v1, 0
+; GFX1250-SDAG-NEXT:    s_mov_b32 s6, 0x3ff
+; GFX1250-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-SDAG-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s6, v0 bitop3:0x40
 ; GFX1250-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-SDAG-NEXT:    buffer_load_u8 v0, v0, s[0:3], null offen nv
 ; GFX1250-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
@@ -183,9 +184,10 @@ define amdgpu_kernel void @ptr_s_buffer_load_i8_divergent_offset(ptr addrspace(1
 ; GFX1250-GISEL-NEXT:    v_nop
 ; GFX1250-GISEL-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-GISEL-NEXT:    s_load_b128 s[0:3], s[4:5], 0x10 nv
-; GFX1250-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX1250-GISEL-NEXT:    s_mov_b32 s6, 0x3ff
+; GFX1250-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s6, v0 bitop3:0x40
 ; GFX1250-GISEL-NEXT:    s_load_b64 s[4:5], s[4:5], 0x0 nv
-; GFX1250-GISEL-NEXT:    v_mov_b32_e32 v1, 0
 ; GFX1250-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-GISEL-NEXT:    buffer_load_u8 v0, v0, s[0:3], null offen nv
 ; GFX1250-GISEL-NEXT:    s_wait_loadcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.atomic.buffer.load.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.atomic.buffer.load.ll
index 37647c5acec4e..ee7b058104040 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.atomic.buffer.load.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.atomic.buffer.load.ll
@@ -482,101 +482,30 @@ define amdgpu_kernel void @raw_atomic_buffer_load_i64(<4 x i32> %addr) {
 ; GFX11-NEXT:  ; %bb.2: ; %bb2
 ; GFX11-NEXT:    s_endpgm
 ;
-; GFX12-SDAG-TRUE16-LABEL: raw_atomic_buffer_load_i64:
-; GFX12-SDAG-TRUE16:       ; %bb.0: ; %bb
-; GFX12-SDAG-TRUE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-SDAG-TRUE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-SDAG-TRUE16-NEXT:    v_nop
-; GFX12-SDAG-TRUE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-SDAG-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-SDAG-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-SDAG-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-SDAG-TRUE16-NEXT:  .LBB5_1: ; %bb1
-; GFX12-SDAG-TRUE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    buffer_load_b64 v[2:3], off, s[0:3], null offset:4 th:TH_LOAD_NT
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
-; GFX12-SDAG-TRUE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-SDAG-TRUE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-SDAG-TRUE16-NEXT:    s_cbranch_execnz .LBB5_1
-; GFX12-SDAG-TRUE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-SDAG-TRUE16-NEXT:    s_endpgm
-;
-; GFX12-SDAG-FAKE16-LABEL: raw_atomic_buffer_load_i64:
-; GFX12-SDAG-FAKE16:       ; %bb.0: ; %bb
-; GFX12-SDAG-FAKE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-SDAG-FAKE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-SDAG-FAKE16-NEXT:    v_nop
-; GFX12-SDAG-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-SDAG-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-SDAG-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-SDAG-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-SDAG-FAKE16-NEXT:  .LBB5_1: ; %bb1
-; GFX12-SDAG-FAKE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    buffer_load_b64 v[2:3], off, s[0:3], null offset:4 th:TH_LOAD_NT
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
-; GFX12-SDAG-FAKE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-SDAG-FAKE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-SDAG-FAKE16-NEXT:    s_cbranch_execnz .LBB5_1
-; GFX12-SDAG-FAKE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-SDAG-FAKE16-NEXT:    s_endpgm
-;
-; GFX12-GISEL-TRUE16-LABEL: raw_atomic_buffer_load_i64:
-; GFX12-GISEL-TRUE16:       ; %bb.0: ; %bb
-; GFX12-GISEL-TRUE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-GISEL-TRUE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-GISEL-TRUE16-NEXT:    v_nop
-; GFX12-GISEL-TRUE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-GISEL-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-GISEL-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-GISEL-TRUE16-NEXT:  .LBB5_1: ; %bb1
-; GFX12-GISEL-TRUE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    buffer_load_b64 v[2:3], off, s[0:3], null offset:4 th:TH_LOAD_NT
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
-; GFX12-GISEL-TRUE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-TRUE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-GISEL-TRUE16-NEXT:    s_cbranch_execnz .LBB5_1
-; GFX12-GISEL-TRUE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-GISEL-TRUE16-NEXT:    s_endpgm
-;
-; GFX12-GISEL-FAKE16-LABEL: raw_atomic_buffer_load_i64:
-; GFX12-GISEL-FAKE16:       ; %bb.0: ; %bb
-; GFX12-GISEL-FAKE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-GISEL-FAKE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-GISEL-FAKE16-NEXT:    v_nop
-; GFX12-GISEL-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-GISEL-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-GISEL-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-GISEL-FAKE16-NEXT:  .LBB5_1: ; %bb1
-; GFX12-GISEL-FAKE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    buffer_load_b64 v[2:3], off, s[0:3], null offset:4 th:TH_LOAD_NT
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
-; GFX12-GISEL-FAKE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-FAKE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-GISEL-FAKE16-NEXT:    s_cbranch_execnz .LBB5_1
-; GFX12-GISEL-FAKE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-GISEL-FAKE16-NEXT:    s_endpgm
+; GFX12-LABEL: raw_atomic_buffer_load_i64:
+; GFX12:       ; %bb.0: ; %bb
+; GFX12-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX12-NEXT:    s_mov_b64 s[64:65], 0
+; GFX12-NEXT:    v_nop
+; GFX12-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX12-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX12-NEXT:    s_wait_xcnt 0x0
+; GFX12-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX12-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
+; GFX12-NEXT:    s_mov_b32 s4, 0
+; GFX12-NEXT:  .LBB5_1: ; %bb1
+; GFX12-NEXT:    ; =>This Inner Loop Header: Depth=1
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    buffer_load_b64 v[2:3], off, s[0:3], null offset:4 th:TH_LOAD_NT
+; GFX12-NEXT:    s_wait_loadcnt 0x0
+; GFX12-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
+; GFX12-NEXT:    s_or_b32 s4, vcc_lo, s4
+; GFX12-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT:    s_cbranch_execnz .LBB5_1
+; GFX12-NEXT:  ; %bb.2: ; %bb2
+; GFX12-NEXT:    s_endpgm
 bb:
   %id = tail call i32 @llvm.amdgcn.workitem.id.x()
   %id.zext = zext i32 %id to i64
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.ptr.atomic.buffer.load.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.ptr.atomic.buffer.load.ll
index ca2679f6b155d..9a919c269eaf8 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.ptr.atomic.buffer.load.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.raw.ptr.atomic.buffer.load.ll
@@ -482,101 +482,30 @@ define amdgpu_kernel void @raw_ptr_atomic_buffer_load_i64(ptr addrspace(8) %ptr)
 ; GFX11-NEXT:  ; %bb.2: ; %bb2
 ; GFX11-NEXT:    s_endpgm
 ;
-; GFX12-SDAG-TRUE16-LABEL: raw_ptr_atomic_buffer_load_i64:
-; GFX12-SDAG-TRUE16:       ; %bb.0: ; %bb
-; GFX12-SDAG-TRUE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-SDAG-TRUE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-SDAG-TRUE16-NEXT:    v_nop
-; GFX12-SDAG-TRUE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-SDAG-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-SDAG-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-SDAG-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-SDAG-TRUE16-NEXT:  .LBB5_1: ; %bb1
-; GFX12-SDAG-TRUE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    buffer_load_b64 v[2:3], off, s[0:3], null offset:4 th:TH_LOAD_NT
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
-; GFX12-SDAG-TRUE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-SDAG-TRUE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-SDAG-TRUE16-NEXT:    s_cbranch_execnz .LBB5_1
-; GFX12-SDAG-TRUE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-SDAG-TRUE16-NEXT:    s_endpgm
-;
-; GFX12-SDAG-FAKE16-LABEL: raw_ptr_atomic_buffer_load_i64:
-; GFX12-SDAG-FAKE16:       ; %bb.0: ; %bb
-; GFX12-SDAG-FAKE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-SDAG-FAKE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-SDAG-FAKE16-NEXT:    v_nop
-; GFX12-SDAG-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-SDAG-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-SDAG-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-SDAG-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-SDAG-FAKE16-NEXT:  .LBB5_1: ; %bb1
-; GFX12-SDAG-FAKE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    buffer_load_b64 v[2:3], off, s[0:3], null offset:4 th:TH_LOAD_NT
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
-; GFX12-SDAG-FAKE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-SDAG-FAKE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-SDAG-FAKE16-NEXT:    s_cbranch_execnz .LBB5_1
-; GFX12-SDAG-FAKE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-SDAG-FAKE16-NEXT:    s_endpgm
-;
-; GFX12-GISEL-TRUE16-LABEL: raw_ptr_atomic_buffer_load_i64:
-; GFX12-GISEL-TRUE16:       ; %bb.0: ; %bb
-; GFX12-GISEL-TRUE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-GISEL-TRUE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-GISEL-TRUE16-NEXT:    v_nop
-; GFX12-GISEL-TRUE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-GISEL-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-GISEL-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-GISEL-TRUE16-NEXT:  .LBB5_1: ; %bb1
-; GFX12-GISEL-TRUE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    buffer_load_b64 v[2:3], off, s[0:3], null offset:4 th:TH_LOAD_NT
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
-; GFX12-GISEL-TRUE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-TRUE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-GISEL-TRUE16-NEXT:    s_cbranch_execnz .LBB5_1
-; GFX12-GISEL-TRUE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-GISEL-TRUE16-NEXT:    s_endpgm
-;
-; GFX12-GISEL-FAKE16-LABEL: raw_ptr_atomic_buffer_load_i64:
-; GFX12-GISEL-FAKE16:       ; %bb.0: ; %bb
-; GFX12-GISEL-FAKE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-GISEL-FAKE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-GISEL-FAKE16-NEXT:    v_nop
-; GFX12-GISEL-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-GISEL-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-GISEL-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-GISEL-FAKE16-NEXT:  .LBB5_1: ; %bb1
-; GFX12-GISEL-FAKE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    buffer_load_b64 v[2:3], off, s[0:3], null offset:4 th:TH_LOAD_NT
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
-; GFX12-GISEL-FAKE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-FAKE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-GISEL-FAKE16-NEXT:    s_cbranch_execnz .LBB5_1
-; GFX12-GISEL-FAKE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-GISEL-FAKE16-NEXT:    s_endpgm
+; GFX12-LABEL: raw_ptr_atomic_buffer_load_i64:
+; GFX12:       ; %bb.0: ; %bb
+; GFX12-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX12-NEXT:    s_mov_b64 s[64:65], 0
+; GFX12-NEXT:    v_nop
+; GFX12-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX12-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX12-NEXT:    s_wait_xcnt 0x0
+; GFX12-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX12-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
+; GFX12-NEXT:    s_mov_b32 s4, 0
+; GFX12-NEXT:  .LBB5_1: ; %bb1
+; GFX12-NEXT:    ; =>This Inner Loop Header: Depth=1
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    buffer_load_b64 v[2:3], off, s[0:3], null offset:4 th:TH_LOAD_NT
+; GFX12-NEXT:    s_wait_loadcnt 0x0
+; GFX12-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[2:3], v[0:1]
+; GFX12-NEXT:    s_or_b32 s4, vcc_lo, s4
+; GFX12-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT:    s_cbranch_execnz .LBB5_1
+; GFX12-NEXT:  ; %bb.2: ; %bb2
+; GFX12-NEXT:    s_endpgm
 bb:
   %id = tail call i32 @llvm.amdgcn.workitem.id.x()
   %id.zext = zext i32 %id to i64
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.rcp.bf16.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.rcp.bf16.ll
index f6e68ab3f4ac1..f677f5dfe06ca 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.rcp.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.rcp.bf16.ll
@@ -91,8 +91,10 @@ define amdgpu_kernel void @rcp_bf16_global_load(ptr addrspace(1) %out, ptr addrs
 ; SDAG-TRUE16-NEXT:    v_nop
 ; SDAG-TRUE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; SDAG-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
-; SDAG-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; SDAG-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
+; SDAG-TRUE16-NEXT:    s_wait_xcnt 0x0
+; SDAG-TRUE16-NEXT:    s_mov_b32 s4, 0x3ff
+; SDAG-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; SDAG-TRUE16-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; SDAG-TRUE16-NEXT:    s_wait_kmcnt 0x0
 ; SDAG-TRUE16-NEXT:    global_load_u16 v0, v0, s[2:3] scale_offset
 ; SDAG-TRUE16-NEXT:    s_wait_loadcnt 0x0
@@ -107,8 +109,10 @@ define amdgpu_kernel void @rcp_bf16_global_load(ptr addrspace(1) %out, ptr addrs
 ; SDAG-FAKE16-NEXT:    v_nop
 ; SDAG-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; SDAG-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
-; SDAG-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; SDAG-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; SDAG-FAKE16-NEXT:    s_wait_xcnt 0x0
+; SDAG-FAKE16-NEXT:    s_mov_b32 s4, 0x3ff
+; SDAG-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; SDAG-FAKE16-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; SDAG-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; SDAG-FAKE16-NEXT:    global_load_u16 v0, v0, s[2:3] scale_offset
 ; SDAG-FAKE16-NEXT:    s_wait_loadcnt 0x0
@@ -123,8 +127,10 @@ define amdgpu_kernel void @rcp_bf16_global_load(ptr addrspace(1) %out, ptr addrs
 ; GI-TRUE16-NEXT:    v_nop
 ; GI-TRUE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GI-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
-; GI-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GI-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
+; GI-TRUE16-NEXT:    s_wait_xcnt 0x0
+; GI-TRUE16-NEXT:    s_mov_b32 s4, 0x3ff
+; GI-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GI-TRUE16-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GI-TRUE16-NEXT:    s_wait_kmcnt 0x0
 ; GI-TRUE16-NEXT:    global_load_u16 v0, v0, s[2:3] scale_offset
 ; GI-TRUE16-NEXT:    s_wait_loadcnt 0x0
@@ -139,8 +145,10 @@ define amdgpu_kernel void @rcp_bf16_global_load(ptr addrspace(1) %out, ptr addrs
 ; GI-FAKE16-NEXT:    v_nop
 ; GI-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GI-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
-; GI-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GI-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GI-FAKE16-NEXT:    s_wait_xcnt 0x0
+; GI-FAKE16-NEXT:    s_mov_b32 s4, 0x3ff
+; GI-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GI-FAKE16-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GI-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GI-FAKE16-NEXT:    global_load_u16 v0, v0, s[2:3] scale_offset
 ; GI-FAKE16-NEXT:    s_wait_loadcnt 0x0
@@ -151,8 +159,9 @@ define amdgpu_kernel void @rcp_bf16_global_load(ptr addrspace(1) %out, ptr addrs
 ; GFX13-SDAG-TRUE16-LABEL: rcp_bf16_global_load:
 ; GFX13-SDAG-TRUE16:       ; %bb.0:
 ; GFX13-SDAG-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
-; GFX13-SDAG-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-SDAG-TRUE16-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-SDAG-TRUE16-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-TRUE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-TRUE16-NEXT:    global_load_d16_b16 v0, v0, s[2:3] scale_offset
 ; GFX13-SDAG-TRUE16-NEXT:    s_wait_loadcnt 0x0
@@ -163,8 +172,9 @@ define amdgpu_kernel void @rcp_bf16_global_load(ptr addrspace(1) %out, ptr addrs
 ; GFX13-SDAG-FAKE16-LABEL: rcp_bf16_global_load:
 ; GFX13-SDAG-FAKE16:       ; %bb.0:
 ; GFX13-SDAG-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
-; GFX13-SDAG-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX13-SDAG-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-SDAG-FAKE16-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-SDAG-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-SDAG-FAKE16-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX13-SDAG-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-SDAG-FAKE16-NEXT:    global_load_u16 v0, v0, s[2:3] scale_offset
 ; GFX13-SDAG-FAKE16-NEXT:    s_wait_loadcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.sched.group.barrier.gfx12.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.sched.group.barrier.gfx12.ll
index 38106eb2017a2..e7ddfc73a8d60 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.sched.group.barrier.gfx12.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.sched.group.barrier.gfx12.ll
@@ -110,11 +110,11 @@ define amdgpu_kernel void @test_sched_group_barrier_pipeline_SWMMAC_cluster(ptr
 ; COEXEC-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; COEXEC-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
 ; COEXEC-NEXT:    v_dual_mov_b32 v48, 0 :: v_dual_lshlrev_b32 v0, 4, v0
-; COEXEC-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; COEXEC-NEXT:    v_and_b32_e32 v0, 0x3ff0, v0
+; COEXEC-NEXT:    s_mov_b32 s2, 0x3ff0
 ; COEXEC-NEXT:    s_wait_kmcnt 0x0
-; COEXEC-NEXT:    v_dual_mov_b32 v49, s1 :: v_dual_add_nc_u32 v4, s0, v0
-; COEXEC-NEXT:    v_add_nc_u32_e32 v50, s1, v0
+; COEXEC-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; COEXEC-NEXT:    v_dual_mov_b32 v49, s1 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
+; COEXEC-NEXT:    v_dual_add_nc_u32 v4, s0, v0 :: v_dual_add_nc_u32 v50, s1, v0
 ; COEXEC-NEXT:    ds_load_b128 v[8:11], v4
 ; COEXEC-NEXT:    ds_load_b128 v[12:15], v4 offset:512
 ; COEXEC-NEXT:    ds_load_b128 v[16:19], v4 offset:5120
@@ -329,11 +329,11 @@ define amdgpu_kernel void @test_sched_group_barrier_pipeline_SWMMAC_interleaved(
 ; COEXEC-NEXT:    v_nop
 ; COEXEC-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; COEXEC-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; COEXEC-NEXT:    v_mov_b32_e32 v16, 0
-; COEXEC-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; COEXEC-NEXT:    s_mov_b32 s2, 0x3ff
+; COEXEC-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; COEXEC-NEXT:    v_dual_mov_b32 v16, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; COEXEC-NEXT:    s_wait_kmcnt 0x0
 ; COEXEC-NEXT:    v_mov_b32_e32 v17, s1
-; COEXEC-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; COEXEC-NEXT:    v_lshl_add_u32 v18, v0, 5, s0
 ; COEXEC-NEXT:    v_lshl_add_u32 v19, v0, 4, s1
 ; COEXEC-NEXT:    ds_load_b128 v[8:11], v18 offset:1024
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.atomic.buffer.load.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.atomic.buffer.load.ll
index 7f795e8341165..80da36097a770 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.atomic.buffer.load.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.atomic.buffer.load.ll
@@ -94,9 +94,10 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i32_const_idx(<4 x i32> %ad
 ; GFX12-NEXT:    v_nop
 ; GFX12-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX12-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-NEXT:    v_mov_b32_e32 v1, 15
 ; GFX12-NEXT:    s_wait_xcnt 0x0
+; GFX12-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX12-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT:    v_dual_mov_b32 v1, 15 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX12-NEXT:    s_mov_b32 s4, 0
 ; GFX12-NEXT:  .LBB1_1: ; %bb1
 ; GFX12-NEXT:    ; =>This Inner Loop Header: Depth=1
@@ -326,9 +327,10 @@ define amdgpu_kernel void @struct_nonatomic_buffer_load_i32(<4 x i32> %addr, i32
 ; GFX12-NEXT:    s_clause 0x1
 ; GFX12-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
 ; GFX12-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX12-NEXT:    s_wait_xcnt 0x0
+; GFX12-NEXT:    s_mov_b32 s4, 0x3ff
 ; GFX12-NEXT:    s_wait_kmcnt 0x0
-; GFX12-NEXT:    v_mov_b32_e32 v1, s6
+; GFX12-NEXT:    v_dual_mov_b32 v1, s6 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX12-NEXT:    buffer_load_b32 v1, v1, s[0:3], null idxen offset:4 th:TH_LOAD_NT
 ; GFX12-NEXT:    s_wait_xcnt 0x0
 ; GFX12-NEXT:    s_mov_b32 s0, 0
@@ -376,113 +378,33 @@ define amdgpu_kernel void @struct_atomic_buffer_load_i64(<4 x i32> %addr, i32 %i
 ; GFX11-NEXT:  ; %bb.2: ; %bb2
 ; GFX11-NEXT:    s_endpgm
 ;
-; GFX12-SDAG-TRUE16-LABEL: struct_atomic_buffer_load_i64:
-; GFX12-SDAG-TRUE16:       ; %bb.0: ; %bb
-; GFX12-SDAG-TRUE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-SDAG-TRUE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-SDAG-TRUE16-NEXT:    v_nop
-; GFX12-SDAG-TRUE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-SDAG-TRUE16-NEXT:    s_clause 0x1
-; GFX12-SDAG-TRUE16-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
-; GFX12-SDAG-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-SDAG-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-SDAG-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    v_mov_b32_e32 v2, s6
-; GFX12-SDAG-TRUE16-NEXT:  .LBB6_1: ; %bb1
-; GFX12-SDAG-TRUE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-SDAG-TRUE16-NEXT:    buffer_load_b64 v[4:5], v2, s[0:3], null idxen offset:4 th:TH_LOAD_NT
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
-; GFX12-SDAG-TRUE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-SDAG-TRUE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-SDAG-TRUE16-NEXT:    s_cbranch_execnz .LBB6_1
-; GFX12-SDAG-TRUE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-SDAG-TRUE16-NEXT:    s_endpgm
-;
-; GFX12-SDAG-FAKE16-LABEL: struct_atomic_buffer_load_i64:
-; GFX12-SDAG-FAKE16:       ; %bb.0: ; %bb
-; GFX12-SDAG-FAKE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-SDAG-FAKE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-SDAG-FAKE16-NEXT:    v_nop
-; GFX12-SDAG-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-SDAG-FAKE16-NEXT:    s_clause 0x1
-; GFX12-SDAG-FAKE16-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
-; GFX12-SDAG-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-SDAG-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-SDAG-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    v_mov_b32_e32 v2, s6
-; GFX12-SDAG-FAKE16-NEXT:  .LBB6_1: ; %bb1
-; GFX12-SDAG-FAKE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-SDAG-FAKE16-NEXT:    buffer_load_b64 v[4:5], v2, s[0:3], null idxen offset:4 th:TH_LOAD_NT
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
-; GFX12-SDAG-FAKE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-SDAG-FAKE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-SDAG-FAKE16-NEXT:    s_cbranch_execnz .LBB6_1
-; GFX12-SDAG-FAKE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-SDAG-FAKE16-NEXT:    s_endpgm
-;
-; GFX12-GISEL-TRUE16-LABEL: struct_atomic_buffer_load_i64:
-; GFX12-GISEL-TRUE16:       ; %bb.0: ; %bb
-; GFX12-GISEL-TRUE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-GISEL-TRUE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-GISEL-TRUE16-NEXT:    v_nop
-; GFX12-GISEL-TRUE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-GISEL-TRUE16-NEXT:    s_clause 0x1
-; GFX12-GISEL-TRUE16-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
-; GFX12-GISEL-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-GISEL-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    v_mov_b32_e32 v2, s6
-; GFX12-GISEL-TRUE16-NEXT:  .LBB6_1: ; %bb1
-; GFX12-GISEL-TRUE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-GISEL-TRUE16-NEXT:    buffer_load_b64 v[4:5], v2, s[0:3], null idxen offset:4 th:TH_LOAD_NT
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
-; GFX12-GISEL-TRUE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-TRUE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-GISEL-TRUE16-NEXT:    s_cbranch_execnz .LBB6_1
-; GFX12-GISEL-TRUE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-GISEL-TRUE16-NEXT:    s_endpgm
-;
-; GFX12-GISEL-FAKE16-LABEL: struct_atomic_buffer_load_i64:
-; GFX12-GISEL-FAKE16:       ; %bb.0: ; %bb
-; GFX12-GISEL-FAKE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-GISEL-FAKE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-GISEL-FAKE16-NEXT:    v_nop
-; GFX12-GISEL-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-GISEL-FAKE16-NEXT:    s_clause 0x1
-; GFX12-GISEL-FAKE16-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
-; GFX12-GISEL-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-GISEL-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    v_mov_b32_e32 v2, s6
-; GFX12-GISEL-FAKE16-NEXT:  .LBB6_1: ; %bb1
-; GFX12-GISEL-FAKE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-GISEL-FAKE16-NEXT:    buffer_load_b64 v[4:5], v2, s[0:3], null idxen offset:4 th:TH_LOAD_NT
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
-; GFX12-GISEL-FAKE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-FAKE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-GISEL-FAKE16-NEXT:    s_cbranch_execnz .LBB6_1
-; GFX12-GISEL-FAKE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-GISEL-FAKE16-NEXT:    s_endpgm
+; GFX12-LABEL: struct_atomic_buffer_load_i64:
+; GFX12:       ; %bb.0: ; %bb
+; GFX12-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX12-NEXT:    s_mov_b64 s[64:65], 0
+; GFX12-NEXT:    v_nop
+; GFX12-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX12-NEXT:    s_clause 0x1
+; GFX12-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
+; GFX12-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX12-NEXT:    s_wait_xcnt 0x0
+; GFX12-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX12-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
+; GFX12-NEXT:    s_mov_b32 s4, 0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    v_mov_b32_e32 v2, s6
+; GFX12-NEXT:  .LBB6_1: ; %bb1
+; GFX12-NEXT:    ; =>This Inner Loop Header: Depth=1
+; GFX12-NEXT:    buffer_load_b64 v[4:5], v2, s[0:3], null idxen offset:4 th:TH_LOAD_NT
+; GFX12-NEXT:    s_wait_loadcnt 0x0
+; GFX12-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
+; GFX12-NEXT:    s_or_b32 s4, vcc_lo, s4
+; GFX12-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT:    s_cbranch_execnz .LBB6_1
+; GFX12-NEXT:  ; %bb.2: ; %bb2
+; GFX12-NEXT:    s_endpgm
 bb:
   %id = tail call i32 @llvm.amdgcn.workitem.id.x()
   %id.zext = zext i32 %id to i64
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.ptr.atomic.buffer.load.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.ptr.atomic.buffer.load.ll
index f29842a69ea11..3934b5d1aee3f 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.ptr.atomic.buffer.load.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.struct.ptr.atomic.buffer.load.ll
@@ -94,9 +94,10 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i32_const_idx(ptr addrs
 ; GFX12-NEXT:    v_nop
 ; GFX12-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX12-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-NEXT:    v_mov_b32_e32 v1, 15
 ; GFX12-NEXT:    s_wait_xcnt 0x0
+; GFX12-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX12-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT:    v_dual_mov_b32 v1, 15 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX12-NEXT:    s_mov_b32 s4, 0
 ; GFX12-NEXT:  .LBB1_1: ; %bb1
 ; GFX12-NEXT:    ; =>This Inner Loop Header: Depth=1
@@ -326,9 +327,10 @@ define amdgpu_kernel void @struct_ptr_nonatomic_buffer_load_i32(ptr addrspace(8)
 ; GFX12-NEXT:    s_clause 0x1
 ; GFX12-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
 ; GFX12-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX12-NEXT:    s_wait_xcnt 0x0
+; GFX12-NEXT:    s_mov_b32 s4, 0x3ff
 ; GFX12-NEXT:    s_wait_kmcnt 0x0
-; GFX12-NEXT:    v_mov_b32_e32 v1, s6
+; GFX12-NEXT:    v_dual_mov_b32 v1, s6 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
 ; GFX12-NEXT:    buffer_load_b32 v1, v1, s[0:3], null idxen offset:4 th:TH_LOAD_NT
 ; GFX12-NEXT:    s_wait_xcnt 0x0
 ; GFX12-NEXT:    s_mov_b32 s0, 0
@@ -376,113 +378,33 @@ define amdgpu_kernel void @struct_ptr_atomic_buffer_load_i64(ptr addrspace(8) %p
 ; GFX11-NEXT:  ; %bb.2: ; %bb2
 ; GFX11-NEXT:    s_endpgm
 ;
-; GFX12-SDAG-TRUE16-LABEL: struct_ptr_atomic_buffer_load_i64:
-; GFX12-SDAG-TRUE16:       ; %bb.0: ; %bb
-; GFX12-SDAG-TRUE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-SDAG-TRUE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-SDAG-TRUE16-NEXT:    v_nop
-; GFX12-SDAG-TRUE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-SDAG-TRUE16-NEXT:    s_clause 0x1
-; GFX12-SDAG-TRUE16-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
-; GFX12-SDAG-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-SDAG-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-SDAG-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    v_mov_b32_e32 v2, s6
-; GFX12-SDAG-TRUE16-NEXT:  .LBB6_1: ; %bb1
-; GFX12-SDAG-TRUE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-SDAG-TRUE16-NEXT:    buffer_load_b64 v[4:5], v2, s[0:3], null idxen offset:4 th:TH_LOAD_NT
-; GFX12-SDAG-TRUE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-SDAG-TRUE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
-; GFX12-SDAG-TRUE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-SDAG-TRUE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-SDAG-TRUE16-NEXT:    s_cbranch_execnz .LBB6_1
-; GFX12-SDAG-TRUE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-SDAG-TRUE16-NEXT:    s_endpgm
-;
-; GFX12-SDAG-FAKE16-LABEL: struct_ptr_atomic_buffer_load_i64:
-; GFX12-SDAG-FAKE16:       ; %bb.0: ; %bb
-; GFX12-SDAG-FAKE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-SDAG-FAKE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-SDAG-FAKE16-NEXT:    v_nop
-; GFX12-SDAG-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-SDAG-FAKE16-NEXT:    s_clause 0x1
-; GFX12-SDAG-FAKE16-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
-; GFX12-SDAG-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-SDAG-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-SDAG-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    v_mov_b32_e32 v2, s6
-; GFX12-SDAG-FAKE16-NEXT:  .LBB6_1: ; %bb1
-; GFX12-SDAG-FAKE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-SDAG-FAKE16-NEXT:    buffer_load_b64 v[4:5], v2, s[0:3], null idxen offset:4 th:TH_LOAD_NT
-; GFX12-SDAG-FAKE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-SDAG-FAKE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
-; GFX12-SDAG-FAKE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-SDAG-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-SDAG-FAKE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-SDAG-FAKE16-NEXT:    s_cbranch_execnz .LBB6_1
-; GFX12-SDAG-FAKE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-SDAG-FAKE16-NEXT:    s_endpgm
-;
-; GFX12-GISEL-TRUE16-LABEL: struct_ptr_atomic_buffer_load_i64:
-; GFX12-GISEL-TRUE16:       ; %bb.0: ; %bb
-; GFX12-GISEL-TRUE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-GISEL-TRUE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-GISEL-TRUE16-NEXT:    v_nop
-; GFX12-GISEL-TRUE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-GISEL-TRUE16-NEXT:    s_clause 0x1
-; GFX12-GISEL-TRUE16-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
-; GFX12-GISEL-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-GISEL-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    v_mov_b32_e32 v2, s6
-; GFX12-GISEL-TRUE16-NEXT:  .LBB6_1: ; %bb1
-; GFX12-GISEL-TRUE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-GISEL-TRUE16-NEXT:    buffer_load_b64 v[4:5], v2, s[0:3], null idxen offset:4 th:TH_LOAD_NT
-; GFX12-GISEL-TRUE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-GISEL-TRUE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
-; GFX12-GISEL-TRUE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-TRUE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-GISEL-TRUE16-NEXT:    s_cbranch_execnz .LBB6_1
-; GFX12-GISEL-TRUE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-GISEL-TRUE16-NEXT:    s_endpgm
-;
-; GFX12-GISEL-FAKE16-LABEL: struct_ptr_atomic_buffer_load_i64:
-; GFX12-GISEL-FAKE16:       ; %bb.0: ; %bb
-; GFX12-GISEL-FAKE16-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
-; GFX12-GISEL-FAKE16-NEXT:    s_mov_b64 s[64:65], 0
-; GFX12-GISEL-FAKE16-NEXT:    v_nop
-; GFX12-GISEL-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX12-GISEL-FAKE16-NEXT:    s_clause 0x1
-; GFX12-GISEL-FAKE16-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
-; GFX12-GISEL-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
-; GFX12-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX12-GISEL-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_xcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    s_mov_b32 s4, 0
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_kmcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    v_mov_b32_e32 v2, s6
-; GFX12-GISEL-FAKE16-NEXT:  .LBB6_1: ; %bb1
-; GFX12-GISEL-FAKE16-NEXT:    ; =>This Inner Loop Header: Depth=1
-; GFX12-GISEL-FAKE16-NEXT:    buffer_load_b64 v[4:5], v2, s[0:3], null idxen offset:4 th:TH_LOAD_NT
-; GFX12-GISEL-FAKE16-NEXT:    s_wait_loadcnt 0x0
-; GFX12-GISEL-FAKE16-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
-; GFX12-GISEL-FAKE16-NEXT:    s_or_b32 s4, vcc_lo, s4
-; GFX12-GISEL-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX12-GISEL-FAKE16-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
-; GFX12-GISEL-FAKE16-NEXT:    s_cbranch_execnz .LBB6_1
-; GFX12-GISEL-FAKE16-NEXT:  ; %bb.2: ; %bb2
-; GFX12-GISEL-FAKE16-NEXT:    s_endpgm
+; GFX12-LABEL: struct_ptr_atomic_buffer_load_i64:
+; GFX12:       ; %bb.0: ; %bb
+; GFX12-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX12-NEXT:    s_mov_b64 s[64:65], 0
+; GFX12-NEXT:    v_nop
+; GFX12-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX12-NEXT:    s_clause 0x1
+; GFX12-NEXT:    s_load_b32 s6, s[4:5], 0x34 nv
+; GFX12-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
+; GFX12-NEXT:    s_wait_xcnt 0x0
+; GFX12-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX12-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s4, v0 bitop3:0x40
+; GFX12-NEXT:    s_mov_b32 s4, 0
+; GFX12-NEXT:    s_wait_kmcnt 0x0
+; GFX12-NEXT:    v_mov_b32_e32 v2, s6
+; GFX12-NEXT:  .LBB6_1: ; %bb1
+; GFX12-NEXT:    ; =>This Inner Loop Header: Depth=1
+; GFX12-NEXT:    buffer_load_b64 v[4:5], v2, s[0:3], null idxen offset:4 th:TH_LOAD_NT
+; GFX12-NEXT:    s_wait_loadcnt 0x0
+; GFX12-NEXT:    v_cmp_ne_u64_e32 vcc_lo, v[4:5], v[0:1]
+; GFX12-NEXT:    s_or_b32 s4, vcc_lo, s4
+; GFX12-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX12-NEXT:    s_and_not1_b32 exec_lo, exec_lo, s4
+; GFX12-NEXT:    s_cbranch_execnz .LBB6_1
+; GFX12-NEXT:  ; %bb.2: ; %bb2
+; GFX12-NEXT:    s_endpgm
 bb:
   %id = tail call i32 @llvm.amdgcn.workitem.id.x()
   %id.zext = zext i32 %id to i64
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.sqrt.bf16.ll b/llvm/test/CodeGen/AMDGPU/llvm.sqrt.bf16.ll
index dc4e360bb41cb..30d966eb1b7c2 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.sqrt.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.sqrt.bf16.ll
@@ -162,12 +162,13 @@ define amdgpu_kernel void @sqrt_v2bf16(ptr addrspace(1) %r, ptr addrspace(1) %a)
 ; GFX12-FAKE16-SDAG-NEXT:    s_mov_b32 s5, s1
 ; GFX12-FAKE16-SDAG-NEXT:    s_wait_loadcnt 0x0
 ; GFX12-FAKE16-SDAG-NEXT:    v_sqrt_bf16_e32 v1, v0
+; GFX12-FAKE16-SDAG-NEXT:    s_mov_b32 s0, 0xffff
 ; GFX12-FAKE16-SDAG-NEXT:    v_nop
-; GFX12-FAKE16-SDAG-NEXT:    v_lshrrev_b32_e32 v0, 16, v0
-; GFX12-FAKE16-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(TRANS32_DEP_2)
+; GFX12-FAKE16-SDAG-NEXT:    s_delay_alu instid0(TRANS32_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX12-FAKE16-SDAG-NEXT:    v_dual_lshrrev_b32 v0, 16, v0 :: v_dual_bitop2_b32 v1, s0, v1 bitop3:0x40
 ; GFX12-FAKE16-SDAG-NEXT:    v_sqrt_bf16_e32 v0, v0
-; GFX12-FAKE16-SDAG-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX12-FAKE16-SDAG-NEXT:    s_delay_alu instid0(TRANS32_DEP_1) | instid1(VALU_DEP_1)
+; GFX12-FAKE16-SDAG-NEXT:    v_nop
+; GFX12-FAKE16-SDAG-NEXT:    s_delay_alu instid0(TRANS32_DEP_1)
 ; GFX12-FAKE16-SDAG-NEXT:    v_lshl_or_b32 v0, v0, 16, v1
 ; GFX12-FAKE16-SDAG-NEXT:    buffer_store_b32 v0, off, s[4:7], null
 ; GFX12-FAKE16-SDAG-NEXT:    s_endpgm
diff --git a/llvm/test/CodeGen/AMDGPU/load-constant-i1.ll b/llvm/test/CodeGen/AMDGPU/load-constant-i1.ll
index 70d0d38eb656a..34910b6f180e7 100644
--- a/llvm/test/CodeGen/AMDGPU/load-constant-i1.ll
+++ b/llvm/test/CodeGen/AMDGPU/load-constant-i1.ll
@@ -1898,18 +1898,20 @@ define amdgpu_kernel void @constant_zextload_v8i1_to_v8i32(ptr addrspace(1) %out
 ; GFX1250-FAKE16-NEXT:    global_load_u8 v0, v8, s[2:3] nv
 ; GFX1250-FAKE16-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-FAKE16-NEXT:    v_readfirstlane_b32 s2, v0
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v0, 0xffff, v0
+; GFX1250-FAKE16-NEXT:    s_bfe_u32 s5, s2, 0x10005
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s3, 0xffff
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v1, s5 :: v_dual_bitop2_b32 v0, s3, v0 bitop3:0x40
 ; GFX1250-FAKE16-NEXT:    s_bfe_u32 s3, s2, 0x10003
 ; GFX1250-FAKE16-NEXT:    s_bfe_u32 s4, s2, 0x10001
-; GFX1250-FAKE16-NEXT:    s_bfe_u32 s5, s2, 0x10005
 ; GFX1250-FAKE16-NEXT:    s_and_b32 s6, s2, 1
 ; GFX1250-FAKE16-NEXT:    s_bfe_u32 s7, s2, 0x10002
 ; GFX1250-FAKE16-NEXT:    s_bfe_u32 s2, s2, 0x10004
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b32_e32 v3, 7, v0
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v2, v0, 6, 1
-; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s5
-; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v4, s6 :: v_dual_mov_b32 v5, s4
-; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v6, s7 :: v_dual_mov_b32 v7, s3
+; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v4, s6
+; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v5, s4 :: v_dual_mov_b32 v6, s7
+; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v7, s3
 ; GFX1250-FAKE16-NEXT:    s_clause 0x1
 ; GFX1250-FAKE16-NEXT:    global_store_b128 v8, v[0:3], s[0:1] offset:16
 ; GFX1250-FAKE16-NEXT:    global_store_b128 v8, v[4:7], s[0:1]
@@ -5928,17 +5930,19 @@ define amdgpu_kernel void @constant_zextload_v3i1_to_v3i64(ptr addrspace(1) %out
 ; GFX1250-FAKE16-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x24 nv
 ; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v5, 0
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v3, v5
 ; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-FAKE16-NEXT:    global_load_u8 v0, v5, s[2:3] nv
-; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v3, v5
 ; GFX1250-FAKE16-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v1, 0xffff, v0
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v6, v0, 1, 1
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250-FAKE16-NEXT:    v_dual_lshrrev_b32 v2, 2, v1 :: v_dual_bitop2_b32 v0, 1, v0 bitop3:0x40
-; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v1, v5
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v4, 0xffff, v2
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_4)
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s2, 0xffff
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v1, v5 :: v_dual_bitop2_b32 v4, s2, v2 bitop3:0x40
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_3)
 ; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v2, 0xffff, v6
 ; GFX1250-FAKE16-NEXT:    s_clause 0x1
 ; GFX1250-FAKE16-NEXT:    global_store_b64 v5, v[4:5], s[0:1] offset:16
@@ -6315,15 +6319,15 @@ define amdgpu_kernel void @constant_zextload_v4i1_to_v4i64(ptr addrspace(1) %out
 ; GFX1250-FAKE16-NEXT:    v_readfirstlane_b32 s2, v0
 ; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v0, 0xffff, v0
 ; GFX1250-FAKE16-NEXT:    s_bfe_u32 s3, s2, 0x10002
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
-; GFX1250-FAKE16-NEXT:    v_lshrrev_b32_e32 v2, 3, v0
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
 ; GFX1250-FAKE16-NEXT:    s_and_b32 s3, 0xffff, s3
-; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v0, s3
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshrrev_b32 v2, 3, v0 :: v_dual_mov_b32 v0, s3
 ; GFX1250-FAKE16-NEXT:    s_bfe_u32 s3, s2, 0x10001
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_2)
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v2, 0xffff, v2
 ; GFX1250-FAKE16-NEXT:    s_and_b32 s2, s2, 1
 ; GFX1250-FAKE16-NEXT:    s_and_b32 s3, 0xffff, s3
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v2, 0xffff, v2
 ; GFX1250-FAKE16-NEXT:    global_store_b128 v1, v[0:3], s[0:1] offset:16
 ; GFX1250-FAKE16-NEXT:    s_wait_xcnt 0x0
 ; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v2, s3
@@ -6741,15 +6745,16 @@ define amdgpu_kernel void @constant_zextload_v8i1_to_v8i64(ptr addrspace(1) %out
 ; GFX1250-TRUE16-NEXT:    v_dual_mov_b32 v11, v1 :: v_dual_mov_b32 v13, v1
 ; GFX1250-TRUE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-TRUE16-NEXT:    global_load_u8 v8, v1, s[2:3] nv
+; GFX1250-TRUE16-NEXT:    s_wait_xcnt 0x0
+; GFX1250-TRUE16-NEXT:    s_mov_b32 s2, 0xffff
 ; GFX1250-TRUE16-NEXT:    s_wait_loadcnt 0x0
-; GFX1250-TRUE16-NEXT:    v_and_b32_e32 v0, 0xffff, v8
+; GFX1250-TRUE16-NEXT:    v_dual_mov_b32 v15, v1 :: v_dual_bitop2_b32 v0, s2, v8 bitop3:0x40
 ; GFX1250-TRUE16-NEXT:    v_mov_b16_e32 v12.l, v8.l
-; GFX1250-TRUE16-NEXT:    v_mov_b32_e32 v15, v1
 ; GFX1250-TRUE16-NEXT:    v_bfe_u32 v6, v8, 5, 1
 ; GFX1250-TRUE16-NEXT:    v_bfe_u32 v4, v8, 4, 1
+; GFX1250-TRUE16-NEXT:    v_bfe_u32 v10, v8, 3, 1
 ; GFX1250-TRUE16-NEXT:    v_lshrrev_b32_e32 v2, 7, v0
 ; GFX1250-TRUE16-NEXT:    v_bfe_u32 v0, v0, 6, 1
-; GFX1250-TRUE16-NEXT:    v_bfe_u32 v10, v8, 3, 1
 ; GFX1250-TRUE16-NEXT:    v_bfe_u32 v14, v8, 1, 1
 ; GFX1250-TRUE16-NEXT:    v_bfe_u32 v8, v8, 2, 1
 ; GFX1250-TRUE16-NEXT:    v_and_b32_e32 v12, 1, v12
@@ -6772,17 +6777,17 @@ define amdgpu_kernel void @constant_zextload_v8i1_to_v8i64(ptr addrspace(1) %out
 ; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v3, v1 :: v_dual_mov_b32 v5, v1
 ; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v7, v1 :: v_dual_mov_b32 v9, v1
 ; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v11, v1 :: v_dual_mov_b32 v13, v1
+; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v15, v1
 ; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-FAKE16-NEXT:    global_load_u8 v12, v1, s[2:3] nv
 ; GFX1250-FAKE16-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v0, 0xffff, v12
-; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v15, v1
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v6, v12, 5, 1
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v4, v12, 4, 1
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v10, v12, 3, 1
+; GFX1250-FAKE16-NEXT:    v_bfe_u32 v8, v12, 2, 1
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b32_e32 v2, 7, v0
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v0, v0, 6, 1
-; GFX1250-FAKE16-NEXT:    v_bfe_u32 v8, v12, 2, 1
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v14, v12, 1, 1
 ; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v12, 1, v12
 ; GFX1250-FAKE16-NEXT:    s_clause 0x3
@@ -7503,17 +7508,19 @@ define amdgpu_kernel void @constant_zextload_v16i1_to_v16i64(ptr addrspace(1) %o
 ; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v27, v1 :: v_dual_mov_b32 v29, v1
 ; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-FAKE16-NEXT:    global_load_u16 v12, v1, s[2:3] nv
+; GFX1250-FAKE16-NEXT:    s_wait_xcnt 0x0
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s2, 0xffff
 ; GFX1250-FAKE16-NEXT:    s_wait_loadcnt 0x0
-; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v31, v1 :: v_dual_bitop2_b32 v28, 1, v12 bitop3:0x40
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v22, 0xffff, v12
+; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v31, v1 :: v_dual_bitop2_b32 v22, s2, v12 bitop3:0x40
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v0, v12, 10, 1
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v6, v12, 9, 1
+; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v28, 1, v12
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v14, v12, 13, 1
-; GFX1250-FAKE16-NEXT:    v_bfe_u32 v18, v12, 7, 1
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v2, v22, 11, 1
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v4, v22, 8, 1
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b32_e32 v10, 15, v22
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v8, v22, 14, 1
+; GFX1250-FAKE16-NEXT:    v_bfe_u32 v18, v12, 7, 1
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v26, v12, 3, 1
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v30, v12, 1, 1
 ; GFX1250-FAKE16-NEXT:    v_bfe_u32 v24, v12, 2, 1
diff --git a/llvm/test/CodeGen/AMDGPU/loop-prefetch-data.ll b/llvm/test/CodeGen/AMDGPU/loop-prefetch-data.ll
index c345990cfc115..018adb44df059 100644
--- a/llvm/test/CodeGen/AMDGPU/loop-prefetch-data.ll
+++ b/llvm/test/CodeGen/AMDGPU/loop-prefetch-data.ll
@@ -672,15 +672,15 @@ define amdgpu_kernel void @copy_flat_divergent(ptr nocapture %d, ptr nocapture r
 ; GFX1250-NEXT:    s_cbranch_scc1 .LBB4_3
 ; GFX1250-NEXT:  ; %bb.1: ; %for.body.preheader
 ; GFX1250-NEXT:    s_load_b128 s[8:11], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
+; GFX1250-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s0, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_mov_b64 s[0:1], 0xffffffffffffff50
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 4, v0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[2:3], s[10:11], v[0:1]
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[0:1], s[8:9], v[0:1]
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2)
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[2:3], 0xb0, v[2:3]
 ; GFX1250-NEXT:  .LBB4_2: ; %for.body
 ; GFX1250-NEXT:    ; =>This Inner Loop Header: Depth=1
diff --git a/llvm/test/CodeGen/AMDGPU/mad-mix-bf16.ll b/llvm/test/CodeGen/AMDGPU/mad-mix-bf16.ll
index 5013ea2ff2ccd..11e100bbb97ef 100644
--- a/llvm/test/CodeGen/AMDGPU/mad-mix-bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/mad-mix-bf16.ll
@@ -61,12 +61,14 @@ define <2 x float> @v_mad_mix_v2f32(<2 x bfloat> %src0, <2 x bfloat> %src1, <2 x
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_lshlrev_b32 v6, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v1
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_bitop2_b32 v5, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v1 :: v_dual_bitop2_b32 v7, s0, v1 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
 ; GFX1250-NEXT:    v_pk_fma_f32 v[0:1], v[4:5], v[6:7], v[0:1]
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %src0.ext = fpext <2 x bfloat> %src0 to <2 x float>
@@ -81,9 +83,11 @@ define <2 x float> @v_mad_mix_v2f32_shuffle(<2 x bfloat> %src0, <2 x bfloat> %sr
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v5, 16, v0 :: v_dual_lshlrev_b32 v6, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v4, 0xffff0000, v0
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v1
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v5, 16, v0 :: v_dual_bitop2_b32 v4, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v1 :: v_dual_bitop2_b32 v7, s0, v1 bitop3:0x40
 ; GFX1250-NEXT:    v_and_b32_e32 v0, 0xffff0000, v2
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_pk_fma_f32 v[0:1], v[4:5], v[6:7], v[0:1] op_sel_hi:[1,1,0]
@@ -267,9 +271,11 @@ define <2 x float> @v_mad_mix_v2f32_f32imm1(<2 x bfloat> %src0, <2 x bfloat> %sr
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_lshlrev_b32 v4, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v1
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v3, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v1 :: v_dual_bitop2_b32 v5, s0, v1 bitop3:0x40
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_pk_fma_f32 v[0:1], v[2:3], v[4:5], 1.0 op_sel_hi:[1,1,0]
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
@@ -284,9 +290,11 @@ define <2 x float> @v_mad_mix_v2f32_cvtbf16imminv2pi(<2 x bfloat> %src0, <2 x bf
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_lshlrev_b32 v4, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v1
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v3, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v1 :: v_dual_bitop2_b32 v5, s0, v1 bitop3:0x40
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_pk_fma_f32 v[0:1], v[2:3], v[4:5], 0x3e230000 op_sel_hi:[1,1,0]
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
@@ -302,9 +310,11 @@ define <2 x float> @v_mad_mix_v2f32_f32imminv2pi(<2 x bfloat> %src0, <2 x bfloat
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_lshlrev_b32 v4, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v1
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v3, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v1 :: v_dual_bitop2_b32 v5, s0, v1 bitop3:0x40
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_pk_fma_f32 v[0:1], v[2:3], v[4:5], 0.15915494 op_sel_hi:[1,1,0]
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
@@ -679,9 +689,11 @@ define <2 x float> @v_mad_mix_v2f32_cvt_add(<2 x bfloat> %src0, <2 x bfloat> %sr
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_lshlrev_b32 v4, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v1
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v3, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v1 :: v_dual_bitop2_b32 v5, s0, v1 bitop3:0x40
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_pk_add_f32 v[0:1], v[2:3], v[4:5]
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
@@ -696,9 +708,11 @@ define <2 x float> @v_mad_mix_v2f32_shuffle_cvt_add(<2 x bfloat> %src0, <2 x bfl
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v3, 16, v0 :: v_dual_lshlrev_b32 v4, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v0
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v1
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v3, 16, v0 :: v_dual_bitop2_b32 v2, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v1 :: v_dual_bitop2_b32 v5, s0, v1 bitop3:0x40
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_pk_add_f32 v[0:1], v[2:3], v[4:5]
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
@@ -1041,9 +1055,11 @@ define <2 x float> @v_mad_mix_v2f32_cvt_mul(<2 x bfloat> %src0, <2 x bfloat> %sr
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_lshlrev_b32 v4, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v1
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v0 :: v_dual_bitop2_b32 v3, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v1 :: v_dual_bitop2_b32 v5, s0, v1 bitop3:0x40
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_pk_mul_f32 v[0:1], v[2:3], v[4:5]
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
@@ -1058,9 +1074,11 @@ define <2 x float> @v_mad_mix_v2f32_shuffle_cvt_mul(<2 x bfloat> %src0, <2 x bfl
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v3, 16, v0 :: v_dual_lshlrev_b32 v4, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v0
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v1
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v3, 16, v0 :: v_dual_bitop2_b32 v2, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v1 :: v_dual_bitop2_b32 v5, s0, v1 bitop3:0x40
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_pk_mul_f32 v[0:1], v[2:3], v[4:5]
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/mad-mix-lo-bf16.ll b/llvm/test/CodeGen/AMDGPU/mad-mix-lo-bf16.ll
index c15cf022b3177..690cf79704e4d 100644
--- a/llvm/test/CodeGen/AMDGPU/mad-mix-lo-bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/mad-mix-lo-bf16.ll
@@ -109,13 +109,16 @@ define <2 x bfloat> @v_mad_mix_v2f32(<2 x bfloat> %src0, <2 x bfloat> %src1, <2
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_lshlrev_b32 v6, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v1
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_bitop2_b32 v5, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v1 :: v_dual_bitop2_b32 v7, s0, v1 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
 ; GFX1250-NEXT:    v_pk_fma_f32 v[0:1], v[4:5], v[6:7], v[0:1]
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v1
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %src0.ext = fpext <2 x bfloat> %src0 to <2 x float>
@@ -161,17 +164,21 @@ define <4 x bfloat> @v_mad_mix_v4f32(<4 x bfloat> %src0, <4 x bfloat> %src1, <4
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v1
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v6, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v0
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
-; GFX1250-NEXT:    v_and_b32_e32 v9, 0xffff0000, v3
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v8, 16, v3
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v2
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
-; GFX1250-NEXT:    v_and_b32_e32 v11, 0xffff0000, v5
-; GFX1250-NEXT:    v_and_b32_e32 v13, 0xffff0000, v4
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v12, 16, v4 :: v_dual_lshlrev_b32 v10, 16, v5
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v1 :: v_dual_bitop2_b32 v7, s0, v1 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v3 :: v_dual_bitop2_b32 v1, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v0 :: v_dual_bitop2_b32 v9, s0, v3 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v12, 16, v4 :: v_dual_bitop2_b32 v3, s0, v2 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v2 :: v_dual_bitop2_b32 v11, s0, v5 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v10, 16, v5 :: v_dual_bitop2_b32 v13, s0, v4 bitop3:0x40
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_pk_fma_f32 v[0:1], v[0:1], v[2:3], v[12:13]
 ; GFX1250-NEXT:    v_pk_fma_f32 v[2:3], v[6:7], v[8:9], v[10:11]
@@ -192,13 +199,16 @@ define <2 x bfloat> @v_mad_mix_v2f32_clamp_postcvt(<2 x bfloat> %src0, <2 x bflo
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_lshlrev_b32 v6, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v1
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_bitop2_b32 v5, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v1 :: v_dual_bitop2_b32 v7, s0, v1 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
 ; GFX1250-NEXT:    v_pk_fma_f32 v[0:1], v[4:5], v[6:7], v[0:1]
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v1 clamp
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %src0.ext = fpext <2 x bfloat> %src0 to <2 x float>
@@ -251,15 +261,21 @@ define <4 x bfloat> @v_mad_mix_v4f32_clamp_postcvt(<4 x bfloat> %src0, <4 x bflo
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v0 :: v_dual_lshlrev_b32 v8, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v9, 0xffff0000, v1
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_lshlrev_b32 v10, 16, v3
-; GFX1250-NEXT:    v_and_b32_e32 v11, 0xffff0000, v3
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v4
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v4 :: v_dual_lshlrev_b32 v12, 16, v5
-; GFX1250-NEXT:    v_and_b32_e32 v13, 0xffff0000, v5
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v0 :: v_dual_bitop2_b32 v7, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v1 :: v_dual_bitop2_b32 v9, s0, v1 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v10, 16, v3 :: v_dual_bitop2_b32 v11, s0, v3 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v4 :: v_dual_bitop2_b32 v3, s0, v4 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v12, 16, v5 :: v_dual_bitop2_b32 v13, s0, v5 bitop3:0x40
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_pk_fma_f32 v[0:1], v[6:7], v[0:1], v[2:3]
 ; GFX1250-NEXT:    v_pk_fma_f32 v[2:3], v[8:9], v[10:11], v[12:13]
@@ -282,16 +298,19 @@ define <2 x bfloat> @v_mad_mix_v2f32_clamp_postcvt_lo(<2 x bfloat> %src0, <2 x b
 ; GFX1250-FAKE16:       ; %bb.0:
 ; GFX1250-FAKE16-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v5, 0xffff0000, v0
-; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_lshlrev_b32 v6, 16, v1
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v7, 0xffff0000, v1
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-FAKE16-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_bitop2_b32 v5, s0, v0 bitop3:0x40
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v6, 16, v1 :: v_dual_bitop2_b32 v7, s0, v1 bitop3:0x40
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
 ; GFX1250-FAKE16-NEXT:    v_pk_fma_f32 v[0:1], v[4:5], v[6:7], v[0:1]
-; GFX1250-FAKE16-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v1
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v1
 ; GFX1250-FAKE16-NEXT:    v_pk_max_num_bf16 v1, v0, v0 clamp
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-FAKE16-NEXT:    v_bfi_b32 v0, 0xffff, v1, v0
 ; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
 ;
@@ -299,15 +318,17 @@ define <2 x bfloat> @v_mad_mix_v2f32_clamp_postcvt_lo(<2 x bfloat> %src0, <2 x b
 ; GFX1250-REAL16:       ; %bb.0:
 ; GFX1250-REAL16-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-REAL16-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-REAL16-NEXT:    v_and_b32_e32 v5, 0xffff0000, v0
-; GFX1250-REAL16-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_lshlrev_b32 v6, 16, v1
-; GFX1250-REAL16-NEXT:    v_and_b32_e32 v7, 0xffff0000, v1
-; GFX1250-REAL16-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-REAL16-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
-; GFX1250-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-REAL16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-REAL16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-REAL16-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_bitop2_b32 v5, s0, v0 bitop3:0x40
+; GFX1250-REAL16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-REAL16-NEXT:    v_dual_lshlrev_b32 v6, 16, v1 :: v_dual_bitop2_b32 v7, s0, v1 bitop3:0x40
+; GFX1250-REAL16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-REAL16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-REAL16-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
 ; GFX1250-REAL16-NEXT:    v_pk_fma_f32 v[0:1], v[4:5], v[6:7], v[0:1]
+; GFX1250-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-REAL16-NEXT:    v_cvt_pk_bf16_f32 v1, v0, v1
-; GFX1250-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-REAL16-NEXT:    v_pk_max_num_bf16 v0, v1, v1 op_sel_hi:[0,0] clamp
 ; GFX1250-REAL16-NEXT:    v_mov_b16_e32 v0.h, v1.h
 ; GFX1250-REAL16-NEXT:    s_set_pc_i64 s[30:31]
@@ -328,18 +349,20 @@ define <2 x bfloat> @v_mad_mix_v2f32_clamp_postcvt_hi(<2 x bfloat> %src0, <2 x b
 ; GFX1250-FAKE16:       ; %bb.0:
 ; GFX1250-FAKE16-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v5, 0xffff0000, v0
-; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_lshlrev_b32 v6, 16, v1
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v7, 0xffff0000, v1
-; GFX1250-FAKE16-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-FAKE16-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_bitop2_b32 v5, s0, v0 bitop3:0x40
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v6, 16, v1 :: v_dual_bitop2_b32 v7, s0, v1 bitop3:0x40
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
 ; GFX1250-FAKE16-NEXT:    v_pk_fma_f32 v[0:1], v[4:5], v[6:7], v[0:1]
-; GFX1250-FAKE16-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v1
 ; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v1
 ; GFX1250-FAKE16-NEXT:    v_lshrrev_b32_e32 v1, 16, v0
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-FAKE16-NEXT:    v_pk_max_num_bf16 v1, v1, v1 clamp
-; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-FAKE16-NEXT:    v_perm_b32 v0, v1, v0, 0x5040100
 ; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
 ;
@@ -347,18 +370,20 @@ define <2 x bfloat> @v_mad_mix_v2f32_clamp_postcvt_hi(<2 x bfloat> %src0, <2 x b
 ; GFX1250-REAL16:       ; %bb.0:
 ; GFX1250-REAL16-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-REAL16-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-REAL16-NEXT:    v_and_b32_e32 v5, 0xffff0000, v0
-; GFX1250-REAL16-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_lshlrev_b32 v6, 16, v1
-; GFX1250-REAL16-NEXT:    v_and_b32_e32 v7, 0xffff0000, v1
-; GFX1250-REAL16-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-REAL16-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
-; GFX1250-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-REAL16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-REAL16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-REAL16-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_bitop2_b32 v5, s0, v0 bitop3:0x40
+; GFX1250-REAL16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-REAL16-NEXT:    v_dual_lshlrev_b32 v6, 16, v1 :: v_dual_bitop2_b32 v7, s0, v1 bitop3:0x40
+; GFX1250-REAL16-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-REAL16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-REAL16-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
 ; GFX1250-REAL16-NEXT:    v_pk_fma_f32 v[0:1], v[4:5], v[6:7], v[0:1]
-; GFX1250-REAL16-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v1
 ; GFX1250-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-REAL16-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v1
 ; GFX1250-REAL16-NEXT:    v_mov_b16_e32 v1.l, v0.h
+; GFX1250-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-REAL16-NEXT:    v_pk_max_num_bf16 v1, v1, v1 op_sel_hi:[0,0] clamp
-; GFX1250-REAL16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-REAL16-NEXT:    v_mov_b16_e32 v0.h, v1.l
 ; GFX1250-REAL16-NEXT:    s_set_pc_i64 s[30:31]
   %src0.ext = fpext <2 x bfloat> %src0 to <2 x float>
@@ -378,16 +403,19 @@ define <2 x bfloat> @v_mad_mix_v2f32_clamp_precvt(<2 x bfloat> %src0, <2 x bfloa
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v5, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_lshlrev_b32 v6, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v1
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v0 :: v_dual_bitop2_b32 v5, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v1 :: v_dual_bitop2_b32 v7, s0, v1 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
 ; GFX1250-NEXT:    v_pk_fma_f32 v[0:1], v[4:5], v[6:7], v[0:1]
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_max_num_f32_e64 v1, v1, v1 clamp
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_max_num_f32_e64 v0, v0, v0 clamp
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_cvt_pk_bf16_f32 v0, v0, v1
 ; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
   %src0.ext = fpext <2 x bfloat> %src0 to <2 x float>
@@ -439,15 +467,21 @@ define <4 x bfloat> @v_mad_mix_v4f32_clamp_precvt(<4 x bfloat> %src0, <4 x bfloa
 ; GFX1250:       ; %bb.0:
 ; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    v_and_b32_e32 v7, 0xffff0000, v0
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v0 :: v_dual_lshlrev_b32 v8, 16, v1
-; GFX1250-NEXT:    v_and_b32_e32 v9, 0xffff0000, v1
-; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_lshlrev_b32 v10, 16, v3
-; GFX1250-NEXT:    v_and_b32_e32 v11, 0xffff0000, v3
-; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v4
-; GFX1250-NEXT:    v_and_b32_e32 v13, 0xffff0000, v5
-; GFX1250-NEXT:    v_dual_lshlrev_b32 v12, 16, v5 :: v_dual_lshlrev_b32 v2, 16, v4
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v6, 16, v0 :: v_dual_bitop2_b32 v7, s0, v0 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v8, 16, v1 :: v_dual_bitop2_b32 v9, s0, v1 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s0, v2 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v10, 16, v3 :: v_dual_bitop2_b32 v11, s0, v3 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v12, 16, v5 :: v_dual_bitop2_b32 v3, s0, v4 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s0, 0xffff0000
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v4 :: v_dual_bitop2_b32 v13, s0, v5 bitop3:0x40
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_pk_fma_f32 v[4:5], v[8:9], v[10:11], v[12:13]
 ; GFX1250-NEXT:    v_pk_fma_f32 v[0:1], v[6:7], v[0:1], v[2:3]
diff --git a/llvm/test/CodeGen/AMDGPU/mul.ll b/llvm/test/CodeGen/AMDGPU/mul.ll
index a1ad790911aa3..6fa32dc889920 100644
--- a/llvm/test/CodeGen/AMDGPU/mul.ll
+++ b/llvm/test/CodeGen/AMDGPU/mul.ll
@@ -4067,7 +4067,10 @@ define amdgpu_kernel void @v_mul_i128(ptr addrspace(1) %out, ptr addrspace(1) %a
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x2c nv
-; GFX1250-NEXT:    v_and_b32_e32 v16, 0x3ff, v0
+; GFX1250-NEXT:    s_wait_xcnt 0x0
+; GFX1250-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v13, 0 :: v_dual_bitop2_b32 v16, s4, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    s_clause 0x1
 ; GFX1250-NEXT:    global_load_b128 v[0:3], v16, s[0:1] scale_offset
@@ -4076,7 +4079,7 @@ define amdgpu_kernel void @v_mul_i128(ptr addrspace(1) %out, ptr addrspace(1) %a
 ; GFX1250-NEXT:    v_mad_nc_u64_u32 v[8:9], v0, v4, 0
 ; GFX1250-NEXT:    v_mad_nc_u64_u32 v[10:11], v4, v2, 0
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
-; GFX1250-NEXT:    v_dual_mov_b32 v13, 0 :: v_dual_mov_b32 v12, v9
+; GFX1250-NEXT:    v_mov_b32_e32 v12, v9
 ; GFX1250-NEXT:    v_mad_u32 v3, v4, v3, v11
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_2)
 ; GFX1250-NEXT:    v_mad_nc_u64_u32 v[14:15], v1, v4, v[12:13]
@@ -4101,8 +4104,9 @@ define amdgpu_kernel void @v_mul_i128(ptr addrspace(1) %out, ptr addrspace(1) %a
 ; GFX13-LABEL: v_mul_i128:
 ; GFX13:       ; %bb.0: ; %entry
 ; GFX13-NEXT:    s_load_b128 s[0:3], s[4:5], 0x2c nv
-; GFX13-NEXT:    v_and_b32_e32 v13, 0x3ff, v0
-; GFX13-NEXT:    v_mov_b32_e32 v10, 0
+; GFX13-NEXT:    s_mov_b32 s4, 0x3ff
+; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX13-NEXT:    v_dual_mov_b32 v10, 0 :: v_dual_bitop2_b32 v13, s4, v0 bitop3:0x40
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    global_load_b128 v[0:3], v13, s[0:1] scale_offset
diff --git a/llvm/test/CodeGen/AMDGPU/packed-fp32.ll b/llvm/test/CodeGen/AMDGPU/packed-fp32.ll
index 55f60193236a0..a678d67892474 100644
--- a/llvm/test/CodeGen/AMDGPU/packed-fp32.ll
+++ b/llvm/test/CodeGen/AMDGPU/packed-fp32.ll
@@ -156,11 +156,13 @@ define amdgpu_kernel void @fadd_v4_vs(ptr addrspace(1) %a, <4 x float> %x) {
 ; GFX1250-SDAG-NEXT:    s_clause 0x1
 ; GFX1250-SDAG-NEXT:    s_load_b64 s[6:7], s[4:5], 0x24 nv
 ; GFX1250-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
-; GFX1250-SDAG-NEXT:    v_and_b32_e32 v8, 0x3ff, v0
+; GFX1250-SDAG-NEXT:    s_wait_xcnt 0x0
+; GFX1250-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
 ; GFX1250-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-SDAG-NEXT:    v_dual_mov_b32 v4, s2 :: v_dual_bitop2_b32 v8, s4, v0 bitop3:0x40
 ; GFX1250-SDAG-NEXT:    global_load_b128 v[0:3], v8, s[6:7] scale_offset
 ; GFX1250-SDAG-NEXT:    v_mov_b64_e32 v[6:7], s[0:1]
-; GFX1250-SDAG-NEXT:    v_dual_mov_b32 v4, s2 :: v_dual_mov_b32 v5, s3
+; GFX1250-SDAG-NEXT:    v_mov_b32_e32 v5, s3
 ; GFX1250-SDAG-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-SDAG-NEXT:    v_pk_add_f32 v[2:3], v[2:3], v[4:5]
@@ -1085,12 +1087,12 @@ define amdgpu_kernel void @fadd_v2_v_fneg(ptr addrspace(1) %a, float %x) {
 ; GFX1250-GISEL-NEXT:    v_nop
 ; GFX1250-GISEL-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-GISEL-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
-; GFX1250-GISEL-NEXT:    v_and_b32_e32 v4, 0x3ff, v0
 ; GFX1250-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-GISEL-NEXT:    global_load_b64 v[0:1], v4, s[0:1] scale_offset
 ; GFX1250-GISEL-NEXT:    v_max_num_f32_e64 v2, -s2, -s2
-; GFX1250-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1250-GISEL-NEXT:    v_mov_b32_e32 v3, v2
+; GFX1250-GISEL-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT:    v_dual_mov_b32 v3, v2 :: v_dual_bitop2_b32 v4, s2, v0 bitop3:0x40
+; GFX1250-GISEL-NEXT:    global_load_b64 v[0:1], v4, s[0:1] scale_offset
 ; GFX1250-GISEL-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-GISEL-NEXT:    v_pk_add_f32 v[0:1], v[0:1], v[2:3]
 ; GFX1250-GISEL-NEXT:    global_store_b64 v4, v[0:1], s[0:1] scale_offset
@@ -1670,11 +1672,13 @@ define amdgpu_kernel void @fmul_v4_vs(ptr addrspace(1) %a, <4 x float> %x) {
 ; GFX1250-SDAG-NEXT:    s_clause 0x1
 ; GFX1250-SDAG-NEXT:    s_load_b64 s[6:7], s[4:5], 0x24 nv
 ; GFX1250-SDAG-NEXT:    s_load_b128 s[0:3], s[4:5], 0x34 nv
-; GFX1250-SDAG-NEXT:    v_and_b32_e32 v8, 0x3ff, v0
+; GFX1250-SDAG-NEXT:    s_wait_xcnt 0x0
+; GFX1250-SDAG-NEXT:    s_mov_b32 s4, 0x3ff
 ; GFX1250-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-SDAG-NEXT:    v_dual_mov_b32 v4, s2 :: v_dual_bitop2_b32 v8, s4, v0 bitop3:0x40
 ; GFX1250-SDAG-NEXT:    global_load_b128 v[0:3], v8, s[6:7] scale_offset
 ; GFX1250-SDAG-NEXT:    v_mov_b64_e32 v[6:7], s[0:1]
-; GFX1250-SDAG-NEXT:    v_dual_mov_b32 v4, s2 :: v_dual_mov_b32 v5, s3
+; GFX1250-SDAG-NEXT:    v_mov_b32_e32 v5, s3
 ; GFX1250-SDAG-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-SDAG-NEXT:    v_pk_mul_f32 v[2:3], v[2:3], v[4:5]
@@ -2462,12 +2466,12 @@ define amdgpu_kernel void @fmul_v2_v_fneg(ptr addrspace(1) %a, float %x) {
 ; GFX1250-GISEL-NEXT:    v_nop
 ; GFX1250-GISEL-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-GISEL-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
-; GFX1250-GISEL-NEXT:    v_and_b32_e32 v4, 0x3ff, v0
 ; GFX1250-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-GISEL-NEXT:    global_load_b64 v[0:1], v4, s[0:1] scale_offset
 ; GFX1250-GISEL-NEXT:    v_max_num_f32_e64 v2, -s2, -s2
-; GFX1250-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1250-GISEL-NEXT:    v_mov_b32_e32 v3, v2
+; GFX1250-GISEL-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT:    v_dual_mov_b32 v3, v2 :: v_dual_bitop2_b32 v4, s2, v0 bitop3:0x40
+; GFX1250-GISEL-NEXT:    global_load_b64 v[0:1], v4, s[0:1] scale_offset
 ; GFX1250-GISEL-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-GISEL-NEXT:    v_pk_mul_f32 v[0:1], v[0:1], v[2:3]
 ; GFX1250-GISEL-NEXT:    global_store_b64 v4, v[0:1], s[0:1] scale_offset
@@ -3520,12 +3524,12 @@ define amdgpu_kernel void @fma_v2_v_fneg(ptr addrspace(1) %a, float %x) {
 ; GFX1250-GISEL-NEXT:    v_nop
 ; GFX1250-GISEL-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-GISEL-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
-; GFX1250-GISEL-NEXT:    v_and_b32_e32 v4, 0x3ff, v0
 ; GFX1250-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-GISEL-NEXT:    global_load_b64 v[0:1], v4, s[0:1] scale_offset
 ; GFX1250-GISEL-NEXT:    v_max_num_f32_e64 v2, -s2, -s2
-; GFX1250-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1250-GISEL-NEXT:    v_mov_b32_e32 v3, v2
+; GFX1250-GISEL-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1250-GISEL-NEXT:    v_dual_mov_b32 v3, v2 :: v_dual_bitop2_b32 v4, s2, v0 bitop3:0x40
+; GFX1250-GISEL-NEXT:    global_load_b64 v[0:1], v4, s[0:1] scale_offset
 ; GFX1250-GISEL-NEXT:    s_wait_loadcnt 0x0
 ; GFX1250-GISEL-NEXT:    v_pk_fma_f32 v[0:1], v[0:1], v[2:3], v[2:3]
 ; GFX1250-GISEL-NEXT:    global_store_b64 v4, v[0:1], s[0:1] scale_offset
diff --git a/llvm/test/CodeGen/AMDGPU/packed-fp64.ll b/llvm/test/CodeGen/AMDGPU/packed-fp64.ll
index d7c94df43b354..2d2e79eb70a68 100644
--- a/llvm/test/CodeGen/AMDGPU/packed-fp64.ll
+++ b/llvm/test/CodeGen/AMDGPU/packed-fp64.ll
@@ -448,9 +448,9 @@ define amdgpu_kernel void @fadd_v2_v_v_vgpr_splat(ptr addrspace(1) %a) {
 ; GFX1251-NEXT:    v_nop
 ; GFX1251-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1251-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1251-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1251-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1251-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1251-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1251-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1251-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1251-NEXT:    v_mov_b64_e32 v[2:3], v[0:1]
 ; GFX1251-NEXT:    s_wait_kmcnt 0x0
 ; GFX1251-NEXT:    global_load_b128 v[4:7], v0, s[0:1] scale_offset
@@ -1379,9 +1379,9 @@ define amdgpu_kernel void @fmul_v2_v_v_splat(ptr addrspace(1) %a) {
 ; GFX1251-NEXT:    v_nop
 ; GFX1251-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1251-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1251-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1251-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1251-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1251-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1251-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1251-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1251-NEXT:    v_mov_b64_e32 v[2:3], v[0:1]
 ; GFX1251-NEXT:    s_wait_kmcnt 0x0
 ; GFX1251-NEXT:    global_load_b128 v[4:7], v0, s[0:1] scale_offset
@@ -1964,9 +1964,9 @@ define amdgpu_kernel void @fma_v2_v_v_splat(ptr addrspace(1) %a) {
 ; GFX1251-NEXT:    v_nop
 ; GFX1251-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1251-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1251-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1251-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1251-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1251-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1251-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1251-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1251-NEXT:    v_mov_b64_e32 v[2:3], v[0:1]
 ; GFX1251-NEXT:    s_wait_kmcnt 0x0
 ; GFX1251-NEXT:    global_load_b128 v[4:7], v0, s[0:1] scale_offset
@@ -2529,14 +2529,14 @@ define amdgpu_kernel void @fneg_v2f64_pkfma(ptr addrspace(1) %out) {
 ; GFX1251-SDAG-NEXT:    s_mov_b64 s[64:65], 0
 ; GFX1251-SDAG-NEXT:    v_nop
 ; GFX1251-SDAG-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
-; GFX1251-SDAG-NEXT:    v_and_b32_e32 v1, 0x3ff, v0
+; GFX1251-SDAG-NEXT:    s_mov_b32 s0, 0x3ff
+; GFX1251-SDAG-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
+; GFX1251-SDAG-NEXT:    v_dual_mov_b32 v0, 0 :: v_dual_bitop2_b32 v1, s0, v0 bitop3:0x40
 ; GFX1251-SDAG-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1251-SDAG-NEXT:    v_mov_b32_e32 v0, 0
-; GFX1251-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1251-SDAG-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 0, v1
 ; GFX1251-SDAG-NEXT:    v_cndmask_b32_e64 v1, 0x3ff00000, 0, vcc_lo
+; GFX1251-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1251-SDAG-NEXT:    v_mov_b64_e32 v[2:3], v[0:1]
-; GFX1251-SDAG-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1251-SDAG-NEXT:    v_pk_fma_f64 v[2:5], v[0:3], 0, v[0:3] neg_lo:[0,0,1] neg_hi:[0,0,1]
 ; GFX1251-SDAG-NEXT:    s_wait_kmcnt 0x0
 ; GFX1251-SDAG-NEXT:    global_store_b128 v0, v[2:5], s[0:1]
@@ -2550,13 +2550,14 @@ define amdgpu_kernel void @fneg_v2f64_pkfma(ptr addrspace(1) %out) {
 ; GFX1251-GISEL-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1251-GISEL-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
 ; GFX1251-GISEL-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1251-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1251-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1251-GISEL-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 0, v0
-; GFX1251-GISEL-NEXT:    v_mov_b32_e32 v0, 0
 ; GFX1251-GISEL-NEXT:    v_cndmask_b32_e64 v1, 0x3ff00000, 0, vcc_lo
+; GFX1251-GISEL-NEXT:    s_mov_b32 s2, 0x80000000
+; GFX1251-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX1251-GISEL-NEXT:    v_dual_mov_b32 v0, 0 :: v_dual_bitop2_b32 v5, s2, v1 bitop3:0x14
+; GFX1251-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
 ; GFX1251-GISEL-NEXT:    v_mov_b32_e32 v4, v0
-; GFX1251-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(SKIP_1) | instid1(VALU_DEP_2)
-; GFX1251-GISEL-NEXT:    v_xor_b32_e32 v5, 0x80000000, v1
 ; GFX1251-GISEL-NEXT:    v_mov_b64_e32 v[2:3], v[0:1]
 ; GFX1251-GISEL-NEXT:    v_mov_b64_e32 v[6:7], v[4:5]
 ; GFX1251-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
diff --git a/llvm/test/CodeGen/AMDGPU/packed-u64.ll b/llvm/test/CodeGen/AMDGPU/packed-u64.ll
index d8ceebf08b758..54d43e61ebbc9 100644
--- a/llvm/test/CodeGen/AMDGPU/packed-u64.ll
+++ b/llvm/test/CodeGen/AMDGPU/packed-u64.ll
@@ -448,9 +448,9 @@ define amdgpu_kernel void @add_v2_v_v_vgpr_splat(ptr addrspace(1) %a) {
 ; GFX1251-NEXT:    v_nop
 ; GFX1251-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1251-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1251-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1251-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1251-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1251-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1251-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1251-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1251-NEXT:    v_mov_b64_e32 v[2:3], v[0:1]
 ; GFX1251-NEXT:    s_wait_kmcnt 0x0
 ; GFX1251-NEXT:    global_load_b128 v[4:7], v0, s[0:1] scale_offset
@@ -1144,9 +1144,9 @@ define amdgpu_kernel void @sub_v2_v_v_splat(ptr addrspace(1) %a) {
 ; GFX1251-NEXT:    v_nop
 ; GFX1251-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1251-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1251-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1251-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1251-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1251-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1251-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1251-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1251-NEXT:    v_mov_b64_e32 v[2:3], v[0:1]
 ; GFX1251-NEXT:    s_wait_kmcnt 0x0
 ; GFX1251-NEXT:    global_load_b128 v[4:7], v0, s[0:1] scale_offset
diff --git a/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm-gfx12.ll b/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm-gfx12.ll
index 1a04889eba18e..8ebd6d63f0546 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm-gfx12.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm-gfx12.ll
@@ -15,15 +15,15 @@ define amdgpu_kernel void @promote_async_load_offset_negative(ptr addrspace(1) %
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
+; GFX1250-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_mov_b64 s[2:3], 0xffffffffffffff00
 ; GFX1250-NEXT:    v_add_nc_u32_e64 v4, 0xfffffe00, 0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_nc_u32_e32 v0, 0x100, v0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[2:3], s[0:1], v[0:1]
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[2:3], s[2:3], v[2:3]
 ; GFX1250-NEXT:    global_load_async_to_lds_b128 v1, v0, s[0:1]
 ; GFX1250-NEXT:    s_clause 0x1
@@ -65,12 +65,13 @@ define amdgpu_kernel void @promote_async_load_offset_positive(ptr addrspace(1) %
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX1250-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1250-NEXT:    v_add_nc_u32_e64 v4, 0xffffff00, 0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[2:3], s[0:1], v[0:1]
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[2:3], 0x100, v[2:3]
 ; GFX1250-NEXT:    global_load_async_to_lds_b128 v1, v0, s[0:1]
 ; GFX1250-NEXT:    s_clause 0x1
@@ -114,15 +115,15 @@ define amdgpu_kernel void @promote_async_store_offset_negative(ptr addrspace(1)
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    v_mov_b32_e32 v1, 0
+; GFX1250-NEXT:    s_mov_b32 s2, 0x3ff
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_2) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_bitop2_b32 v0, s2, v0 bitop3:0x40
 ; GFX1250-NEXT:    s_mov_b64 s[2:3], 0xffffffffffffff00
 ; GFX1250-NEXT:    v_add_nc_u32_e64 v4, 0xfffffe00, 0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_nc_u32_e32 v0, 0x100, v0
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[2:3], s[0:1], v[0:1]
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_nc_u64_e32 v[2:3], s[2:3], v[2:3]
 ; GFX1250-NEXT:    global_store_async_from_lds_b128 v0, v1, s[0:1]
 ; GFX1250-NEXT:    s_clause 0x1
diff --git a/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm-gfx12.mir b/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm-gfx12.mir
index 5c3d5ae8426f8..72b4a936b8eaf 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm-gfx12.mir
+++ b/llvm/test/CodeGen/AMDGPU/promote-constOffset-to-imm-gfx12.mir
@@ -106,8 +106,8 @@ body:             |
     ; GFX1250-NEXT: frame-setup CFI_INSTRUCTION undefined $sgpr0
     ; GFX1250-NEXT: frame-setup CFI_INSTRUCTION undefined $sgpr1
     ; GFX1250-NEXT: early-clobber renamable $sgpr0_sgpr1 = S_LOAD_DWORDX2_IMM_ec killed renamable $sgpr4_sgpr5, 36, 32 :: (dereferenceable invariant load (s64), align 4, addrspace 4)
-    ; GFX1250-NEXT: renamable $vgpr1 = V_MOV_B32_e32 0, implicit $exec
-    ; GFX1250-NEXT: renamable $vgpr0 = V_AND_B32_e32 1023, killed $vgpr0, implicit $exec
+    ; GFX1250-NEXT: $sgpr2 = S_MOV_B32 1023
+    ; GFX1250-NEXT: renamable $vgpr1, renamable $vgpr0 = V_DUAL_MOV_B32_e32_X_BITOP2_B32_e64_e96_gfx1250 0, $sgpr2, killed $vgpr0, 64, implicit $exec, implicit $exec, implicit $exec
     ; GFX1250-NEXT: renamable $vgpr4 = V_ADD_U32_e64 -256, 0, 0, implicit $exec
     ; GFX1250-NEXT: renamable $vgpr2_vgpr3 = V_ADD_U64_e32 $sgpr0_sgpr1, $vgpr0_vgpr1, implicit $exec
     ; GFX1250-NEXT: renamable $vgpr2_vgpr3 = V_ADD_U64_e32 256, killed $vgpr2_vgpr3, implicit $exec
diff --git a/llvm/test/CodeGen/AMDGPU/reassoc-mul-add-1-to-mad.ll b/llvm/test/CodeGen/AMDGPU/reassoc-mul-add-1-to-mad.ll
index 8fa4634d5bbe8..f702e53c93cd8 100644
--- a/llvm/test/CodeGen/AMDGPU/reassoc-mul-add-1-to-mad.ll
+++ b/llvm/test/CodeGen/AMDGPU/reassoc-mul-add-1-to-mad.ll
@@ -5313,36 +5313,37 @@ define amdgpu_kernel void @compute_mad(ptr addrspace(4) %i18, ptr addrspace(4) %
 ; GFX1250-NEXT:    v_nop
 ; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-NEXT:    s_load_b96 s[0:2], s[4:5], 0x10 nv
-; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
-; GFX1250-NEXT:    s_wait_xcnt 0x0
-; GFX1250-NEXT:    s_load_b128 s[4:7], s[4:5], 0x0 nv
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    s_add_co_i32 s2, s2, 1
-; GFX1250-NEXT:    s_load_b32 s6, s[6:7], 0x4 nv
+; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX1250-NEXT:    s_load_b128 s[4:7], s[4:5], 0x0 nv
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_mul_lo_u32 v1, s2, v0
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_dual_add_nc_u32 v2, s2, v1 :: v_dual_add_nc_u32 v1, 1, v1
 ; GFX1250-NEXT:    s_bfe_u32 s2, ttmp6, 0x4000c
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT:    v_mul_lo_u32 v2, v2, v0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    s_load_b32 s6, s[6:7], 0x4 nv
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
 ; GFX1250-NEXT:    s_add_co_i32 s7, s2, 1
-; GFX1250-NEXT:    v_mul_lo_u32 v2, v2, v0
 ; GFX1250-NEXT:    s_load_b64 s[2:3], s[4:5], 0x0 nv
 ; GFX1250-NEXT:    s_wait_xcnt 0x0
 ; GFX1250-NEXT:    s_and_b32 s4, ttmp6, 15
 ; GFX1250-NEXT:    s_mul_i32 s5, ttmp9, s7
+; GFX1250-NEXT:    v_mul_lo_u32 v3, v2, v1
 ; GFX1250-NEXT:    s_getreg_b32 s7, hwreg(HW_REG_IB_STS2, 6, 4)
 ; GFX1250-NEXT:    s_add_co_i32 s4, s4, s5
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    v_add_nc_u32_e32 v1, v3, v1
+; GFX1250-NEXT:    v_mul_lo_u32 v1, v1, v2
+; GFX1250-NEXT:    v_add_nc_u32_e32 v2, 1, v3
 ; GFX1250-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-NEXT:    s_and_b32 s5, s6, 0xffff
 ; GFX1250-NEXT:    s_cmp_eq_u32 s7, 0
-; GFX1250-NEXT:    v_mul_lo_u32 v3, v2, v1
 ; GFX1250-NEXT:    s_cselect_b32 s4, ttmp9, s4
-; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_2)
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1250-NEXT:    v_mad_u32 v0, s4, s5, v0
-; GFX1250-NEXT:    v_add_nc_u32_e32 v1, v3, v1
-; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX1250-NEXT:    v_mul_lo_u32 v1, v1, v2
-; GFX1250-NEXT:    v_add_nc_u32_e32 v2, 1, v3
 ; GFX1250-NEXT:    v_mul_lo_u32 v3, v1, v2
 ; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX1250-NEXT:    v_add_nc_u32_e32 v2, v3, v2
@@ -5359,30 +5360,32 @@ define amdgpu_kernel void @compute_mad(ptr addrspace(4) %i18, ptr addrspace(4) %
 ;
 ; GFX13-LABEL: compute_mad:
 ; GFX13:       ; %bb.0: ; %bb
+; GFX13-NEXT:    s_clause 0x1
 ; GFX13-NEXT:    s_load_b96 s[0:2], s[4:5], 0x10 nv
-; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
 ; GFX13-NEXT:    s_load_b128 s[4:7], s[4:5], 0x0 nv
 ; GFX13-NEXT:    s_bfe_u32 s8, ttmp6, 0x4000c
 ; GFX13-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
 ; GFX13-NEXT:    s_add_co_i32 s8, s8, 1
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-NEXT:    s_add_co_i32 s2, s2, 1
+; GFX13-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
 ; GFX13-NEXT:    s_load_b32 s6, s[6:7], 0x4 nv
-; GFX13-NEXT:    v_mul_lo_u32 v1, s2, v0
 ; GFX13-NEXT:    s_and_b32 s7, ttmp6, 15
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_mul_lo_u32 v1, s2, v0
 ; GFX13-NEXT:    v_dual_add_nc_u32 v2, s2, v1 :: v_dual_add_nc_u32 v1, 1, v1
 ; GFX13-NEXT:    s_load_b64 s[2:3], s[4:5], 0x0 nv
 ; GFX13-NEXT:    s_mul_i32 s4, ttmp9, s8
 ; GFX13-NEXT:    s_getreg_b32 s5, hwreg(HW_REG_IB_STS2, 6, 4)
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1)
 ; GFX13-NEXT:    v_mul_lo_u32 v2, v2, v0
 ; GFX13-NEXT:    s_add_co_i32 s7, s7, s4
 ; GFX13-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-NEXT:    s_and_b32 s4, s6, 0xffff
 ; GFX13-NEXT:    s_cmp_eq_u32 s5, 0
-; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
-; GFX13-NEXT:    v_mul_lo_u32 v3, v2, v1
 ; GFX13-NEXT:    s_cselect_b32 s5, ttmp9, s7
+; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX13-NEXT:    v_mul_lo_u32 v3, v2, v1
 ; GFX13-NEXT:    v_add_nc_u32_e32 v1, v3, v1
 ; GFX13-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
 ; GFX13-NEXT:    v_mul_lo_u32 v2, v1, v2
diff --git a/llvm/test/CodeGen/AMDGPU/v_ashr_pk.ll b/llvm/test/CodeGen/AMDGPU/v_ashr_pk.ll
index b06c6709cb370..59197827e1efd 100644
--- a/llvm/test/CodeGen/AMDGPU/v_ashr_pk.ll
+++ b/llvm/test/CodeGen/AMDGPU/v_ashr_pk.ll
@@ -31,8 +31,9 @@ define amdgpu_kernel void @v_ashr_pk_i8_i32(ptr addrspace(1) %out, i32 %src0, i3
 ; GFX1250-TRUE16-NEXT:    s_clause 0x1
 ; GFX1250-TRUE16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x2c nv
 ; GFX1250-TRUE16-NEXT:    s_load_b64 s[6:7], s[4:5], 0x24 nv
-; GFX1250-TRUE16-NEXT:    v_mov_b32_e32 v0, 0x7f
-; GFX1250-TRUE16-NEXT:    v_mov_b32_e32 v2, 0
+; GFX1250-TRUE16-NEXT:    s_mov_b32 s3, 0x7f
+; GFX1250-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX1250-TRUE16-NEXT:    v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v2, 0
 ; GFX1250-TRUE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-TRUE16-NEXT:    s_ashr_i32 s0, s0, s2
 ; GFX1250-TRUE16-NEXT:    s_ashr_i32 s1, s1, s2
@@ -52,8 +53,9 @@ define amdgpu_kernel void @v_ashr_pk_i8_i32(ptr addrspace(1) %out, i32 %src0, i3
 ; GFX1250-FAKE16-NEXT:    s_clause 0x1
 ; GFX1250-FAKE16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x2c nv
 ; GFX1250-FAKE16-NEXT:    s_load_b64 s[6:7], s[4:5], 0x24 nv
-; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v0, 0x7f
-; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v2, 0
+; GFX1250-FAKE16-NEXT:    s_mov_b32 s3, 0x7f
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v2, 0
 ; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-FAKE16-NEXT:    s_ashr_i32 s0, s0, s2
 ; GFX1250-FAKE16-NEXT:    s_ashr_i32 s1, s1, s2
@@ -69,8 +71,9 @@ define amdgpu_kernel void @v_ashr_pk_i8_i32(ptr addrspace(1) %out, i32 %src0, i3
 ; GFX13-TRUE16-NEXT:    s_clause 0x1
 ; GFX13-TRUE16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x2c nv
 ; GFX13-TRUE16-NEXT:    s_load_b64 s[4:5], s[4:5], 0x24 nv
-; GFX13-TRUE16-NEXT:    v_mov_b32_e32 v0, 0x7f
-; GFX13-TRUE16-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-TRUE16-NEXT:    s_mov_b32 s3, 0x7f
+; GFX13-TRUE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-TRUE16-NEXT:    v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v2, 0
 ; GFX13-TRUE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-TRUE16-NEXT:    s_ashr_i32 s0, s0, s2
 ; GFX13-TRUE16-NEXT:    s_ashr_i32 s1, s1, s2
@@ -86,8 +89,9 @@ define amdgpu_kernel void @v_ashr_pk_i8_i32(ptr addrspace(1) %out, i32 %src0, i3
 ; GFX13-FAKE16-NEXT:    s_clause 0x1
 ; GFX13-FAKE16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x2c nv
 ; GFX13-FAKE16-NEXT:    s_load_b64 s[4:5], s[4:5], 0x24 nv
-; GFX13-FAKE16-NEXT:    v_mov_b32_e32 v0, 0x7f
-; GFX13-FAKE16-NEXT:    v_mov_b32_e32 v2, 0
+; GFX13-FAKE16-NEXT:    s_mov_b32 s3, 0x7f
+; GFX13-FAKE16-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX13-FAKE16-NEXT:    v_dual_mov_b32 v0, s3 :: v_dual_mov_b32 v2, 0
 ; GFX13-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GFX13-FAKE16-NEXT:    s_ashr_i32 s0, s0, s2
 ; GFX13-FAKE16-NEXT:    s_ashr_i32 s1, s1, s2
diff --git a/llvm/test/CodeGen/AMDGPU/vopd-combine-gfx1250.mir b/llvm/test/CodeGen/AMDGPU/vopd-combine-gfx1250.mir
index 3ee98c0161de1..af317d454506c 100644
--- a/llvm/test/CodeGen/AMDGPU/vopd-combine-gfx1250.mir
+++ b/llvm/test/CodeGen/AMDGPU/vopd-combine-gfx1250.mir
@@ -869,34 +869,34 @@ body:             |
     $vgpr6 = V_SUB_F32_e32 12345, $vgpr1, implicit $mode, implicit $exec
 ...
 
-# Below 2 tests cannot use VOPD because of the vdst parity and cannot use
-# VOPD3 because of the literal use.
+# Below 2 tests cannot use VOPD because of the vdst parity, and VOPD3 cannot
+# encode the literal, so the literal is moved into a scalar register first.
 ---
-name:            vopd_no_combine_literal_x
+name:            vopd_combine_moved_literal_x
 tracksRegLiveness: true
 body:             |
   bb.0:
 
-    ; SCHED-LABEL: name: vopd_no_combine_literal_x
+    ; SCHED-LABEL: name: vopd_combine_moved_literal_x
     ; SCHED: $vgpr0 = IMPLICIT_DEF
     ; SCHED-NEXT: $vgpr1 = IMPLICIT_DEF
     ; SCHED-NEXT: $vgpr3 = V_SUB_F32_e32 12345, $vgpr1, implicit $mode, implicit $exec
-    ; SCHED-NEXT: $vgpr4 = V_BFM_B32_e32 $vgpr0, killed $vgpr1, implicit $exec
-    ; SCHED-NEXT: $vgpr5 = V_MUL_F32_e32 killed $vgpr0, $vgpr0, implicit $mode, implicit $exec
+    ; SCHED-NEXT: $vgpr5 = V_MUL_F32_e32 $vgpr0, $vgpr0, implicit $mode, implicit $exec
+    ; SCHED-NEXT: $vgpr4 = V_BFM_B32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
     ;
-    ; PAIR-LABEL: name: vopd_no_combine_literal_x
+    ; PAIR-LABEL: name: vopd_combine_moved_literal_x
     ; PAIR: $vgpr0 = IMPLICIT_DEF
     ; PAIR-NEXT: $vgpr1 = IMPLICIT_DEF
-    ; PAIR-NEXT: $vgpr3 = V_SUB_F32_e32 12345, $vgpr1, implicit $mode, implicit $exec
-    ; PAIR-NEXT: $vgpr4 = V_BFM_B32_e32 $vgpr0, killed $vgpr1, implicit $exec
-    ; PAIR-NEXT: $vgpr5 = V_MUL_F32_e32 killed $vgpr0, $vgpr0, implicit $mode, implicit $exec
+    ; PAIR-NEXT: $sgpr0 = S_MOV_B32 12345
+    ; PAIR-NEXT: $vgpr3, $vgpr5 = V_DUAL_SUB_F32_e32_X_MUL_F32_e32_e96_gfx1250 0, $sgpr0, 0, $vgpr1, 0, $vgpr0, 0, $vgpr0, implicit $mode, implicit $exec, implicit $mode, implicit $exec, implicit $mode, implicit $exec
+    ; PAIR-NEXT: $vgpr4 = V_BFM_B32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
     ;
-    ; LOWER-LABEL: name: vopd_no_combine_literal_x
+    ; LOWER-LABEL: name: vopd_combine_moved_literal_x
     ; LOWER: $vgpr0 = IMPLICIT_DEF
     ; LOWER-NEXT: $vgpr1 = IMPLICIT_DEF
-    ; LOWER-NEXT: $vgpr3 = V_SUB_F32_e32 12345, $vgpr1, implicit $mode, implicit $exec
-    ; LOWER-NEXT: $vgpr4 = V_BFM_B32_e32 $vgpr0, killed $vgpr1, implicit $exec
-    ; LOWER-NEXT: $vgpr5 = V_MUL_F32_e32 killed $vgpr0, $vgpr0, implicit $mode, implicit $exec
+    ; LOWER-NEXT: $sgpr0 = S_MOV_B32 12345
+    ; LOWER-NEXT: $vgpr3, $vgpr5 = V_DUAL_SUB_F32_e32_X_MUL_F32_e32_e96_gfx1250 0, $sgpr0, 0, $vgpr1, 0, $vgpr0, 0, $vgpr0, implicit $mode, implicit $exec, implicit $mode, implicit $exec, implicit $mode, implicit $exec
+    ; LOWER-NEXT: $vgpr4 = V_BFM_B32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
     $vgpr0 = IMPLICIT_DEF
     $vgpr1 = IMPLICIT_DEF
     $vgpr3 = V_SUB_F32_e32 12345, $vgpr1, implicit $mode, implicit $exec
@@ -905,31 +905,31 @@ body:             |
 ...
 
 ---
-name:            vopd_no_combine_literal_y
+name:            vopd_combine_moved_literal_y
 tracksRegLiveness: true
 body:             |
   bb.0:
 
-    ; SCHED-LABEL: name: vopd_no_combine_literal_y
+    ; SCHED-LABEL: name: vopd_combine_moved_literal_y
     ; SCHED: $vgpr0 = IMPLICIT_DEF
     ; SCHED-NEXT: $vgpr1 = IMPLICIT_DEF
     ; SCHED-NEXT: $vgpr3 = V_MUL_F32_e32 $vgpr0, $vgpr0, implicit $mode, implicit $exec
-    ; SCHED-NEXT: $vgpr4 = V_BFM_B32_e32 killed $vgpr0, $vgpr1, implicit $exec
-    ; SCHED-NEXT: $vgpr5 = V_SUB_F32_e32 12345, killed $vgpr1, implicit $mode, implicit $exec
+    ; SCHED-NEXT: $vgpr5 = V_SUB_F32_e32 12345, $vgpr1, implicit $mode, implicit $exec
+    ; SCHED-NEXT: $vgpr4 = V_BFM_B32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
     ;
-    ; PAIR-LABEL: name: vopd_no_combine_literal_y
+    ; PAIR-LABEL: name: vopd_combine_moved_literal_y
     ; PAIR: $vgpr0 = IMPLICIT_DEF
     ; PAIR-NEXT: $vgpr1 = IMPLICIT_DEF
-    ; PAIR-NEXT: $vgpr3 = V_MUL_F32_e32 $vgpr0, $vgpr0, implicit $mode, implicit $exec
-    ; PAIR-NEXT: $vgpr4 = V_BFM_B32_e32 killed $vgpr0, $vgpr1, implicit $exec
-    ; PAIR-NEXT: $vgpr5 = V_SUB_F32_e32 12345, killed $vgpr1, implicit $mode, implicit $exec
+    ; PAIR-NEXT: $sgpr0 = S_MOV_B32 12345
+    ; PAIR-NEXT: $vgpr3, $vgpr5 = V_DUAL_MUL_F32_e32_X_SUB_F32_e32_e96_gfx1250 0, $vgpr0, 0, $vgpr0, 0, $sgpr0, 0, $vgpr1, implicit $mode, implicit $exec, implicit $mode, implicit $exec, implicit $mode, implicit $exec
+    ; PAIR-NEXT: $vgpr4 = V_BFM_B32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
     ;
-    ; LOWER-LABEL: name: vopd_no_combine_literal_y
+    ; LOWER-LABEL: name: vopd_combine_moved_literal_y
     ; LOWER: $vgpr0 = IMPLICIT_DEF
     ; LOWER-NEXT: $vgpr1 = IMPLICIT_DEF
-    ; LOWER-NEXT: $vgpr3 = V_MUL_F32_e32 $vgpr0, $vgpr0, implicit $mode, implicit $exec
-    ; LOWER-NEXT: $vgpr4 = V_BFM_B32_e32 killed $vgpr0, $vgpr1, implicit $exec
-    ; LOWER-NEXT: $vgpr5 = V_SUB_F32_e32 12345, killed $vgpr1, implicit $mode, implicit $exec
+    ; LOWER-NEXT: $sgpr0 = S_MOV_B32 12345
+    ; LOWER-NEXT: $vgpr3, $vgpr5 = V_DUAL_MUL_F32_e32_X_SUB_F32_e32_e96_gfx1250 0, $vgpr0, 0, $vgpr0, 0, $sgpr0, 0, $vgpr1, implicit $mode, implicit $exec, implicit $mode, implicit $exec, implicit $mode, implicit $exec
+    ; LOWER-NEXT: $vgpr4 = V_BFM_B32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
     $vgpr0 = IMPLICIT_DEF
     $vgpr1 = IMPLICIT_DEF
     $vgpr3 = V_MUL_F32_e32 $vgpr0, $vgpr0, implicit $mode, implicit $exec
@@ -1228,36 +1228,110 @@ body:             |
 ...
 
 ---
-name:            vopd_no_combine_sub_u32_sub_u32_lit
+name:            vopd_combine_sub_u32_sub_u32_moved_lit
 tracksRegLiveness: true
 body:             |
   bb.0:
 
-    ; SCHED-LABEL: name: vopd_no_combine_sub_u32_sub_u32_lit
+    ; SCHED-LABEL: name: vopd_combine_sub_u32_sub_u32_moved_lit
     ; SCHED: $vgpr0 = IMPLICIT_DEF
     ; SCHED-NEXT: $vgpr1 = IMPLICIT_DEF
     ; SCHED-NEXT: $vgpr2 = IMPLICIT_DEF
     ; SCHED-NEXT: $vgpr4 = V_SUB_U32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
     ; SCHED-NEXT: $vgpr5 = V_SUB_U32_e32 300, killed $vgpr2, implicit $mode, implicit $exec
     ;
-    ; PAIR-LABEL: name: vopd_no_combine_sub_u32_sub_u32_lit
+    ; PAIR-LABEL: name: vopd_combine_sub_u32_sub_u32_moved_lit
     ; PAIR: $vgpr0 = IMPLICIT_DEF
     ; PAIR-NEXT: $vgpr1 = IMPLICIT_DEF
     ; PAIR-NEXT: $vgpr2 = IMPLICIT_DEF
+    ; PAIR-NEXT: $sgpr0 = S_MOV_B32 300
+    ; PAIR-NEXT: $vgpr4, $vgpr5 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 killed $vgpr0, killed $vgpr1, $sgpr0, killed $vgpr2, implicit $exec, implicit $exec, implicit $mode, implicit $exec
+    ;
+    ; LOWER-LABEL: name: vopd_combine_sub_u32_sub_u32_moved_lit
+    ; LOWER: $vgpr0 = IMPLICIT_DEF
+    ; LOWER-NEXT: $vgpr1 = IMPLICIT_DEF
+    ; LOWER-NEXT: $vgpr2 = IMPLICIT_DEF
+    ; LOWER-NEXT: $sgpr0 = S_MOV_B32 300
+    ; LOWER-NEXT: $vgpr4, $vgpr5 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 killed $vgpr0, killed $vgpr1, $sgpr0, killed $vgpr2, implicit $exec, implicit $exec, implicit $mode, implicit $exec
+    $vgpr0 = IMPLICIT_DEF
+    $vgpr1 = IMPLICIT_DEF
+    $vgpr2 = IMPLICIT_DEF
+    $vgpr4 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 300, $vgpr2, implicit $mode, implicit $exec
+...
+
+# Null reads nothing, so it needs no scalar bus source. The pair holds null and
+# two scalar registers, which the encoding allows.
+---
+name:            vopd_combine_null_with_two_sgprs
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $sgpr0, $sgpr1, $vgpr0, $vgpr1, $vgpr2, $vgpr3
+    ; SCHED-LABEL: name: vopd_combine_null_with_two_sgprs
+    ; SCHED: liveins: $sgpr0, $sgpr1, $vgpr0, $vgpr1, $vgpr2, $vgpr3
+    ; SCHED-NEXT: {{  $}}
+    ; SCHED-NEXT: $vgpr4 = V_CNDMASK_B32_e64 0, $sgpr_null, 0, killed $vgpr1, killed $sgpr0, implicit $exec
+    ; SCHED-NEXT: $vgpr5 = V_CNDMASK_B32_e64 0, killed $vgpr2, 0, killed $vgpr3, killed $sgpr1, implicit $exec
+    ;
+    ; PAIR-LABEL: name: vopd_combine_null_with_two_sgprs
+    ; PAIR: liveins: $sgpr0, $sgpr1, $vgpr0, $vgpr1, $vgpr2, $vgpr3
+    ; PAIR-NEXT: {{  $}}
+    ; PAIR-NEXT: $vgpr4, $vgpr5 = V_DUAL_CNDMASK_B32_e32_X_CNDMASK_B32_e32_e96_gfx1250 0, $sgpr_null, 0, killed $vgpr1, killed $sgpr0, 0, killed $vgpr2, 0, killed $vgpr3, killed $sgpr1, implicit $exec, implicit $exec, implicit $exec
+    ;
+    ; LOWER-LABEL: name: vopd_combine_null_with_two_sgprs
+    ; LOWER: liveins: $sgpr0, $sgpr1, $vgpr0, $vgpr1, $vgpr2, $vgpr3
+    ; LOWER-NEXT: {{  $}}
+    ; LOWER-NEXT: $vgpr4, $vgpr5 = V_DUAL_CNDMASK_B32_e32_X_CNDMASK_B32_e32_e96_gfx1250 0, $sgpr_null, 0, killed $vgpr1, killed $sgpr0, 0, killed $vgpr2, 0, killed $vgpr3, killed $sgpr1, implicit $exec, implicit $exec, implicit $exec
+    $vgpr4 = V_CNDMASK_B32_e64 0, $sgpr_null, 0, $vgpr1, $sgpr0, implicit $exec
+    $vgpr5 = V_CNDMASK_B32_e64 0, $vgpr2, 0, $vgpr3, $sgpr1, implicit $exec
+...
+
+# Same pair, but every SGPR is live, so the literal has nowhere to go and the
+# two instructions are left alone.
+---
+name:            vopd_no_combine_lit_without_free_sgpr
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105
+
+    ; SCHED-LABEL: name: vopd_no_combine_lit_without_free_sgpr
+    ; SCHED: liveins: $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105
+    ; SCHED-NEXT: {{  $}}
+    ; SCHED-NEXT: $vgpr0 = IMPLICIT_DEF
+    ; SCHED-NEXT: $vgpr1 = IMPLICIT_DEF
+    ; SCHED-NEXT: $vgpr2 = IMPLICIT_DEF
+    ; SCHED-NEXT: $vgpr4 = V_SUB_U32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
+    ; SCHED-NEXT: $vgpr5 = V_SUB_U32_e32 300, killed $vgpr2, implicit $mode, implicit $exec
+    ; SCHED-NEXT: S_NOP 0, implicit killed $sgpr0_sgpr1_sgpr2_sgpr3, implicit killed $sgpr4_sgpr5_sgpr6_sgpr7, implicit killed $sgpr8_sgpr9_sgpr10_sgpr11, implicit killed $sgpr12_sgpr13_sgpr14_sgpr15, implicit killed $sgpr16_sgpr17_sgpr18_sgpr19, implicit killed $sgpr20_sgpr21_sgpr22_sgpr23, implicit killed $sgpr24_sgpr25_sgpr26_sgpr27, implicit killed $sgpr28_sgpr29_sgpr30_sgpr31, implicit killed $sgpr32_sgpr33_sgpr34_sgpr35, implicit killed $sgpr36_sgpr37_sgpr38_sgpr39, implicit killed $sgpr40_sgpr41_sgpr42_sgpr43, implicit killed $sgpr44_sgpr45_sgpr46_sgpr47, implicit killed $sgpr48_sgpr49_sgpr50_sgpr51, implicit killed $sgpr52_sgpr53_sgpr54_sgpr55, implicit killed $sgpr56_sgpr57_sgpr58_sgpr59, implicit killed $sgpr60_sgpr61_sgpr62_sgpr63, implicit killed $sgpr64_sgpr65_sgpr66_sgpr67, implicit killed $sgpr68_sgpr69_sgpr70_sgpr71, implicit killed $sgpr72_sgpr73_sgpr74_sgpr75, implicit killed $sgpr76_sgpr77_sgpr78_sgpr79, implicit killed $sgpr80_sgpr81_sgpr82_sgpr83, implicit killed $sgpr84_sgpr85_sgpr86_sgpr87, implicit killed $sgpr88_sgpr89_sgpr90_sgpr91, implicit killed $sgpr92_sgpr93_sgpr94_sgpr95, implicit killed $sgpr96_sgpr97_sgpr98_sgpr99, implicit killed $sgpr100_sgpr101_sgpr102_sgpr103, implicit killed $sgpr104_sgpr105
+    ;
+    ; PAIR-LABEL: name: vopd_no_combine_lit_without_free_sgpr
+    ; PAIR: liveins: $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105
+    ; PAIR-NEXT: {{  $}}
+    ; PAIR-NEXT: $vgpr0 = IMPLICIT_DEF
+    ; PAIR-NEXT: $vgpr1 = IMPLICIT_DEF
+    ; PAIR-NEXT: $vgpr2 = IMPLICIT_DEF
     ; PAIR-NEXT: $vgpr4 = V_SUB_U32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
     ; PAIR-NEXT: $vgpr5 = V_SUB_U32_e32 300, killed $vgpr2, implicit $mode, implicit $exec
+    ; PAIR-NEXT: S_NOP 0, implicit killed $sgpr0_sgpr1_sgpr2_sgpr3, implicit killed $sgpr4_sgpr5_sgpr6_sgpr7, implicit killed $sgpr8_sgpr9_sgpr10_sgpr11, implicit killed $sgpr12_sgpr13_sgpr14_sgpr15, implicit killed $sgpr16_sgpr17_sgpr18_sgpr19, implicit killed $sgpr20_sgpr21_sgpr22_sgpr23, implicit killed $sgpr24_sgpr25_sgpr26_sgpr27, implicit killed $sgpr28_sgpr29_sgpr30_sgpr31, implicit killed $sgpr32_sgpr33_sgpr34_sgpr35, implicit killed $sgpr36_sgpr37_sgpr38_sgpr39, implicit killed $sgpr40_sgpr41_sgpr42_sgpr43, implicit killed $sgpr44_sgpr45_sgpr46_sgpr47, implicit killed $sgpr48_sgpr49_sgpr50_sgpr51, implicit killed $sgpr52_sgpr53_sgpr54_sgpr55, implicit killed $sgpr56_sgpr57_sgpr58_sgpr59, implicit killed $sgpr60_sgpr61_sgpr62_sgpr63, implicit killed $sgpr64_sgpr65_sgpr66_sgpr67, implicit killed $sgpr68_sgpr69_sgpr70_sgpr71, implicit killed $sgpr72_sgpr73_sgpr74_sgpr75, implicit killed $sgpr76_sgpr77_sgpr78_sgpr79, implicit killed $sgpr80_sgpr81_sgpr82_sgpr83, implicit killed $sgpr84_sgpr85_sgpr86_sgpr87, implicit killed $sgpr88_sgpr89_sgpr90_sgpr91, implicit killed $sgpr92_sgpr93_sgpr94_sgpr95, implicit killed $sgpr96_sgpr97_sgpr98_sgpr99, implicit killed $sgpr100_sgpr101_sgpr102_sgpr103, implicit killed $sgpr104_sgpr105
     ;
-    ; LOWER-LABEL: name: vopd_no_combine_sub_u32_sub_u32_lit
-    ; LOWER: $vgpr0 = IMPLICIT_DEF
+    ; LOWER-LABEL: name: vopd_no_combine_lit_without_free_sgpr
+    ; LOWER: liveins: $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105
+    ; LOWER-NEXT: {{  $}}
+    ; LOWER-NEXT: $vgpr0 = IMPLICIT_DEF
     ; LOWER-NEXT: $vgpr1 = IMPLICIT_DEF
     ; LOWER-NEXT: $vgpr2 = IMPLICIT_DEF
     ; LOWER-NEXT: $vgpr4 = V_SUB_U32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
     ; LOWER-NEXT: $vgpr5 = V_SUB_U32_e32 300, killed $vgpr2, implicit $mode, implicit $exec
+    ; LOWER-NEXT: S_NOP 0, implicit killed $sgpr0_sgpr1_sgpr2_sgpr3, implicit killed $sgpr4_sgpr5_sgpr6_sgpr7, implicit killed $sgpr8_sgpr9_sgpr10_sgpr11, implicit killed $sgpr12_sgpr13_sgpr14_sgpr15, implicit killed $sgpr16_sgpr17_sgpr18_sgpr19, implicit killed $sgpr20_sgpr21_sgpr22_sgpr23, implicit killed $sgpr24_sgpr25_sgpr26_sgpr27, implicit killed $sgpr28_sgpr29_sgpr30_sgpr31, implicit killed $sgpr32_sgpr33_sgpr34_sgpr35, implicit killed $sgpr36_sgpr37_sgpr38_sgpr39, implicit killed $sgpr40_sgpr41_sgpr42_sgpr43, implicit killed $sgpr44_sgpr45_sgpr46_sgpr47, implicit killed $sgpr48_sgpr49_sgpr50_sgpr51, implicit killed $sgpr52_sgpr53_sgpr54_sgpr55, implicit killed $sgpr56_sgpr57_sgpr58_sgpr59, implicit killed $sgpr60_sgpr61_sgpr62_sgpr63, implicit killed $sgpr64_sgpr65_sgpr66_sgpr67, implicit killed $sgpr68_sgpr69_sgpr70_sgpr71, implicit killed $sgpr72_sgpr73_sgpr74_sgpr75, implicit killed $sgpr76_sgpr77_sgpr78_sgpr79, implicit killed $sgpr80_sgpr81_sgpr82_sgpr83, implicit killed $sgpr84_sgpr85_sgpr86_sgpr87, implicit killed $sgpr88_sgpr89_sgpr90_sgpr91, implicit killed $sgpr92_sgpr93_sgpr94_sgpr95, implicit killed $sgpr96_sgpr97_sgpr98_sgpr99, implicit killed $sgpr100_sgpr101_sgpr102_sgpr103, implicit killed $sgpr104_sgpr105
     $vgpr0 = IMPLICIT_DEF
     $vgpr1 = IMPLICIT_DEF
     $vgpr2 = IMPLICIT_DEF
     $vgpr4 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
     $vgpr5 = V_SUB_U32_e32 300, $vgpr2, implicit $mode, implicit $exec
+    S_NOP 0, implicit $sgpr0_sgpr1_sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33_sgpr34_sgpr35, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+
 ...
 
 ---
@@ -4606,3 +4680,67 @@ body:             |
     $vgpr6 = V_FMAC_F32_e64 0, $vgpr2, 0, $vgpr3, 0, $vgpr6, 0, 0, implicit $mode, implicit $exec
     $vgpr2_vgpr3 = V_ADD_F64_pseudo_e32 10, $vgpr0_vgpr1, implicit $mode, implicit $exec
 ...
+
+# Two VOPD3 components which need the same value can be clustered. Changing
+# the second value makes the pair need two moves, so it must not be clustered.
+---
+name:            vopd_cluster_one_materialized_value
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; SCHED-LABEL: name: vopd_cluster_one_materialized_value
+    ; SCHED: liveins: $vgpr0, $vgpr1
+    ; SCHED-NEXT: {{  $}}
+    ; SCHED-NEXT: $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; SCHED-NEXT: $vgpr5 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; SCHED-NEXT: $vgpr8 = V_BFM_B32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
+    ;
+    ; PAIR-LABEL: name: vopd_cluster_one_materialized_value
+    ; PAIR: liveins: $vgpr0, $vgpr1
+    ; PAIR-NEXT: {{  $}}
+    ; PAIR-NEXT: $sgpr0 = S_MOV_B32 300
+    ; PAIR-NEXT: $vgpr4, $vgpr5 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $sgpr0, $vgpr1, $sgpr0, $vgpr1, implicit $exec, implicit $mode, implicit $exec, implicit $mode, implicit $exec
+    ; PAIR-NEXT: $vgpr8 = V_BFM_B32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
+    ;
+    ; LOWER-LABEL: name: vopd_cluster_one_materialized_value
+    ; LOWER: liveins: $vgpr0, $vgpr1
+    ; LOWER-NEXT: {{  $}}
+    ; LOWER-NEXT: $sgpr0 = S_MOV_B32 300
+    ; LOWER-NEXT: $vgpr4, $vgpr5 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $sgpr0, $vgpr1, $sgpr0, $vgpr1, implicit $exec, implicit $mode, implicit $exec, implicit $mode, implicit $exec
+    ; LOWER-NEXT: $vgpr8 = V_BFM_B32_e32 killed $vgpr0, killed $vgpr1, implicit $exec
+    $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr8 = V_BFM_B32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+...
+
+---
+name:            vopd_no_cluster_two_materialized_values
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; SCHED-LABEL: name: vopd_no_cluster_two_materialized_values
+    ; SCHED: liveins: $vgpr0, $vgpr1
+    ; SCHED-NEXT: {{  $}}
+    ; SCHED-NEXT: $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; SCHED-NEXT: $vgpr8 = V_BFM_B32_e32 killed $vgpr0, $vgpr1, implicit $exec
+    ; SCHED-NEXT: $vgpr5 = V_SUB_U32_e32 400, killed $vgpr1, implicit $mode, implicit $exec
+    ;
+    ; PAIR-LABEL: name: vopd_no_cluster_two_materialized_values
+    ; PAIR: liveins: $vgpr0, $vgpr1
+    ; PAIR-NEXT: {{  $}}
+    ; PAIR-NEXT: $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; PAIR-NEXT: $vgpr8 = V_BFM_B32_e32 killed $vgpr0, $vgpr1, implicit $exec
+    ; PAIR-NEXT: $vgpr5 = V_SUB_U32_e32 400, killed $vgpr1, implicit $mode, implicit $exec
+    ;
+    ; LOWER-LABEL: name: vopd_no_cluster_two_materialized_values
+    ; LOWER: liveins: $vgpr0, $vgpr1
+    ; LOWER-NEXT: {{  $}}
+    ; LOWER-NEXT: $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; LOWER-NEXT: $vgpr8 = V_BFM_B32_e32 killed $vgpr0, $vgpr1, implicit $exec
+    ; LOWER-NEXT: $vgpr5 = V_SUB_U32_e32 400, killed $vgpr1, implicit $mode, implicit $exec
+    $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr8 = V_BFM_B32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 400, $vgpr1, implicit $mode, implicit $exec
+...
diff --git a/llvm/test/CodeGen/AMDGPU/vopd-combine.mir b/llvm/test/CodeGen/AMDGPU/vopd-combine.mir
index e0c558c95a89f..c813d4126dcae 100644
--- a/llvm/test/CodeGen/AMDGPU/vopd-combine.mir
+++ b/llvm/test/CodeGen/AMDGPU/vopd-combine.mir
@@ -21,6 +21,7 @@
   define void @vopd_constants_same() { ret void }
   define void @vopd_mov_fmaak_constants_same() { ret void }
   define void @vopd_debug() { ret void }
+  define void @vopd_debug_at_search_position() { ret void }
   define void @vopd_schedule_unconstrained() { ret void }
   define void @vopd_schedule_unconstrained_2() { ret void }
   define void @vopd_mov_fixup() { ret void }
@@ -514,6 +515,56 @@ body:             |
 
 ...
 
+# The search skips debug instructions, but it can still land on one: at the
+# start of a block, or right after a pair it has just matched.
+
+---
+name:            vopd_debug_at_search_position
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+
+    ; SCHED-LABEL: name: vopd_debug_at_search_position
+    ; SCHED: liveins: $vgpr0, $vgpr1
+    ; SCHED-NEXT: {{  $}}
+    ; SCHED-NEXT: DBG_VALUE $vgpr0, 0, 0
+    ; SCHED-NEXT: $vgpr3 = V_SUB_F32_e32 killed $vgpr1, $vgpr1, implicit $mode, implicit $exec
+    ; SCHED-NEXT: $vgpr6 = V_MUL_F32_e32 killed $vgpr0, $vgpr0, implicit $mode, implicit $exec
+    ; SCHED-NEXT: DBG_VALUE $vgpr1, 0, 0
+    ; SCHED-NEXT: S_ENDPGM 0
+    ;
+    ; PAIR-GFX1100-LABEL: name: vopd_debug_at_search_position
+    ; PAIR-GFX1100: liveins: $vgpr0, $vgpr1
+    ; PAIR-GFX1100-NEXT: {{  $}}
+    ; PAIR-GFX1100-NEXT: DBG_VALUE $vgpr0, 0, 0
+    ; PAIR-GFX1100-NEXT: $vgpr3, $vgpr6 = V_DUAL_SUB_F32_e32_X_MUL_F32_e32_gfx11 killed $vgpr1, $vgpr1, killed $vgpr0, $vgpr0, implicit $mode, implicit $exec, implicit $mode, implicit $exec, implicit $mode, implicit $exec
+    ; PAIR-GFX1100-NEXT: DBG_VALUE $vgpr1, 0, 0
+    ; PAIR-GFX1100-NEXT: S_ENDPGM 0
+    ;
+    ; PAIR-GFX1170-LABEL: name: vopd_debug_at_search_position
+    ; PAIR-GFX1170: liveins: $vgpr0, $vgpr1
+    ; PAIR-GFX1170-NEXT: {{  $}}
+    ; PAIR-GFX1170-NEXT: DBG_VALUE $vgpr0, 0, 0
+    ; PAIR-GFX1170-NEXT: $vgpr3, $vgpr6 = V_DUAL_SUB_F32_e32_X_MUL_F32_e32_gfx1170 killed $vgpr1, $vgpr1, killed $vgpr0, $vgpr0, implicit $mode, implicit $exec, implicit $mode, implicit $exec, implicit $mode, implicit $exec
+    ; PAIR-GFX1170-NEXT: DBG_VALUE $vgpr1, 0, 0
+    ; PAIR-GFX1170-NEXT: S_ENDPGM 0
+    ;
+    ; PAIR-GFX12-LABEL: name: vopd_debug_at_search_position
+    ; PAIR-GFX12: liveins: $vgpr0, $vgpr1
+    ; PAIR-GFX12-NEXT: {{  $}}
+    ; PAIR-GFX12-NEXT: DBG_VALUE $vgpr0, 0, 0
+    ; PAIR-GFX12-NEXT: $vgpr3, $vgpr6 = V_DUAL_SUB_F32_e32_X_MUL_F32_e32_gfx12 killed $vgpr1, $vgpr1, killed $vgpr0, $vgpr0, implicit $mode, implicit $exec, implicit $mode, implicit $exec, implicit $mode, implicit $exec
+    ; PAIR-GFX12-NEXT: DBG_VALUE $vgpr1, 0, 0
+    ; PAIR-GFX12-NEXT: S_ENDPGM 0
+    DBG_VALUE $vgpr0, 0, 0
+    $vgpr3 = V_SUB_F32_e32 $vgpr1, $vgpr1, implicit $mode, implicit $exec
+    $vgpr6 = V_MUL_F32_e32 $vgpr0, $vgpr0, implicit $mode, implicit $exec
+    DBG_VALUE $vgpr1, 0, 0
+    S_ENDPGM 0
+
+...
+
 ---
 name:            vopd_schedule_unconstrained
 tracksRegLiveness: true
diff --git a/llvm/test/CodeGen/AMDGPU/vopd3-imm-fold.ll b/llvm/test/CodeGen/AMDGPU/vopd3-imm-fold.ll
new file mode 100644
index 0000000000000..ebee18351e99d
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/vopd3-imm-fold.ll
@@ -0,0 +1,373 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgpu12.50-amd-amdhsa -verify-machineinstrs < %s | FileCheck -check-prefix=GFX1250 %s
+; RUN: llc -mtriple=amdgpu12.00-amd-amdhsa -verify-machineinstrs < %s | FileCheck -check-prefix=GFX1200 %s
+
+; A VOPD3 component cannot encode a 32-bit literal, and on GFX1250 v_and_b32 can
+; only be dual issued through VOPD3 as v_dual_bitop2_b32. A folded literal
+; therefore costs one issue slot per use.
+;
+; The pair is built after register allocation, so the literal is folded first
+; and moved back into a scalar register only where that buys a pair. Each pair
+; gets its own short-lived register.
+;
+; GFX1200 has no VOPD3, so the immediate stays folded there.
+
+declare i32 @llvm.amdgcn.workitem.id.x()
+
+; Two v_and_b32 operations use the same mask. Each pair gets one S_MOV_B32.
+define amdgpu_kernel void @mask_shared_by_two_pairs(ptr addrspace(1) %out, ptr addrspace(1) %in) {
+; GFX1250-LABEL: mask_shared_by_two_pairs:
+; GFX1250:       ; %bb.0: ; %entry
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT:    v_and_b32_e32 v4, 0x3ff, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 2, v4
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    global_load_b64 v[2:3], v0, s[2:3]
+; GFX1250-NEXT:    s_wait_xcnt 0x0
+; GFX1250-NEXT:    s_mov_b32 s2, 0xffff0000
+; GFX1250-NEXT:    s_wait_loadcnt 0x0
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, s2, v2 bitop3:0x40
+; GFX1250-NEXT:    s_mov_b32 s2, 0xffff0000
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v3 :: v_dual_bitop2_b32 v3, s2, v3 bitop3:0x40
+; GFX1250-NEXT:    global_store_b128 v4, v[0:3], s[0:1] scale_offset
+; GFX1250-NEXT:    s_endpgm
+;
+; GFX1200-LABEL: mask_shared_by_two_pairs:
+; GFX1200:       ; %bb.0: ; %entry
+; GFX1200-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0
+; GFX1200-NEXT:    v_and_b32_e32 v4, 0x3ff, v0
+; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v0, 2, v4
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v4, 4, v4
+; GFX1200-NEXT:    s_wait_kmcnt 0x0
+; GFX1200-NEXT:    global_load_b64 v[2:3], v0, s[2:3]
+; GFX1200-NEXT:    s_wait_loadcnt 0x0
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
+; GFX1200-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v2, 16, v3
+; GFX1200-NEXT:    v_and_b32_e32 v3, 0xffff0000, v3
+; GFX1200-NEXT:    global_store_b128 v4, v[0:3], s[0:1]
+; GFX1200-NEXT:    s_endpgm
+entry:
+  %id = call i32 @llvm.amdgcn.workitem.id.x()
+  %in.gep0 = getelementptr i32, ptr addrspace(1) %in, i32 %id
+  %id1 = add i32 %id, 1
+  %in.gep1 = getelementptr i32, ptr addrspace(1) %in, i32 %id1
+  %data0 = load i32, ptr addrspace(1) %in.gep0
+  %data1 = load i32, ptr addrspace(1) %in.gep1
+  %lo0 = shl i32 %data0, 16
+  %hi0 = and i32 %data0, -65536
+  %lo1 = shl i32 %data1, 16
+  %hi1 = and i32 %data1, -65536
+  %ins0 = insertelement <4 x i32> poison, i32 %lo0, i32 0
+  %ins1 = insertelement <4 x i32> %ins0, i32 %hi0, i32 1
+  %ins2 = insertelement <4 x i32> %ins1, i32 %lo1, i32 2
+  %ins3 = insertelement <4 x i32> %ins2, i32 %hi1, i32 3
+  %out.gep = getelementptr <4 x i32>, ptr addrspace(1) %out, i32 %id
+  store <4 x i32> %ins3, ptr addrspace(1) %out.gep
+  ret void
+}
+
+; The pair is inside a loop. The move stays next to it, so it runs once per
+; iteration: one scalar instruction bought for one dual-issue slot.
+define amdgpu_kernel void @mask_in_loop_body(ptr addrspace(1) %out, ptr addrspace(1) %in, i32 %n) {
+; GFX1250-LABEL: mask_in_loop_body:
+; GFX1250:       ; %bb.0: ; %entry
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT:    s_clause 0x1
+; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT:    s_load_b32 s6, s[4:5], 0x10 nv
+; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v2, v0
+; GFX1250-NEXT:  .LBB1_1: ; %loop
+; GFX1250-NEXT:    ; =>This Inner Loop Header: Depth=1
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    global_load_b32 v3, v2, s[2:3] scale_offset
+; GFX1250-NEXT:    s_wait_xcnt 0x0
+; GFX1250-NEXT:    v_add_nc_u32_e32 v2, 1, v2
+; GFX1250-NEXT:    s_add_co_i32 s6, s6, -1
+; GFX1250-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_3) | instid1(VALU_DEP_1)
+; GFX1250-NEXT:    s_cmp_lg_u32 s6, 0
+; GFX1250-NEXT:    s_mov_b32 s4, 0xffff0000
+; GFX1250-NEXT:    s_wait_loadcnt 0x0
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v4, 16, v3 :: v_dual_bitop2_b32 v3, s4, v3 bitop3:0x40
+; GFX1250-NEXT:    v_add3_u32 v1, v4, v3, v1
+; GFX1250-NEXT:    s_cbranch_scc1 .LBB1_1
+; GFX1250-NEXT:  ; %bb.2: ; %exit
+; GFX1250-NEXT:    global_store_b32 v0, v1, s[0:1] scale_offset
+; GFX1250-NEXT:    s_endpgm
+;
+; GFX1200-LABEL: mask_in_loop_body:
+; GFX1200:       ; %bb.0: ; %entry
+; GFX1200-NEXT:    s_clause 0x1
+; GFX1200-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0
+; GFX1200-NEXT:    s_load_b32 s4, s[4:5], 0x10
+; GFX1200-NEXT:    v_dual_mov_b32 v3, 0 :: v_dual_and_b32 v2, 0x3ff, v0
+; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1200-NEXT:    v_mov_b32_e32 v0, v2
+; GFX1200-NEXT:  .LBB1_1: ; %loop
+; GFX1200-NEXT:    ; =>This Inner Loop Header: Depth=1
+; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_2) | instid1(SALU_CYCLE_1)
+; GFX1200-NEXT:    v_ashrrev_i32_e32 v1, 31, v0
+; GFX1200-NEXT:    s_wait_kmcnt 0x0
+; GFX1200-NEXT:    s_add_co_i32 s4, s4, -1
+; GFX1200-NEXT:    s_cmp_lg_u32 s4, 0
+; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
+; GFX1200-NEXT:    v_lshlrev_b64_e32 v[4:5], 2, v[0:1]
+; GFX1200-NEXT:    v_add_nc_u32_e32 v0, 1, v0
+; GFX1200-NEXT:    v_add_co_u32 v4, vcc_lo, s2, v4
+; GFX1200-NEXT:    s_wait_alu depctr_va_vcc(0)
+; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1200-NEXT:    v_add_co_ci_u32_e64 v5, null, s3, v5, vcc_lo
+; GFX1200-NEXT:    global_load_b32 v1, v[4:5], off
+; GFX1200-NEXT:    s_wait_loadcnt 0x0
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v4, 16, v1
+; GFX1200-NEXT:    v_and_b32_e32 v1, 0xffff0000, v1
+; GFX1200-NEXT:    v_add3_u32 v3, v4, v1, v3
+; GFX1200-NEXT:    s_cbranch_scc1 .LBB1_1
+; GFX1200-NEXT:  ; %bb.2: ; %exit
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v0, 2, v2
+; GFX1200-NEXT:    global_store_b32 v0, v3, s[0:1]
+; GFX1200-NEXT:    s_endpgm
+entry:
+  %id = call i32 @llvm.amdgcn.workitem.id.x()
+  br label %loop
+
+loop:
+  %i = phi i32 [ 0, %entry ], [ %i.next, %loop ]
+  %acc = phi i32 [ 0, %entry ], [ %acc.next, %loop ]
+  %idx = add i32 %i, %id
+  %in.gep = getelementptr i32, ptr addrspace(1) %in, i32 %idx
+  %data = load i32, ptr addrspace(1) %in.gep
+  %lo = shl i32 %data, 16
+  %hi = and i32 %data, -65536
+  %sum = add i32 %lo, %hi
+  %acc.next = add i32 %acc, %sum
+  %i.next = add i32 %i, 1
+  %cmp = icmp eq i32 %i.next, %n
+  br i1 %cmp, label %exit, label %loop
+
+exit:
+  %out.gep = getelementptr i32, ptr addrspace(1) %out, i32 %id
+  store i32 %acc.next, ptr addrspace(1) %out.gep
+  ret void
+}
+
+; v_lshl_add_u32 is not a VOPD component, so the v_and_b32 has no partner. No
+; pair is built, and the literal keeps the encoding it was folded into.
+define amdgpu_kernel void @mask_without_pair(ptr addrspace(1) %out, ptr addrspace(1) %in) {
+; GFX1250-LABEL: mask_without_pair:
+; GFX1250:       ; %bb.0: ; %entry
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    global_load_b32 v1, v0, s[2:3] scale_offset
+; GFX1250-NEXT:    s_wait_loadcnt 0x0
+; GFX1250-NEXT:    v_and_b32_e32 v2, 0xffff0000, v1
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT:    v_lshl_add_u32 v1, v1, 16, v2
+; GFX1250-NEXT:    global_store_b32 v0, v1, s[0:1] scale_offset
+; GFX1250-NEXT:    s_endpgm
+;
+; GFX1200-LABEL: mask_without_pair:
+; GFX1200:       ; %bb.0: ; %entry
+; GFX1200-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0
+; GFX1200-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_4) | instid1(VALU_DEP_1)
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v0, 2, v0
+; GFX1200-NEXT:    s_wait_kmcnt 0x0
+; GFX1200-NEXT:    global_load_b32 v1, v0, s[2:3]
+; GFX1200-NEXT:    s_wait_loadcnt 0x0
+; GFX1200-NEXT:    v_and_b32_e32 v2, 0xffff0000, v1
+; GFX1200-NEXT:    v_lshl_add_u32 v1, v1, 16, v2
+; GFX1200-NEXT:    global_store_b32 v0, v1, s[0:1]
+; GFX1200-NEXT:    s_endpgm
+entry:
+  %id = call i32 @llvm.amdgcn.workitem.id.x()
+  %in.gep = getelementptr i32, ptr addrspace(1) %in, i32 %id
+  %data = load i32, ptr addrspace(1) %in.gep
+  %lo = shl i32 %data, 16
+  %hi = and i32 %data, -65536
+  %res = add i32 %lo, %hi
+  %out.gep = getelementptr i32, ptr addrspace(1) %out, i32 %id
+  store i32 %res, ptr addrspace(1) %out.gep
+  ret void
+}
+
+; An inline constant is legal in VOPD3, so the pair forms without a move.
+define amdgpu_kernel void @mask_inline_constant(ptr addrspace(1) %out, ptr addrspace(1) %in) {
+; GFX1250-LABEL: mask_inline_constant:
+; GFX1250:       ; %bb.0: ; %entry
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT:    v_and_b32_e32 v4, 0x3ff, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 2, v4
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    global_load_b64 v[2:3], v0, s[2:3]
+; GFX1250-NEXT:    s_wait_loadcnt 0x0
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v0, 16, v2 :: v_dual_bitop2_b32 v1, 63, v2 bitop3:0x40
+; GFX1250-NEXT:    v_dual_lshlrev_b32 v2, 16, v3 :: v_dual_bitop2_b32 v3, 63, v3 bitop3:0x40
+; GFX1250-NEXT:    global_store_b128 v4, v[0:3], s[0:1] scale_offset
+; GFX1250-NEXT:    s_endpgm
+;
+; GFX1200-LABEL: mask_inline_constant:
+; GFX1200:       ; %bb.0: ; %entry
+; GFX1200-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0
+; GFX1200-NEXT:    v_and_b32_e32 v4, 0x3ff, v0
+; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v0, 2, v4
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v4, 4, v4
+; GFX1200-NEXT:    s_wait_kmcnt 0x0
+; GFX1200-NEXT:    global_load_b64 v[2:3], v0, s[2:3]
+; GFX1200-NEXT:    s_wait_loadcnt 0x0
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
+; GFX1200-NEXT:    v_and_b32_e32 v1, 63, v2
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v2, 16, v3
+; GFX1200-NEXT:    v_and_b32_e32 v3, 63, v3
+; GFX1200-NEXT:    global_store_b128 v4, v[0:3], s[0:1]
+; GFX1200-NEXT:    s_endpgm
+entry:
+  %id = call i32 @llvm.amdgcn.workitem.id.x()
+  %in.gep0 = getelementptr i32, ptr addrspace(1) %in, i32 %id
+  %id1 = add i32 %id, 1
+  %in.gep1 = getelementptr i32, ptr addrspace(1) %in, i32 %id1
+  %data0 = load i32, ptr addrspace(1) %in.gep0
+  %data1 = load i32, ptr addrspace(1) %in.gep1
+  %lo0 = shl i32 %data0, 16
+  %hi0 = and i32 %data0, 63
+  %lo1 = shl i32 %data1, 16
+  %hi1 = and i32 %data1, 63
+  %ins0 = insertelement <4 x i32> poison, i32 %lo0, i32 0
+  %ins1 = insertelement <4 x i32> %ins0, i32 %hi0, i32 1
+  %ins2 = insertelement <4 x i32> %ins1, i32 %lo1, i32 2
+  %ins3 = insertelement <4 x i32> %ins2, i32 %hi1, i32 3
+  %out.gep = getelementptr <4 x i32>, ptr addrspace(1) %out, i32 %id
+  store <4 x i32> %ins3, ptr addrspace(1) %out.gep
+  ret void
+}
+
+; Same shape as mask_shared_by_two_pairs, but the function asked for small
+; code. A move only makes the code longer, so the literals stay folded.
+define amdgpu_kernel void @mask_optsize(ptr addrspace(1) %out, ptr addrspace(1) %in) optsize {
+; GFX1250-LABEL: mask_optsize:
+; GFX1250:       ; %bb.0: ; %entry
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT:    v_and_b32_e32 v4, 0x3ff, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 2, v4
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    global_load_b64 v[2:3], v0, s[2:3]
+; GFX1250-NEXT:    s_wait_loadcnt 0x0
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
+; GFX1250-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v2, 16, v3
+; GFX1250-NEXT:    v_and_b32_e32 v3, 0xffff0000, v3
+; GFX1250-NEXT:    global_store_b128 v4, v[0:3], s[0:1] scale_offset
+; GFX1250-NEXT:    s_endpgm
+;
+; GFX1200-LABEL: mask_optsize:
+; GFX1200:       ; %bb.0: ; %entry
+; GFX1200-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0
+; GFX1200-NEXT:    v_and_b32_e32 v4, 0x3ff, v0
+; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v0, 2, v4
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v4, 4, v4
+; GFX1200-NEXT:    s_wait_kmcnt 0x0
+; GFX1200-NEXT:    global_load_b64 v[2:3], v0, s[2:3]
+; GFX1200-NEXT:    s_wait_loadcnt 0x0
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v0, 16, v2
+; GFX1200-NEXT:    v_and_b32_e32 v1, 0xffff0000, v2
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v2, 16, v3
+; GFX1200-NEXT:    v_and_b32_e32 v3, 0xffff0000, v3
+; GFX1200-NEXT:    global_store_b128 v4, v[0:3], s[0:1]
+; GFX1200-NEXT:    s_endpgm
+entry:
+  %id = call i32 @llvm.amdgcn.workitem.id.x()
+  %in.gep0 = getelementptr i32, ptr addrspace(1) %in, i32 %id
+  %id1 = add i32 %id, 1
+  %in.gep1 = getelementptr i32, ptr addrspace(1) %in, i32 %id1
+  %data0 = load i32, ptr addrspace(1) %in.gep0
+  %data1 = load i32, ptr addrspace(1) %in.gep1
+  %lo0 = shl i32 %data0, 16
+  %hi0 = and i32 %data0, -65536
+  %lo1 = shl i32 %data1, 16
+  %hi1 = and i32 %data1, -65536
+  %ins0 = insertelement <4 x i32> poison, i32 %lo0, i32 0
+  %ins1 = insertelement <4 x i32> %ins0, i32 %hi0, i32 1
+  %ins2 = insertelement <4 x i32> %ins1, i32 %lo1, i32 2
+  %ins3 = insertelement <4 x i32> %ins2, i32 %hi1, i32 3
+  %out.gep = getelementptr <4 x i32>, ptr addrspace(1) %out, i32 %id
+  store <4 x i32> %ins3, ptr addrspace(1) %out.gep
+  ret void
+}
+
+; v_add_f32 is a plain VOPD component, which may carry a literal, so the pair
+; forms with the literal in place.
+define amdgpu_kernel void @plain_vopd_component(ptr addrspace(1) %out, ptr addrspace(1) %in) {
+; GFX1250-LABEL: plain_vopd_component:
+; GFX1250:       ; %bb.0: ; %entry
+; GFX1250-NEXT:    s_setreg_imm32_b32 hwreg(HW_REG_WAVE_MODE, 25, 1), 1 ; msbs: dst=0 src0=0 src1=0 src2=0
+; GFX1250-NEXT:    s_mov_b64 s[64:65], 0
+; GFX1250-NEXT:    v_nop
+; GFX1250-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
+; GFX1250-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX1250-NEXT:    v_and_b32_e32 v2, 0x3ff, v0
+; GFX1250-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-NEXT:    v_lshlrev_b32_e32 v0, 2, v2
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    global_load_b64 v[0:1], v0, s[2:3]
+; GFX1250-NEXT:    s_wait_loadcnt 0x0
+; GFX1250-NEXT:    v_dual_add_f32 v0, 0x3dcccccd, v0 :: v_dual_add_f32 v1, 0x3dcccccd, v1
+; GFX1250-NEXT:    global_store_b64 v2, v[0:1], s[0:1] scale_offset
+; GFX1250-NEXT:    s_endpgm
+;
+; GFX1200-LABEL: plain_vopd_component:
+; GFX1200:       ; %bb.0: ; %entry
+; GFX1200-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0
+; GFX1200-NEXT:    v_and_b32_e32 v2, 0x3ff, v0
+; GFX1200-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v0, 2, v2
+; GFX1200-NEXT:    v_lshlrev_b32_e32 v2, 3, v2
+; GFX1200-NEXT:    s_wait_kmcnt 0x0
+; GFX1200-NEXT:    global_load_b64 v[0:1], v0, s[2:3]
+; GFX1200-NEXT:    s_wait_loadcnt 0x0
+; GFX1200-NEXT:    v_dual_add_f32 v0, 0x3dcccccd, v0 :: v_dual_add_f32 v1, 0x3dcccccd, v1
+; GFX1200-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
+; GFX1200-NEXT:    s_endpgm
+entry:
+  %id = call i32 @llvm.amdgcn.workitem.id.x()
+  %in.gep0 = getelementptr float, ptr addrspace(1) %in, i32 %id
+  %id1 = add i32 %id, 1
+  %in.gep1 = getelementptr float, ptr addrspace(1) %in, i32 %id1
+  %data0 = load float, ptr addrspace(1) %in.gep0
+  %data1 = load float, ptr addrspace(1) %in.gep1
+  %add0 = fadd float %data0, 0x3FB99999A0000000
+  %add1 = fadd float %data1, 0x3FB99999A0000000
+  %ins0 = insertelement <2 x float> poison, float %add0, i32 0
+  %ins1 = insertelement <2 x float> %ins0, float %add1, i32 1
+  %out.gep = getelementptr <2 x float>, ptr addrspace(1) %out, i32 %id
+  store <2 x float> %ins1, ptr addrspace(1) %out.gep
+  ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/vopd3-imm-fold.mir b/llvm/test/CodeGen/AMDGPU/vopd3-imm-fold.mir
new file mode 100644
index 0000000000000..001258bf6e62c
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/vopd3-imm-fold.mir
@@ -0,0 +1,433 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6
+# RUN: llc -mtriple=amdgpu12.50 -run-pass=gcn-create-vopd -verify-machineinstrs %s -o - | FileCheck %s
+
+# A VOPD3 component cannot encode a literal, so the pass moves the value into a
+# free scalar register. The move goes next to the pair it serves. These tests
+# cover where it lands and when the pair has to be given up instead.
+
+# The pair is in a loop. The move goes with it, inside the loop body, so it
+# needs no register live across the back edge.
+---
+name:            move_stays_in_loop_body
+tracksRegLiveness: true
+body:             |
+  ; CHECK-LABEL: name: move_stays_in_loop_body
+  ; CHECK: bb.0:
+  ; CHECK-NEXT:   successors: %bb.1(0x80000000)
+  ; CHECK-NEXT:   liveins: $vgpr0, $vgpr1
+  ; CHECK-NEXT: {{  $}}
+  ; CHECK-NEXT:   S_BRANCH %bb.1
+  ; CHECK-NEXT: {{  $}}
+  ; CHECK-NEXT: bb.1:
+  ; CHECK-NEXT:   successors: %bb.1(0x40000000), %bb.2(0x40000000)
+  ; CHECK-NEXT:   liveins: $vgpr0, $vgpr1
+  ; CHECK-NEXT: {{  $}}
+  ; CHECK-NEXT:   $sgpr0 = S_MOV_B32 300
+  ; CHECK-NEXT:   $vgpr4, $vgpr5 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $vgpr0, $vgpr1, $sgpr0, $vgpr1, implicit $exec, implicit $exec, implicit $mode, implicit $exec
+  ; CHECK-NEXT:   S_CBRANCH_SCC1 %bb.1, implicit undef $scc
+  ; CHECK-NEXT:   S_BRANCH %bb.2
+  ; CHECK-NEXT: {{  $}}
+  ; CHECK-NEXT: bb.2:
+  ; CHECK-NEXT:   S_ENDPGM 0
+  bb.0:
+    successors: %bb.1
+    liveins: $vgpr0, $vgpr1
+    S_BRANCH %bb.1
+
+  bb.1:
+    successors: %bb.1, %bb.2
+    liveins: $vgpr0, $vgpr1
+    $vgpr4 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    S_CBRANCH_SCC1 %bb.1, implicit undef $scc
+    S_BRANCH %bb.2
+
+  bb.2:
+    S_ENDPGM 0
+...
+
+# An unreachable block needs no special treatment: the move goes next to its
+# pair, in that same block.
+---
+name:            unreachable_use
+tracksRegLiveness: true
+body:             |
+  ; CHECK-LABEL: name: unreachable_use
+  ; CHECK: bb.0:
+  ; CHECK-NEXT:   liveins: $vgpr0, $vgpr1
+  ; CHECK-NEXT: {{  $}}
+  ; CHECK-NEXT:   S_ENDPGM 0
+  ; CHECK-NEXT: {{  $}}
+  ; CHECK-NEXT: bb.1:
+  ; CHECK-NEXT:   liveins: $vgpr0, $vgpr1
+  ; CHECK-NEXT: {{  $}}
+  ; CHECK-NEXT:   $sgpr0 = S_MOV_B32 300
+  ; CHECK-NEXT:   $vgpr4, $vgpr5 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $vgpr0, $vgpr1, $sgpr0, $vgpr1, implicit $exec, implicit $exec, implicit $mode, implicit $exec
+  ; CHECK-NEXT:   S_ENDPGM 0
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    S_ENDPGM 0
+
+  bb.1:
+    liveins: $vgpr0, $vgpr1
+    $vgpr4 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    S_ENDPGM 0
+...
+
+# The first two instructions match, but the pair would need a move for each of
+# its two literals, which is more than it saves. Giving it up has to release the
+# second instruction, which pairs with the third one for a single move.
+---
+name:            rejected_pair_releases_second_instruction
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: rejected_pair_releases_second_instruction
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $sgpr0 = S_MOV_B32 400
+    ; CHECK-NEXT: $vgpr5, $vgpr6 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $sgpr0, $vgpr1, $vgpr0, $vgpr1, implicit $exec, implicit $mode, implicit $exec, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 400, $vgpr1, implicit $mode, implicit $exec
+    $vgpr6 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    S_ENDPGM 0
+...
+
+# Both overlapping edges form one pair, so prefer the edge which needs no
+# scalar move.
+---
+name:            prefer_pair_without_materialization
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: prefer_pair_without_materialization
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $vgpr5, $vgpr6 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $vgpr0, $vgpr1, $vgpr0, $vgpr1, implicit $exec, implicit $exec, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr6 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    S_ENDPGM 0
+...
+
+# Maximizing the number of pairs takes priority over minimizing moves. Pair
+# the first and second instructions and the third and fourth instructions.
+---
+name:            maximize_pairs_before_minimizing_moves
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: maximize_pairs_before_minimizing_moves
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $sgpr0 = S_MOV_B32 300
+    ; CHECK-NEXT: $vgpr4, $vgpr5 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $sgpr0, $vgpr1, $vgpr0, $vgpr1, implicit $exec, implicit $mode, implicit $exec, implicit $exec
+    ; CHECK-NEXT: $sgpr0 = S_MOV_B32 400
+    ; CHECK-NEXT: $vgpr6, $vgpr7 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $vgpr0, $vgpr1, $sgpr0, $vgpr1, implicit $exec, implicit $exec, implicit $mode, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr6 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr7 = V_SUB_U32_e32 400, $vgpr1, implicit $mode, implicit $exec
+    S_ENDPGM 0
+...
+
+# The first edge cannot use $sgpr0 because that edge reads it, and every other
+# scalar register is live. The overlapping second edge can use $sgpr0.
+---
+name:            no_free_register_edge_does_not_hide_next
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $sgpr0, $sgpr1, $sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105, $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: no_free_register_edge_does_not_hide_next
+    ; CHECK: liveins: $sgpr0, $sgpr1, $sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105, $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $vgpr4 = V_SUB_U32_e32 $sgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: $sgpr0 = S_MOV_B32 300
+    ; CHECK-NEXT: $vgpr5, $vgpr6 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $sgpr0, $vgpr1, $vgpr0, $vgpr1, implicit $exec, implicit $mode, implicit $exec, implicit $exec
+    ; CHECK-NEXT: S_NOP 0, implicit $sgpr1, implicit $sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33_sgpr34_sgpr35, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4 = V_SUB_U32_e32 $sgpr0, $vgpr1, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr6 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    S_NOP 0, implicit $sgpr1, implicit $sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33_sgpr34_sgpr35, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    S_ENDPGM 0
+...
+
+# Every adjacent edge needs two distinct immediate values, so none is eligible.
+---
+name:            every_edge_needs_two_values
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105, $vgpr1
+    ; CHECK-LABEL: name: every_edge_needs_two_values
+    ; CHECK: liveins: $sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $vgpr5 = V_SUB_U32_e32 400, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $vgpr6 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $vgpr7 = V_SUB_U32_e32 500, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: S_NOP 0, implicit $sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33_sgpr34_sgpr35, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 400, $vgpr1, implicit $mode, implicit $exec
+    $vgpr6 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr7 = V_SUB_U32_e32 500, $vgpr1, implicit $mode, implicit $exec
+    S_NOP 0, implicit $sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33_sgpr34_sgpr35, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    S_ENDPGM 0
+...
+
+# The first two edges need two values and are rejected. The last edge needs one
+# value and gets a register.
+---
+name:            two_value_edges_do_not_block_one_value_edge
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105, $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: two_value_edges_do_not_block_one_value_edge
+    ; CHECK: liveins: $sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105, $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $vgpr5 = V_SUB_U32_e32 400, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $sgpr0 = S_MOV_B32 500
+    ; CHECK-NEXT: $vgpr6, $vgpr7 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $sgpr0, $vgpr1, $vgpr0, $vgpr1, implicit $exec, implicit $mode, implicit $exec, implicit $exec
+    ; CHECK-NEXT: S_NOP 0, implicit $sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33_sgpr34_sgpr35, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 400, $vgpr1, implicit $mode, implicit $exec
+    $vgpr6 = V_SUB_U32_e32 500, $vgpr1, implicit $mode, implicit $exec
+    $vgpr7 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    S_NOP 0, implicit $sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33_sgpr34_sgpr35, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    S_ENDPGM 0
+...
+
+# Every edge before the last needs two distinct values and is rejected. The
+# final edge needs one value and forms.
+---
+name:            only_one_value_edge_is_eligible
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104_sgpr105, $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: only_one_value_edge_is_eligible
+    ; CHECK: liveins: $sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34_sgpr35, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104_sgpr105, $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $vgpr5 = V_SUB_U32_e32 500, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $vgpr6 = V_SUB_U32_e32 400, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $vgpr7 = V_SUB_U32_e32 500, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $sgpr0 = S_MOV_B32 600
+    ; CHECK-NEXT: $vgpr8, $vgpr9 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $sgpr0, $vgpr1, $vgpr0, $vgpr1, implicit $exec, implicit $mode, implicit $exec, implicit $exec
+    ; CHECK-NEXT: S_NOP 0, implicit $sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33_sgpr34_sgpr35, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 500, $vgpr1, implicit $mode, implicit $exec
+    $vgpr6 = V_SUB_U32_e32 400, $vgpr1, implicit $mode, implicit $exec
+    $vgpr7 = V_SUB_U32_e32 500, $vgpr1, implicit $mode, implicit $exec
+    $vgpr8 = V_SUB_U32_e32 600, $vgpr1, implicit $mode, implicit $exec
+    $vgpr9 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    S_NOP 0, implicit $sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33_sgpr34_sgpr35, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    S_ENDPGM 0
+...
+
+# The two components name the same 32 bits, written as a negative and as an
+# unsigned value. They share one move.
+---
+name:            same_value_written_two_ways
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: same_value_written_two_ways
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $sgpr0 = S_MOV_B32 -2048
+    ; CHECK-NEXT: $vgpr4, $vgpr5 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $sgpr0, $vgpr1, $sgpr0, $vgpr1, implicit $exec, implicit $mode, implicit $exec, implicit $mode, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4 = V_SUB_U32_e32 -2048, $vgpr1, implicit $mode, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 4294965248, $vgpr1, implicit $mode, implicit $exec
+    S_ENDPGM 0
+...
+
+# A pair which needs two different scalar moves is rejected.
+---
+name:            two_distinct_values_are_not_materialized
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr1
+    ; CHECK-LABEL: name: two_distinct_values_are_not_materialized
+    ; CHECK: liveins: $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $vgpr5 = V_SUB_U32_e32 500, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 500, $vgpr1, implicit $mode, implicit $exec
+    S_ENDPGM 0
+...
+
+# Disjoint pair-local ranges can reuse the same scalar register.
+---
+name:            disjoint_pairs_reuse_register
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: disjoint_pairs_reuse_register
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $sgpr0 = S_MOV_B32 100
+    ; CHECK-NEXT: $vgpr4, $vgpr5 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $sgpr0, $vgpr1, $vgpr0, $vgpr1, implicit $exec, implicit $mode, implicit $exec, implicit $exec
+    ; CHECK-NEXT: $sgpr0 = S_MOV_B32 200
+    ; CHECK-NEXT: $vgpr6, $vgpr7 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $sgpr0, $vgpr1, $vgpr0, $vgpr1, implicit $exec, implicit $mode, implicit $exec, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4 = V_SUB_U32_e32 100, $vgpr1, implicit $mode, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr6 = V_SUB_U32_e32 200, $vgpr1, implicit $mode, implicit $exec
+    $vgpr7 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    S_ENDPGM 0
+...
+
+# Program order is Y then X. The move still goes before the Y instruction,
+# while the VOPD operands and destinations use encoding order.
+---
+name:            literal_with_swapped_xy_order
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: literal_with_swapped_xy_order
+    ; CHECK: liveins: $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $sgpr0 = S_MOV_B32 300
+    ; CHECK-NEXT: $vgpr5, $vgpr4 = V_DUAL_SUB_U32_e32_X_MAX_I32_e32_e96_gfx1250 $vgpr0, $vgpr1, $sgpr0, $vgpr1, implicit $exec, implicit $exec, implicit $mode, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4 = V_MAX_I32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    S_ENDPGM 0
+...
+
+# Only 32-bit values are moved, because one S_MOV_B32 has to produce them. A
+# 64-bit literal is left alone, so the pair stays split.
+---
+name:            literal64_is_not_moved
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $vgpr2, $vgpr3, $vgpr8_vgpr9
+    ; CHECK-LABEL: name: literal64_is_not_moved
+    ; CHECK: liveins: $vgpr2, $vgpr3, $vgpr8_vgpr9
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $vgpr4_vgpr5 = V_ADD_F64_pseudo_e32 1311768465173141112, $vgpr8_vgpr9, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $vgpr6 = V_ADD_F32_e32 $vgpr2, $vgpr3, implicit $mode, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4_vgpr5 = V_ADD_F64_pseudo_e32 1311768465173141112, $vgpr8_vgpr9, implicit $mode, implicit $exec
+    $vgpr6 = V_ADD_F32_e32 $vgpr2, $vgpr3, implicit $mode, implicit $exec
+    S_ENDPGM 0
+...
+
+# The same pair reading a scalar register pairs, so only the width of the
+# value stops the case above.
+---
+name:            scalar64_pairs
+tracksRegLiveness: true
+body:             |
+  bb.0:
+    liveins: $sgpr0_sgpr1, $vgpr2, $vgpr3, $vgpr8_vgpr9
+    ; CHECK-LABEL: name: scalar64_pairs
+    ; CHECK: liveins: $sgpr0_sgpr1, $vgpr2, $vgpr3, $vgpr8_vgpr9
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $vgpr4_vgpr5, $vgpr6 = V_DUAL_ADD_F64_pseudo_e32_X_ADD_F32_e32_e96_gfx1250 0, $sgpr0_sgpr1, 0, $vgpr8_vgpr9, 0, $vgpr2, 0, $vgpr3, implicit $mode, implicit $exec, implicit $mode, implicit $exec, implicit $mode, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr4_vgpr5 = V_ADD_F64_pseudo_e32 $sgpr0_sgpr1, $vgpr8_vgpr9, implicit $mode, implicit $exec
+    $vgpr6 = V_ADD_F32_e32 $vgpr2, $vgpr3, implicit $mode, implicit $exec
+    S_ENDPGM 0
+...
+
+# $sgpr35 is saved and restored, so at a return it holds the caller's value
+# again. That value is not in the live-in list, and the register is not
+# pristine either, so only the live-outs of the return block report it. Every
+# other scalar register is live, so the pair has to be given up.
+---
+name:            restored_callee_saved_reg_is_live
+tracksRegLiveness: true
+frameInfo:
+  isCalleeSavedInfoValid: true
+stack:
+  - { id: 0, type: spill-slot, offset: 0, size: 4, alignment: 4,
+      callee-saved-register: '$sgpr35', callee-saved-restored: true }
+body:             |
+  bb.0:
+    liveins: $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105, $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: restored_callee_saved_reg_is_live
+    ; CHECK: liveins: $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105, $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $vgpr4 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: $vgpr5 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: S_NOP 0, implicit $sgpr0_sgpr1_sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33, implicit $sgpr34, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    ; CHECK-NEXT: SI_RETURN
+    $vgpr4 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    S_NOP 0, implicit $sgpr0_sgpr1_sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33, implicit $sgpr34, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    SI_RETURN
+...
+
+# The same shape, but the restore of $sgpr35 comes after the pair. The register
+# is dead over the range the move needs, so it is still used.
+---
+name:            callee_saved_reg_dead_over_range
+tracksRegLiveness: true
+frameInfo:
+  isCalleeSavedInfoValid: true
+stack:
+  - { id: 0, type: spill-slot, offset: 0, size: 4, alignment: 4,
+      callee-saved-register: '$sgpr35', callee-saved-restored: true }
+body:             |
+  bb.0:
+    liveins: $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105, $vgpr0, $vgpr1
+    ; CHECK-LABEL: name: callee_saved_reg_dead_over_range
+    ; CHECK: liveins: $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr4_sgpr5_sgpr6_sgpr7, $sgpr8_sgpr9_sgpr10_sgpr11, $sgpr12_sgpr13_sgpr14_sgpr15, $sgpr16_sgpr17_sgpr18_sgpr19, $sgpr20_sgpr21_sgpr22_sgpr23, $sgpr24_sgpr25_sgpr26_sgpr27, $sgpr28_sgpr29_sgpr30_sgpr31, $sgpr32_sgpr33_sgpr34, $sgpr36_sgpr37_sgpr38_sgpr39, $sgpr40_sgpr41_sgpr42_sgpr43, $sgpr44_sgpr45_sgpr46_sgpr47, $sgpr48_sgpr49_sgpr50_sgpr51, $sgpr52_sgpr53_sgpr54_sgpr55, $sgpr56_sgpr57_sgpr58_sgpr59, $sgpr60_sgpr61_sgpr62_sgpr63, $sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71, $sgpr72_sgpr73_sgpr74_sgpr75, $sgpr76_sgpr77_sgpr78_sgpr79, $sgpr80_sgpr81_sgpr82_sgpr83, $sgpr84_sgpr85_sgpr86_sgpr87, $sgpr88_sgpr89_sgpr90_sgpr91, $sgpr92_sgpr93_sgpr94_sgpr95, $sgpr96_sgpr97_sgpr98_sgpr99, $sgpr100_sgpr101_sgpr102_sgpr103, $sgpr104, $sgpr105, $vgpr0, $vgpr1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: $sgpr35 = S_MOV_B32 300
+    ; CHECK-NEXT: $vgpr4, $vgpr5 = V_DUAL_SUB_U32_e32_X_SUB_U32_e32_e96_gfx1250 $vgpr0, $vgpr1, $sgpr35, $vgpr1, implicit $exec, implicit $exec, implicit $mode, implicit $exec
+    ; CHECK-NEXT: $sgpr35 = S_MOV_B32 7
+    ; CHECK-NEXT: S_NOP 0, implicit $sgpr0_sgpr1_sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33, implicit $sgpr34, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    ; CHECK-NEXT: SI_RETURN
+    $vgpr4 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    $sgpr35 = S_MOV_B32 7
+    S_NOP 0, implicit $sgpr0_sgpr1_sgpr2_sgpr3, implicit $sgpr4_sgpr5_sgpr6_sgpr7, implicit $sgpr8_sgpr9_sgpr10_sgpr11, implicit $sgpr12_sgpr13_sgpr14_sgpr15, implicit $sgpr16_sgpr17_sgpr18_sgpr19, implicit $sgpr20_sgpr21_sgpr22_sgpr23, implicit $sgpr24_sgpr25_sgpr26_sgpr27, implicit $sgpr28_sgpr29_sgpr30_sgpr31, implicit $sgpr32_sgpr33, implicit $sgpr34, implicit $sgpr36_sgpr37_sgpr38_sgpr39, implicit $sgpr40_sgpr41_sgpr42_sgpr43, implicit $sgpr44_sgpr45_sgpr46_sgpr47, implicit $sgpr48_sgpr49_sgpr50_sgpr51, implicit $sgpr52_sgpr53_sgpr54_sgpr55, implicit $sgpr56_sgpr57_sgpr58_sgpr59, implicit $sgpr60_sgpr61_sgpr62_sgpr63, implicit $sgpr64_sgpr65_sgpr66_sgpr67, implicit $sgpr68_sgpr69_sgpr70_sgpr71, implicit $sgpr72_sgpr73_sgpr74_sgpr75, implicit $sgpr76_sgpr77_sgpr78_sgpr79, implicit $sgpr80_sgpr81_sgpr82_sgpr83, implicit $sgpr84_sgpr85_sgpr86_sgpr87, implicit $sgpr88_sgpr89_sgpr90_sgpr91, implicit $sgpr92_sgpr93_sgpr94_sgpr95, implicit $sgpr96_sgpr97_sgpr98_sgpr99, implicit $sgpr100_sgpr101_sgpr102_sgpr103, implicit $sgpr104_sgpr105
+    SI_RETURN
+...
+
+# Without liveness no free register can be found, so the pair stays as two
+# instructions.
+---
+name:            no_tracked_liveness
+tracksRegLiveness: false
+body:             |
+  bb.0:
+    ; CHECK-LABEL: name: no_tracked_liveness
+    ; CHECK: $vgpr0 = IMPLICIT_DEF
+    ; CHECK-NEXT: $vgpr1 = IMPLICIT_DEF
+    ; CHECK-NEXT: $vgpr4 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    ; CHECK-NEXT: $vgpr5 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    ; CHECK-NEXT: S_ENDPGM 0
+    $vgpr0 = IMPLICIT_DEF
+    $vgpr1 = IMPLICIT_DEF
+    $vgpr4 = V_SUB_U32_e32 $vgpr0, $vgpr1, implicit $exec
+    $vgpr5 = V_SUB_U32_e32 300, $vgpr1, implicit $mode, implicit $exec
+    S_ENDPGM 0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/waitcnt-loop-ds-prefetch-flushed.ll b/llvm/test/CodeGen/AMDGPU/waitcnt-loop-ds-prefetch-flushed.ll
index 9f7e05b3afa17..d54731f15a89f 100644
--- a/llvm/test/CodeGen/AMDGPU/waitcnt-loop-ds-prefetch-flushed.ll
+++ b/llvm/test/CodeGen/AMDGPU/waitcnt-loop-ds-prefetch-flushed.ll
@@ -14,9 +14,9 @@ define amdgpu_kernel void @ds_prefetch_flushed(ptr addrspace(3) %lds, ptr addrsp
 ; CHECK-NEXT:    s_clause 0x1
 ; CHECK-NEXT:    s_load_b32 s1, s[4:5], 0x0 nv
 ; CHECK-NEXT:    s_load_b32 s0, s[4:5], 0x10 nv
-; CHECK-NEXT:    v_and_b32_e32 v10, 0x3ff, v0
-; CHECK-NEXT:    v_mov_b32_e32 v4, 0
-; CHECK-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; CHECK-NEXT:    s_mov_b32 s2, 0x3ff
+; CHECK-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; CHECK-NEXT:    v_dual_mov_b32 v4, 0 :: v_dual_bitop2_b32 v10, s2, v0 bitop3:0x40
 ; CHECK-NEXT:    v_dual_mov_b32 v5, v4 :: v_dual_mov_b32 v6, v4
 ; CHECK-NEXT:    v_dual_mov_b32 v7, v4 :: v_dual_mov_b32 v8, v4
 ; CHECK-NEXT:    v_mov_b32_e32 v9, v4
diff --git a/llvm/test/CodeGen/AMDGPU/waitcnt-loop-ds-prefetch-pattern.ll b/llvm/test/CodeGen/AMDGPU/waitcnt-loop-ds-prefetch-pattern.ll
index ae72c5f640166..e9b3ba942f817 100644
--- a/llvm/test/CodeGen/AMDGPU/waitcnt-loop-ds-prefetch-pattern.ll
+++ b/llvm/test/CodeGen/AMDGPU/waitcnt-loop-ds-prefetch-pattern.ll
@@ -14,12 +14,13 @@ define amdgpu_kernel void @ds_prefetch_pattern(ptr addrspace(3) %lds, ptr addrsp
 ; CHECK-NEXT:    s_clause 0x1
 ; CHECK-NEXT:    s_load_b32 s1, s[4:5], 0x0 nv
 ; CHECK-NEXT:    s_load_b32 s0, s[4:5], 0x10 nv
-; CHECK-NEXT:    v_and_b32_e32 v12, 0x3ff, v0
-; CHECK-NEXT:    v_mov_b32_e32 v4, 0
-; CHECK-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; CHECK-NEXT:    s_mov_b32 s2, 0x3ff
+; CHECK-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; CHECK-NEXT:    v_dual_mov_b32 v4, 0 :: v_dual_bitop2_b32 v12, s2, v0 bitop3:0x40
 ; CHECK-NEXT:    v_dual_mov_b32 v5, v4 :: v_dual_mov_b32 v6, v4
 ; CHECK-NEXT:    v_mov_b32_e32 v7, v4
 ; CHECK-NEXT:    s_wait_kmcnt 0x0
+; CHECK-NEXT:    s_delay_alu instid0(VALU_DEP_3)
 ; CHECK-NEXT:    v_lshl_add_u32 v13, v12, 8, s1
 ; CHECK-NEXT:    s_mov_b32 s1, 0
 ; CHECK-NEXT:    ds_load_b128 v[8:11], v13



More information about the llvm-branch-commits mailing list