[llvm] [AMDGPU] Avoid wmma bank conflict for GFX11 and GFX12 (PR #205530)

via llvm-commits llvm-commits at lists.llvm.org
Wed Jun 24 04:39:58 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-globalisel

Author: Shoreshen

<details>
<summary>Changes</summary>

Trying to solve this issue: https://github.com/llvm/llvm-project/issues/204254

---

Patch is 104.68 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/205530.diff


10 Files Affected:

- (modified) llvm/lib/Target/AMDGPU/GCNPreRAOptimizations.cpp (+66-1) 
- (modified) llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h (+25) 
- (modified) llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp (+104-2) 
- (modified) llvm/lib/Target/AMDGPU/SIRegisterInfo.h (+9) 
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wmma_32.ll (+14-14) 
- (modified) llvm/test/CodeGen/AMDGPU/coexec-scheduler.ll (+241-253) 
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.sched.group.barrier.gfx11.ll (+100-106) 
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.wmma_32.ll (+14-14) 
- (added) llvm/test/CodeGen/AMDGPU/wmma-bank-hint.ll (+610) 
- (modified) llvm/test/CodeGen/AMDGPU/wmma-gfx12-w32-f16-f32-matrix-modifiers.ll (+10-10) 


``````````diff
diff --git a/llvm/lib/Target/AMDGPU/GCNPreRAOptimizations.cpp b/llvm/lib/Target/AMDGPU/GCNPreRAOptimizations.cpp
index 825634d7af65b..d4c2530afcd26 100644
--- a/llvm/lib/Target/AMDGPU/GCNPreRAOptimizations.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNPreRAOptimizations.cpp
@@ -32,8 +32,10 @@
 
 #include "GCNPreRAOptimizations.h"
 #include "AMDGPU.h"
+#include "GCNRegPressure.h"
 #include "GCNSubtarget.h"
 #include "MCTargetDesc/AMDGPUMCTargetDesc.h"
+#include "SIMachineFunctionInfo.h"
 #include "SIRegisterInfo.h"
 #include "llvm/CodeGen/LiveIntervals.h"
 #include "llvm/CodeGen/MachineFunctionPass.h"
@@ -54,6 +56,8 @@ class GCNPreRAOptimizationsImpl {
 
   bool processReg(Register Reg);
   void hintTrue16Copy(const MachineInstr &MI);
+  void recordWMMABankSiblings(const MachineInstr &MI,
+                              SIMachineFunctionInfo &MFI);
   bool optimizeBVHStack(MachineInstr &MI);
 
 public:
@@ -262,6 +266,41 @@ void GCNPreRAOptimizationsImpl::hintTrue16Copy(const MachineInstr &MI) {
     MRI->setRegAllocationHint(Dst, AMDGPURI::Size16, Src);
 }
 
+void GCNPreRAOptimizationsImpl::recordWMMABankSiblings(
+    const MachineInstr &MI, SIMachineFunctionInfo &MFI) {
+  // The consecutive-allocation VGPR bank collision only happens when each
+  // matrix operand is a multiple of 4 VGPRs (128 bits) wide; with narrower
+  // operands the start registers already fall on different banks. Restrict the
+  // hint to WMMAs whose matrix source operands are a multiple of 128 bits.
+  auto Is128BitMultiple = [&](AMDGPU::OpName Name) {
+    const MachineOperand *MO = TII->getNamedOperand(MI, Name);
+    return MO && MO->isReg() && MO->getReg() &&
+           TRI->getRegSizeInBits(MO->getReg(), *MRI) % 128 == 0;
+  };
+  if (!Is128BitMultiple(AMDGPU::OpName::src0) ||
+      !Is128BitMultiple(AMDGPU::OpName::src1))
+    return;
+
+  // Record, for each virtual register operand, the other operands of this WMMA
+  // so SIRegisterInfo::getRegAllocationHints can steer them onto distinct
+  // VGPR banks. A register used by several WMMAs accumulates the union of their
+  // siblings.
+  SmallVector<Register, 4> Regs;
+  for (const MachineOperand &MO : MI.operands())
+    if (MO.isReg() && MO.getReg() && MO.getReg().isVirtual() &&
+        !is_contained(Regs, MO.getReg()))
+      Regs.push_back(MO.getReg());
+
+  DenseMap<Register, SmallVector<Register, 3>> &Siblings =
+      MFI.getWMMABankSiblings();
+  for (Register Reg : Regs) {
+    SmallVector<Register, 3> &S = Siblings[Reg];
+    for (Register Other : Regs)
+      if (Other != Reg && !is_contained(S, Other))
+        S.push_back(Other);
+  }
+}
+
 bool GCNPreRAOptimizationsImpl::optimizeBVHStack(MachineInstr &MI) {
   SmallVector<Register, 2> UseRegs;
 
@@ -321,10 +360,15 @@ bool GCNPreRAOptimizationsImpl::run(MachineFunction &MF) {
 
   const bool HasBVHStack = ST.hasBVHDualAndBVH8Insts();
   const bool HasRealTrue16 = ST.useRealTrue16Insts();
+  // On gfx11/gfx12 the operands of a WMMA should be placed on distinct VGPR
+  // banks to avoid an issue-latency penalty.
+  const bool HasWMMABankHint = ST.hasFeature(AMDGPU::FeatureGFX11) ||
+                               ST.hasFeature(AMDGPU::FeatureGFX12);
 
-  if (!HasRealTrue16 && !HasBVHStack)
+  if (!HasRealTrue16 && !HasBVHStack && !HasWMMABankHint)
     return Changed;
 
+  SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
   for (MachineBasicBlock &MBB : MF) {
     for (MachineInstr &MI : MBB) {
       // Add RA hints to improve True16 COPY elimination.
@@ -332,6 +376,11 @@ bool GCNPreRAOptimizationsImpl::run(MachineFunction &MF) {
         hintTrue16Copy(MI);
         continue;
       }
+      // Record WMMA operand siblings for VGPR bank-conflict avoidance hints.
+      if (HasWMMABankHint && SIInstrInfo::isWMMA(MI)) {
+        recordWMMABankSiblings(MI, *MFI);
+        continue;
+      }
       // Add implicit uses to avoid early wait on intersect ray instructions.
       if (HasBVHStack &&
           (MI.getOpcode() == AMDGPU::DS_BVH_STACK_RTN_B32 ||
@@ -343,5 +392,21 @@ bool GCNPreRAOptimizationsImpl::run(MachineFunction &MF) {
     }
   }
 
+  // If any bank-sensitive WMMA was tagged, record the function's peak ArchVGPR
+  // pressure (the actual VGPR demand). The bank hint anchors its reserved
+  // regions to this instead of the top of the whole VGPR file.
+  if (HasWMMABankHint && !MFI->getWMMABankSiblings().empty()) {
+    unsigned Peak = 0;
+    GCNUpwardRPTracker RPT(*LIS);
+    for (const MachineBasicBlock &MBB : MF) {
+      RPT.reset(*MRI, LIS->getSlotIndexes()->getMBBEndIdx(&MBB).getPrevSlot());
+      for (const MachineInstr &MI : reverse(MBB)) {
+        RPT.recede(MI);
+        Peak = std::max(Peak, RPT.getPressure().getArchVGPRNum());
+      }
+    }
+    MFI->setWMMAPeakVGPRPressure(Peak);
+  }
+
   return Changed;
 }
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 1f43505650222..0717e48132aca 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -617,6 +617,19 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
   // load/store is enabled.
   IndexedMap<uint32_t, VGPRBlock2IndexFunctor> MaskForVGPRBlockOps;
 
+  // Maps each virtual register that is an operand of a bank-sensitive v_wmma_*
+  // (matrix sources a multiple of 128 bits) to that instruction's other operand
+  // registers. Populated once by GCNPreRAOptimizations and consumed by
+  // SIRegisterInfo::getRegAllocationHints for VGPR bank-conflict avoidance, so
+  // the hint hook does not rescan the function for every virtual register.
+  DenseMap<Register, SmallVector<Register, 3>> WMMABankSiblings;
+
+  // Peak ArchVGPR pressure of the function (the "actual demand"), computed once
+  // by GCNPreRAOptimizations. Used by the WMMA bank hint to anchor its reserved
+  // bank regions to the top of the real footprint instead of the top of the
+  // whole VGPR file, so it does not inflate the VGPR count of small kernels.
+  unsigned WMMAPeakVGPRPressure = 0;
+
 private:
   Register VGPRForAGPRCopy;
 
@@ -650,6 +663,18 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
     return MaskForVGPRBlockOps.inBounds(RegisterBlock);
   }
 
+  // Accessors for the WMMA bank-hint map (see WMMABankSiblings).
+  DenseMap<Register, SmallVector<Register, 3>> &getWMMABankSiblings() {
+    return WMMABankSiblings;
+  }
+  const DenseMap<Register, SmallVector<Register, 3>> &
+  getWMMABankSiblings() const {
+    return WMMABankSiblings;
+  }
+
+  unsigned getWMMAPeakVGPRPressure() const { return WMMAPeakVGPRPressure; }
+  void setWMMAPeakVGPRPressure(unsigned P) { WMMAPeakVGPRPressure = P; }
+
 public:
   SIMachineFunctionInfo(const SIMachineFunctionInfo &MFI) = default;
   SIMachineFunctionInfo(const Function &F, const GCNSubtarget *STI);
diff --git a/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp b/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
index 9700720f0373a..ad18d7c940ddc 100644
--- a/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
@@ -4144,8 +4144,110 @@ bool SIRegisterInfo::getRegAllocationHints(Register VirtReg,
     return false;
   }
   default:
-    return TargetRegisterInfo::getRegAllocationHints(VirtReg, Order, Hints, MF,
-                                                     VRM);
+    break;
+  }
+
+  bool BaseImplRetVal =
+      TargetRegisterInfo::getRegAllocationHints(VirtReg, Order, Hints, MF, VRM);
+
+  // Append v_wmma_* bank-conflict avoidance candidates (gfx11/gfx12). This only
+  // augments the candidate order for VirtReg and never changes BaseImplRetVal.
+  addWMMABankConflictHints(VirtReg, Order, Hints, MF, VRM);
+
+  return BaseImplRetVal;
+}
+
+// VGPR bank-conflict avoidance for v_wmma_* operands on gfx11/gfx12.
+//
+// A WMMA reads src0/src1/src2 through per-bank read ports (bank = first VGPR
+// % 4); when two operands share a bank the instruction pays an extra
+// issue-latency cycle. GCNPreRAOptimizations records, for each operand of a
+// bank-sensitive WMMA, that instruction's other operands (WMMABankSiblings).
+// Here, for the siblings that already have a physreg, append the candidates
+// from a not-yet-used bank so the allocator prefers them.
+//
+// The bias is applied via the candidate list (not by recording an MRI
+// allocation hint) on purpose: an MRI hint feeds the allocation-priority
+// computation (VRM::hasKnownPreference) and can reorder allocation so the
+// result register displaces the live-in source operands, introducing copies.
+// Appending here happens after the copy hints emitted by the base
+// implementation, so coalescing keeps priority and only this register's
+// candidate order is affected (never the global allocation order).
+void SIRegisterInfo::addWMMABankConflictHints(Register VirtReg,
+                                              ArrayRef<MCPhysReg> Order,
+                                              SmallVectorImpl<MCPhysReg> &Hints,
+                                              const MachineFunction &MF,
+                                              const VirtRegMap *VRM) const {
+  const MachineRegisterInfo &MRI = MF.getRegInfo();
+  if (!VRM || !isVGPR(MRI, VirtReg))
+    return;
+
+  const SIMachineFunctionInfo &FuncInfo = *MF.getInfo<SIMachineFunctionInfo>();
+  const DenseMap<Register, SmallVector<Register, 3>> &Siblings =
+      FuncInfo.getWMMABankSiblings();
+
+  auto It = Siblings.find(VirtReg);
+  if (It == Siblings.end())
+    return;
+
+  // Banks already taken by allocated siblings.
+  unsigned UsedBankMask = 0;
+  for (Register Sibling : It->second) {
+    Register Phys = Sibling;
+    if (Phys.isVirtual()) {
+      if (!VRM->hasPhys(Phys))
+        continue;
+      Phys = VRM->getPhys(Phys);
+    }
+    if (Phys.isPhysical() && isVGPR(MRI, Phys))
+      UsedBankMask |= 1u << (getHWRegIndex(Phys) % 4);
+  }
+
+  unsigned W = getRegSizeInBits(VirtReg, MRI) / 32;
+  if (W && (W % 4) != 0)
+    return;
+
+  // Build a hardware-index -> physreg map and record the highest start index.
+  // Order is not guaranteed to be sorted (non-kernel functions reorder their
+  // callee-saved VGPRs), so we look candidates up by index instead of trusting
+  // Order's sequence or its last element.
+  DenseMap<unsigned, MCPhysReg> IdxToReg;
+  unsigned MaxIdx = 0;
+  for (MCPhysReg PhysReg : Order) {
+    unsigned Idx = getHWRegIndex(PhysReg);
+    IdxToReg[Idx] = PhysReg;
+    MaxIdx = std::max(MaxIdx, Idx);
+  }
+
+  // Anchor the candidate window to the function's actual VGPR footprint: peak
+  // WMMA-region pressure, rounded up to the allocation granule and capped by
+  // the VGPR budget. Without this the window is the top of the whole VGPR file
+  // (~v256) for any function without an occupancy cap, so operands get steered
+  // to very high registers, inflating the footprint and dropping occupancy on
+  // low-pressure functions (and triggering MSG_DEALLOC_VGPRS).
+  unsigned MaxVGPR = MaxIdx + W;
+  if (unsigned Peak = FuncInfo.getWMMAPeakVGPRPressure()) {
+    unsigned Gran = AMDGPU::IsaInfo::getVGPRAllocGranule(
+        ST, FuncInfo.getDynamicVGPRBlockSize());
+    unsigned Want = alignTo(Peak, std::max(Gran, 1u));
+    if (unsigned Budget = ST.getMaxNumVGPRs(MF))
+      Want = std::min(Want, Budget);
+    if (Want && Want < MaxVGPR)
+      MaxVGPR = Want;
+  }
+  if (MaxVGPR < 4 * W + 3)
+    return;
+  unsigned N = (MaxVGPR - 3) / (4 * W); // blocks per bank region
+  unsigned BankSize = N * W;
+  for (unsigned n = 0; n < N; n++) {
+    for (unsigned Bank = 0; Bank < 4; Bank++) {
+      if (((UsedBankMask >> Bank) & 1) != 0)
+        continue;
+      unsigned Start = Bank * BankSize + Bank + n * W;
+      auto RIt = IdxToReg.find(Start);
+      if (RIt != IdxToReg.end())
+        Hints.push_back(RIt->second);
+    }
   }
 }
 
diff --git a/llvm/lib/Target/AMDGPU/SIRegisterInfo.h b/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
index 5e08e47ad4d83..ba633ba7c5dc1 100644
--- a/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
@@ -364,6 +364,15 @@ class SIRegisterInfo final : public AMDGPUGenRegisterInfo {
                              const MachineFunction &MF, const VirtRegMap *VRM,
                              const LiveRegMatrix *Matrix) const override;
 
+  /// Append physreg candidates that keep the v_wmma_* operand \p VirtReg in a
+  /// VGPR bank distinct from its already-allocated WMMA siblings (gfx11/gfx12).
+  /// Only augments \p Hints for \p VirtReg; the global allocation order is
+  /// unchanged.
+  void addWMMABankConflictHints(Register VirtReg, ArrayRef<MCPhysReg> Order,
+                                SmallVectorImpl<MCPhysReg> &Hints,
+                                const MachineFunction &MF,
+                                const VirtRegMap *VRM) const;
+
   const int *getRegUnitPressureSets(MCRegUnit RegUnit) const override;
 
   MCRegister getReturnAddressReg(const MachineFunction &MF) const;
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wmma_32.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wmma_32.ll
index 57d3db413a277..4a8b2d52b3925 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wmma_32.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wmma_32.ll
@@ -95,16 +95,16 @@ bb:
 define amdgpu_ps void @test_wmma_f16_16x16x16_f16_tied(<16 x half> %A.0, <16 x half> %B.0, <16 x half> %A.1, <16 x half> %B.1, <16 x half> %C, ptr addrspace(1) %out.0, ptr addrspace(1) %out.1) {
 ; W32-LABEL: test_wmma_f16_16x16x16_f16_tied:
 ; W32:       ; %bb.0: ; %bb
-; W32-NEXT:    v_dual_mov_b32 v51, v39 :: v_dual_mov_b32 v50, v38
-; W32-NEXT:    v_dual_mov_b32 v49, v37 :: v_dual_mov_b32 v48, v36
-; W32-NEXT:    v_dual_mov_b32 v47, v35 :: v_dual_mov_b32 v46, v34
-; W32-NEXT:    v_dual_mov_b32 v45, v33 :: v_dual_mov_b32 v44, v32
+; W32-NEXT:    v_dual_mov_b32 v58, v39 :: v_dual_mov_b32 v57, v38
+; W32-NEXT:    v_dual_mov_b32 v56, v37 :: v_dual_mov_b32 v55, v36
+; W32-NEXT:    v_dual_mov_b32 v54, v35 :: v_dual_mov_b32 v53, v34
+; W32-NEXT:    v_dual_mov_b32 v52, v33 :: v_dual_mov_b32 v51, v32
 ; W32-NEXT:    v_wmma_f16_16x16x16_f16 v[32:39], v[16:23], v[24:31], v[32:39]
 ; W32-NEXT:    s_delay_alu instid0(VALU_DEP_2)
-; W32-NEXT:    v_wmma_f16_16x16x16_f16 v[44:51], v[0:7], v[8:15], v[44:51]
+; W32-NEXT:    v_wmma_f16_16x16x16_f16 v[51:58], v[0:7], v[8:15], v[51:58]
 ; W32-NEXT:    s_clause 0x1
-; W32-NEXT:    global_store_b128 v[40:41], v[44:47], off
-; W32-NEXT:    global_store_b128 v[40:41], v[48:51], off offset:16
+; W32-NEXT:    global_store_b128 v[40:41], v[51:54], off
+; W32-NEXT:    global_store_b128 v[40:41], v[55:58], off offset:16
 ; W32-NEXT:    s_clause 0x1
 ; W32-NEXT:    global_store_b128 v[42:43], v[32:35], off
 ; W32-NEXT:    global_store_b128 v[42:43], v[36:39], off offset:16
@@ -170,16 +170,16 @@ bb:
 define amdgpu_ps void @test_wmma_bf16_16x16x16_bf16_tied(<16 x i16> %A.0, <16 x i16> %B.0, <16 x i16> %A.1, <16 x i16> %B.1, <16 x i16> %C, ptr addrspace(1) %out.0, ptr addrspace(1) %out.1) {
 ; W32-LABEL: test_wmma_bf16_16x16x16_bf16_tied:
 ; W32:       ; %bb.0: ; %bb
-; W32-NEXT:    v_dual_mov_b32 v51, v39 :: v_dual_mov_b32 v50, v38
-; W32-NEXT:    v_dual_mov_b32 v49, v37 :: v_dual_mov_b32 v48, v36
-; W32-NEXT:    v_dual_mov_b32 v47, v35 :: v_dual_mov_b32 v46, v34
-; W32-NEXT:    v_dual_mov_b32 v45, v33 :: v_dual_mov_b32 v44, v32
+; W32-NEXT:    v_dual_mov_b32 v58, v39 :: v_dual_mov_b32 v57, v38
+; W32-NEXT:    v_dual_mov_b32 v56, v37 :: v_dual_mov_b32 v55, v36
+; W32-NEXT:    v_dual_mov_b32 v54, v35 :: v_dual_mov_b32 v53, v34
+; W32-NEXT:    v_dual_mov_b32 v52, v33 :: v_dual_mov_b32 v51, v32
 ; W32-NEXT:    v_wmma_bf16_16x16x16_bf16 v[32:39], v[16:23], v[24:31], v[32:39]
 ; W32-NEXT:    s_delay_alu instid0(VALU_DEP_2)
-; W32-NEXT:    v_wmma_bf16_16x16x16_bf16 v[44:51], v[0:7], v[8:15], v[44:51]
+; W32-NEXT:    v_wmma_bf16_16x16x16_bf16 v[51:58], v[0:7], v[8:15], v[51:58]
 ; W32-NEXT:    s_clause 0x1
-; W32-NEXT:    global_store_b128 v[40:41], v[44:47], off
-; W32-NEXT:    global_store_b128 v[40:41], v[48:51], off offset:16
+; W32-NEXT:    global_store_b128 v[40:41], v[51:54], off
+; W32-NEXT:    global_store_b128 v[40:41], v[55:58], off offset:16
 ; W32-NEXT:    s_clause 0x1
 ; W32-NEXT:    global_store_b128 v[42:43], v[32:35], off
 ; W32-NEXT:    global_store_b128 v[42:43], v[36:39], off offset:16
diff --git a/llvm/test/CodeGen/AMDGPU/coexec-scheduler.ll b/llvm/test/CodeGen/AMDGPU/coexec-scheduler.ll
index ac121195de432..e3dfb6feab7c1 100644
--- a/llvm/test/CodeGen/AMDGPU/coexec-scheduler.ll
+++ b/llvm/test/CodeGen/AMDGPU/coexec-scheduler.ll
@@ -14,78 +14,74 @@ define amdgpu_kernel void @ds_wmma(ptr addrspace(3) %base, ptr addrspace(1) %out
 ; COEXEC-NEXT:    v_dual_mov_b32 v1, v0 :: v_dual_mov_b32 v2, v0
 ; COEXEC-NEXT:    v_dual_mov_b32 v3, v0 :: v_dual_mov_b32 v4, v0
 ; COEXEC-NEXT:    v_dual_mov_b32 v5, v0 :: v_dual_mov_b32 v6, v0
-; COEXEC-NEXT:    v_dual_mov_b32 v7, v0 :: v_dual_mov_b32 v8, v0
+; COEXEC-NEXT:    v_dual_mov_b32 v7, v0 :: v_dual_mov_b32 v34, v0
+; COEXEC-NEXT:    v_dual_mov_b32 v35, v0 :: v_dual_mov_b32 v36, v0
+; COEXEC-NEXT:    v_dual_mov_b32 v37, v0 :: v_dual_mov_b32 v38, v0
+; COEXEC-NEXT:    v_dual_mov_b32 v39, v0 :: v_dual_mov_b32 v40, v0
+; COEXEC-NEXT:    v_dual_mov_b32 v41, v0 :: v_dual_mov_b32 v8, v0
 ; COEXEC-NEXT:    v_dual_mov_b32 v9, v0 :: v_dual_mov_b32 v10, v0
 ; COEXEC-NEXT:    v_dual_mov_b32 v11, v0 :: v_dual_mov_b32 v12, v0
 ; COEXEC-NEXT:    v_dual_mov_b32 v13, v0 :: v_dual_mov_b32 v14, v0
-; COEXEC-NEXT:    v_dual_mov_b32 v15, v0 :: v_dual_mov_b32 v16, v0
-; COEXEC-NEXT:    v_dual_mov_b32 v17, v0 :: v_dual_mov_b32 v18, v0
-; COEXEC-NEXT:    v_dual_mov_b32 v19, v0 :: v_dual_mov_b32 v20, v0
-; COEXEC-NEXT:    v_dual_mov_b32 v21, v0 :: v_dual_mov_b32 v22, v0
-; COEXEC-NEXT:    v_dual_mov_b32 v23, v0 :: v_dual_mov_b32 v24, v0
-; COEXEC-NEXT:    v_dual_mov_b32 v25, v0 :: v_dual_mov_b32 v26, v0
-; COEXEC-NEXT:    v_dual_mov_b32 v27, v0 :: v_dual_mov_b32 v28, v0
+; COEXEC-NEXT:    v_dual_mov_b32 v15, v0 :: v_dual_mov_b32 v42, v0
+; COEXEC-NEXT:    v_dual_mov_b32 v43, v0 :: v_dual_mov_b32 v44, v0
+; COEXEC-NEXT:    v_dual_mov_b32 v45, v0 :: v_dual_mov_b32 v46, v0
 ; COEXEC-NEXT:    s_wait_kmcnt 0x0
 ; COEXEC-NEXT:    s_bitcmp1_b32 s0, 0
-; COEXEC-NEXT:    v_mov_b32_e32 v29, v0
+; COEXEC-NEXT:    v_mov_b32_e32 v47, v0
 ; COEXEC-NEXT:    s_cselect_b32 s0, -1, 0
-; COEXEC-NEXT:    v_mov_b32_e32 v30, v0
+; COEXEC-NEXT:    v_mov_b32_e32 v48, v0
 ; COEXEC-NEXT:    s_xor_b32 s0, s0, -1
-; COEXEC-NEXT:    v_mov_b32_e32 v31, v0
-; COEXEC-NEXT:    v_cndmask_b32_e64 v32, 0, 1, s0
-; COEXEC-NEXT:    v_cmp_ne_u32_e64 s0, 1, v32
+; COEXEC-NEXT:    v_mov_b32_e32 v49, v0
+; COEXEC-NEXT:    v_cndmask_b32_e64 v16, 0, 1, s0
+; COEXEC-NEXT:    v_cmp_ne_u32_e64 s0, 1, v16
 ; COEXEC-NEXT:  .LBB0_1: ; %loop
 ; COEXEC-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; COEXEC-NEXT:    s_and_b32 vcc_lo, exec_lo, s0
-; COEXEC-NEXT:    v_nop
-; COEXEC-NEXT:    v_nop
-; COEXEC-NEXT:    v_nop
-; COEXEC-NEXT:    v_nop
-; COEXEC-NEXT:    v_mov_b32_e32 v88, s2
+; COEXEC-NEXT:    v_mov_b32_e32 v32, s2
 ; COEXEC-NEXT:    s_add_co_i32 s2, s2, s1
-; COEXEC-NEXT:    ds_load_tr16_b128 v[36:39], v88 offset:192
-; COEXEC-NEXT:    ds_load_tr16_b128 v[40:43], v88
-; COEXEC-NEXT:    ds_load_tr16_b128 v[44:47], v88 offset:64
-; COEXEC-NEXT:    ds_load_tr16_b128 v[32:35], v88 offset:128
-; COEXEC-NEXT:    ds_load_tr16_b128 v[52:55], v88 offset:448
-; COEXEC-NEXT:    ds_load_tr16_b128 v[48:51], v88 offset:384
-; COEXEC-NEXT:    ds_load_tr16_b128 v[56:59], v88 offset:256
-; COEXEC-NEXT:    ds_load_tr16_b128 v[60:63], v88 offset:320
-; COEXEC-NEXT:    ds_load_tr16_b128 v[68:71], v88 offset:704
-; COEXEC-NEXT:    ds_load_tr16_b128 v[64:67], v88 offset:640
-; COEXEC-NEXT:    ds_load_tr16_b128 v[76:79], v88 offset:576
-; COEXEC-NEXT:    ds_load_tr16_b128 v[72:75], v88 offset:512
-; COEXEC-NEXT:    ds_load_tr16_b128 v[84:87], v88 offset:960
-; COEXEC-NEXT:    ds_load_tr16_b128 v[80:83], v88 offset:896
-; COEXEC-NEXT:    ds_load_tr16_b128 v[92:95], v88 offset:832
-; COEXEC-NEXT:    ds_load_tr16_b128 v[88:91], v88 offset:768
+; COEXEC-NEXT:    ds_load_tr16_b128 v[20:23], v32 offset:192
+; COEXEC-NEXT:    ds_load_tr16_b128 v[24:27], v32
+; COEXEC-NEXT:    ds_load_tr16_b128 v[28:31], v32 offset:64
+; COEXEC-NEXT:    ds_load_tr16_b128 v[16:19], v32 offset:128
+; COEXEC-NEXT:   ...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/205530


More information about the llvm-commits mailing list