[llvm] [AMDGPU] Avoid wmma bank conflict for GFX11 and GFX12 (PR #205530)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Jun 24 04:39:58 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-globalisel
Author: Shoreshen
<details>
<summary>Changes</summary>
Trying to solve this issue: https://github.com/llvm/llvm-project/issues/204254
---
Patch is 104.68 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/205530.diff
10 Files Affected:
- (modified) llvm/lib/Target/AMDGPU/GCNPreRAOptimizations.cpp (+66-1)
- (modified) llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h (+25)
- (modified) llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp (+104-2)
- (modified) llvm/lib/Target/AMDGPU/SIRegisterInfo.h (+9)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wmma_32.ll (+14-14)
- (modified) llvm/test/CodeGen/AMDGPU/coexec-scheduler.ll (+241-253)
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.sched.group.barrier.gfx11.ll (+100-106)
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.wmma_32.ll (+14-14)
- (added) llvm/test/CodeGen/AMDGPU/wmma-bank-hint.ll (+610)
- (modified) llvm/test/CodeGen/AMDGPU/wmma-gfx12-w32-f16-f32-matrix-modifiers.ll (+10-10)
``````````diff
diff --git a/llvm/lib/Target/AMDGPU/GCNPreRAOptimizations.cpp b/llvm/lib/Target/AMDGPU/GCNPreRAOptimizations.cpp
index 825634d7af65b..d4c2530afcd26 100644
--- a/llvm/lib/Target/AMDGPU/GCNPreRAOptimizations.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNPreRAOptimizations.cpp
@@ -32,8 +32,10 @@
#include "GCNPreRAOptimizations.h"
#include "AMDGPU.h"
+#include "GCNRegPressure.h"
#include "GCNSubtarget.h"
#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
+#include "SIMachineFunctionInfo.h"
#include "SIRegisterInfo.h"
#include "llvm/CodeGen/LiveIntervals.h"
#include "llvm/CodeGen/MachineFunctionPass.h"
@@ -54,6 +56,8 @@ class GCNPreRAOptimizationsImpl {
bool processReg(Register Reg);
void hintTrue16Copy(const MachineInstr &MI);
+ void recordWMMABankSiblings(const MachineInstr &MI,
+ SIMachineFunctionInfo &MFI);
bool optimizeBVHStack(MachineInstr &MI);
public:
@@ -262,6 +266,41 @@ void GCNPreRAOptimizationsImpl::hintTrue16Copy(const MachineInstr &MI) {
MRI->setRegAllocationHint(Dst, AMDGPURI::Size16, Src);
}
+void GCNPreRAOptimizationsImpl::recordWMMABankSiblings(
+ const MachineInstr &MI, SIMachineFunctionInfo &MFI) {
+ // The consecutive-allocation VGPR bank collision only happens when each
+ // matrix operand is a multiple of 4 VGPRs (128 bits) wide; with narrower
+ // operands the start registers already fall on different banks. Restrict the
+ // hint to WMMAs whose matrix source operands are a multiple of 128 bits.
+ auto Is128BitMultiple = [&](AMDGPU::OpName Name) {
+ const MachineOperand *MO = TII->getNamedOperand(MI, Name);
+ return MO && MO->isReg() && MO->getReg() &&
+ TRI->getRegSizeInBits(MO->getReg(), *MRI) % 128 == 0;
+ };
+ if (!Is128BitMultiple(AMDGPU::OpName::src0) ||
+ !Is128BitMultiple(AMDGPU::OpName::src1))
+ return;
+
+ // Record, for each virtual register operand, the other operands of this WMMA
+ // so SIRegisterInfo::getRegAllocationHints can steer them onto distinct
+ // VGPR banks. A register used by several WMMAs accumulates the union of their
+ // siblings.
+ SmallVector<Register, 4> Regs;
+ for (const MachineOperand &MO : MI.operands())
+ if (MO.isReg() && MO.getReg() && MO.getReg().isVirtual() &&
+ !is_contained(Regs, MO.getReg()))
+ Regs.push_back(MO.getReg());
+
+ DenseMap<Register, SmallVector<Register, 3>> &Siblings =
+ MFI.getWMMABankSiblings();
+ for (Register Reg : Regs) {
+ SmallVector<Register, 3> &S = Siblings[Reg];
+ for (Register Other : Regs)
+ if (Other != Reg && !is_contained(S, Other))
+ S.push_back(Other);
+ }
+}
+
bool GCNPreRAOptimizationsImpl::optimizeBVHStack(MachineInstr &MI) {
SmallVector<Register, 2> UseRegs;
@@ -321,10 +360,15 @@ bool GCNPreRAOptimizationsImpl::run(MachineFunction &MF) {
const bool HasBVHStack = ST.hasBVHDualAndBVH8Insts();
const bool HasRealTrue16 = ST.useRealTrue16Insts();
+ // On gfx11/gfx12 the operands of a WMMA should be placed on distinct VGPR
+ // banks to avoid an issue-latency penalty.
+ const bool HasWMMABankHint = ST.hasFeature(AMDGPU::FeatureGFX11) ||
+ ST.hasFeature(AMDGPU::FeatureGFX12);
- if (!HasRealTrue16 && !HasBVHStack)
+ if (!HasRealTrue16 && !HasBVHStack && !HasWMMABankHint)
return Changed;
+ SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
for (MachineBasicBlock &MBB : MF) {
for (MachineInstr &MI : MBB) {
// Add RA hints to improve True16 COPY elimination.
@@ -332,6 +376,11 @@ bool GCNPreRAOptimizationsImpl::run(MachineFunction &MF) {
hintTrue16Copy(MI);
continue;
}
+ // Record WMMA operand siblings for VGPR bank-conflict avoidance hints.
+ if (HasWMMABankHint && SIInstrInfo::isWMMA(MI)) {
+ recordWMMABankSiblings(MI, *MFI);
+ continue;
+ }
// Add implicit uses to avoid early wait on intersect ray instructions.
if (HasBVHStack &&
(MI.getOpcode() == AMDGPU::DS_BVH_STACK_RTN_B32 ||
@@ -343,5 +392,21 @@ bool GCNPreRAOptimizationsImpl::run(MachineFunction &MF) {
}
}
+ // If any bank-sensitive WMMA was tagged, record the function's peak ArchVGPR
+ // pressure (the actual VGPR demand). The bank hint anchors its reserved
+ // regions to this instead of the top of the whole VGPR file.
+ if (HasWMMABankHint && !MFI->getWMMABankSiblings().empty()) {
+ unsigned Peak = 0;
+ GCNUpwardRPTracker RPT(*LIS);
+ for (const MachineBasicBlock &MBB : MF) {
+ RPT.reset(*MRI, LIS->getSlotIndexes()->getMBBEndIdx(&MBB).getPrevSlot());
+ for (const MachineInstr &MI : reverse(MBB)) {
+ RPT.recede(MI);
+ Peak = std::max(Peak, RPT.getPressure().getArchVGPRNum());
+ }
+ }
+ MFI->setWMMAPeakVGPRPressure(Peak);
+ }
+
return Changed;
}
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 1f43505650222..0717e48132aca 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -617,6 +617,19 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
// load/store is enabled.
IndexedMap<uint32_t, VGPRBlock2IndexFunctor> MaskForVGPRBlockOps;
+ // Maps each virtual register that is an operand of a bank-sensitive v_wmma_*
+ // (matrix sources a multiple of 128 bits) to that instruction's other operand
+ // registers. Populated once by GCNPreRAOptimizations and consumed by
+ // SIRegisterInfo::getRegAllocationHints for VGPR bank-conflict avoidance, so
+ // the hint hook does not rescan the function for every virtual register.
+ DenseMap<Register, SmallVector<Register, 3>> WMMABankSiblings;
+
+ // Peak ArchVGPR pressure of the function (the "actual demand"), computed once
+ // by GCNPreRAOptimizations. Used by the WMMA bank hint to anchor its reserved
+ // bank regions to the top of the real footprint instead of the top of the
+ // whole VGPR file, so it does not inflate the VGPR count of small kernels.
+ unsigned WMMAPeakVGPRPressure = 0;
+
private:
Register VGPRForAGPRCopy;
@@ -650,6 +663,18 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
return MaskForVGPRBlockOps.inBounds(RegisterBlock);
}
+ // Accessors for the WMMA bank-hint map (see WMMABankSiblings).
+ DenseMap<Register, SmallVector<Register, 3>> &getWMMABankSiblings() {
+ return WMMABankSiblings;
+ }
+ const DenseMap<Register, SmallVector<Register, 3>> &
+ getWMMABankSiblings() const {
+ return WMMABankSiblings;
+ }
+
+ unsigned getWMMAPeakVGPRPressure() const { return WMMAPeakVGPRPressure; }
+ void setWMMAPeakVGPRPressure(unsigned P) { WMMAPeakVGPRPressure = P; }
+
public:
SIMachineFunctionInfo(const SIMachineFunctionInfo &MFI) = default;
SIMachineFunctionInfo(const Function &F, const GCNSubtarget *STI);
diff --git a/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp b/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
index 9700720f0373a..ad18d7c940ddc 100644
--- a/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
@@ -4144,8 +4144,110 @@ bool SIRegisterInfo::getRegAllocationHints(Register VirtReg,
return false;
}
default:
- return TargetRegisterInfo::getRegAllocationHints(VirtReg, Order, Hints, MF,
- VRM);
+ break;
+ }
+
+ bool BaseImplRetVal =
+ TargetRegisterInfo::getRegAllocationHints(VirtReg, Order, Hints, MF, VRM);
+
+ // Append v_wmma_* bank-conflict avoidance candidates (gfx11/gfx12). This only
+ // augments the candidate order for VirtReg and never changes BaseImplRetVal.
+ addWMMABankConflictHints(VirtReg, Order, Hints, MF, VRM);
+
+ return BaseImplRetVal;
+}
+
+// VGPR bank-conflict avoidance for v_wmma_* operands on gfx11/gfx12.
+//
+// A WMMA reads src0/src1/src2 through per-bank read ports (bank = first VGPR
+// % 4); when two operands share a bank the instruction pays an extra
+// issue-latency cycle. GCNPreRAOptimizations records, for each operand of a
+// bank-sensitive WMMA, that instruction's other operands (WMMABankSiblings).
+// Here, for the siblings that already have a physreg, append the candidates
+// from a not-yet-used bank so the allocator prefers them.
+//
+// The bias is applied via the candidate list (not by recording an MRI
+// allocation hint) on purpose: an MRI hint feeds the allocation-priority
+// computation (VRM::hasKnownPreference) and can reorder allocation so the
+// result register displaces the live-in source operands, introducing copies.
+// Appending here happens after the copy hints emitted by the base
+// implementation, so coalescing keeps priority and only this register's
+// candidate order is affected (never the global allocation order).
+void SIRegisterInfo::addWMMABankConflictHints(Register VirtReg,
+ ArrayRef<MCPhysReg> Order,
+ SmallVectorImpl<MCPhysReg> &Hints,
+ const MachineFunction &MF,
+ const VirtRegMap *VRM) const {
+ const MachineRegisterInfo &MRI = MF.getRegInfo();
+ if (!VRM || !isVGPR(MRI, VirtReg))
+ return;
+
+ const SIMachineFunctionInfo &FuncInfo = *MF.getInfo<SIMachineFunctionInfo>();
+ const DenseMap<Register, SmallVector<Register, 3>> &Siblings =
+ FuncInfo.getWMMABankSiblings();
+
+ auto It = Siblings.find(VirtReg);
+ if (It == Siblings.end())
+ return;
+
+ // Banks already taken by allocated siblings.
+ unsigned UsedBankMask = 0;
+ for (Register Sibling : It->second) {
+ Register Phys = Sibling;
+ if (Phys.isVirtual()) {
+ if (!VRM->hasPhys(Phys))
+ continue;
+ Phys = VRM->getPhys(Phys);
+ }
+ if (Phys.isPhysical() && isVGPR(MRI, Phys))
+ UsedBankMask |= 1u << (getHWRegIndex(Phys) % 4);
+ }
+
+ unsigned W = getRegSizeInBits(VirtReg, MRI) / 32;
+ if (W && (W % 4) != 0)
+ return;
+
+ // Build a hardware-index -> physreg map and record the highest start index.
+ // Order is not guaranteed to be sorted (non-kernel functions reorder their
+ // callee-saved VGPRs), so we look candidates up by index instead of trusting
+ // Order's sequence or its last element.
+ DenseMap<unsigned, MCPhysReg> IdxToReg;
+ unsigned MaxIdx = 0;
+ for (MCPhysReg PhysReg : Order) {
+ unsigned Idx = getHWRegIndex(PhysReg);
+ IdxToReg[Idx] = PhysReg;
+ MaxIdx = std::max(MaxIdx, Idx);
+ }
+
+ // Anchor the candidate window to the function's actual VGPR footprint: peak
+ // WMMA-region pressure, rounded up to the allocation granule and capped by
+ // the VGPR budget. Without this the window is the top of the whole VGPR file
+ // (~v256) for any function without an occupancy cap, so operands get steered
+ // to very high registers, inflating the footprint and dropping occupancy on
+ // low-pressure functions (and triggering MSG_DEALLOC_VGPRS).
+ unsigned MaxVGPR = MaxIdx + W;
+ if (unsigned Peak = FuncInfo.getWMMAPeakVGPRPressure()) {
+ unsigned Gran = AMDGPU::IsaInfo::getVGPRAllocGranule(
+ ST, FuncInfo.getDynamicVGPRBlockSize());
+ unsigned Want = alignTo(Peak, std::max(Gran, 1u));
+ if (unsigned Budget = ST.getMaxNumVGPRs(MF))
+ Want = std::min(Want, Budget);
+ if (Want && Want < MaxVGPR)
+ MaxVGPR = Want;
+ }
+ if (MaxVGPR < 4 * W + 3)
+ return;
+ unsigned N = (MaxVGPR - 3) / (4 * W); // blocks per bank region
+ unsigned BankSize = N * W;
+ for (unsigned n = 0; n < N; n++) {
+ for (unsigned Bank = 0; Bank < 4; Bank++) {
+ if (((UsedBankMask >> Bank) & 1) != 0)
+ continue;
+ unsigned Start = Bank * BankSize + Bank + n * W;
+ auto RIt = IdxToReg.find(Start);
+ if (RIt != IdxToReg.end())
+ Hints.push_back(RIt->second);
+ }
}
}
diff --git a/llvm/lib/Target/AMDGPU/SIRegisterInfo.h b/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
index 5e08e47ad4d83..ba633ba7c5dc1 100644
--- a/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIRegisterInfo.h
@@ -364,6 +364,15 @@ class SIRegisterInfo final : public AMDGPUGenRegisterInfo {
const MachineFunction &MF, const VirtRegMap *VRM,
const LiveRegMatrix *Matrix) const override;
+ /// Append physreg candidates that keep the v_wmma_* operand \p VirtReg in a
+ /// VGPR bank distinct from its already-allocated WMMA siblings (gfx11/gfx12).
+ /// Only augments \p Hints for \p VirtReg; the global allocation order is
+ /// unchanged.
+ void addWMMABankConflictHints(Register VirtReg, ArrayRef<MCPhysReg> Order,
+ SmallVectorImpl<MCPhysReg> &Hints,
+ const MachineFunction &MF,
+ const VirtRegMap *VRM) const;
+
const int *getRegUnitPressureSets(MCRegUnit RegUnit) const override;
MCRegister getReturnAddressReg(const MachineFunction &MF) const;
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wmma_32.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wmma_32.ll
index 57d3db413a277..4a8b2d52b3925 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wmma_32.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.wmma_32.ll
@@ -95,16 +95,16 @@ bb:
define amdgpu_ps void @test_wmma_f16_16x16x16_f16_tied(<16 x half> %A.0, <16 x half> %B.0, <16 x half> %A.1, <16 x half> %B.1, <16 x half> %C, ptr addrspace(1) %out.0, ptr addrspace(1) %out.1) {
; W32-LABEL: test_wmma_f16_16x16x16_f16_tied:
; W32: ; %bb.0: ; %bb
-; W32-NEXT: v_dual_mov_b32 v51, v39 :: v_dual_mov_b32 v50, v38
-; W32-NEXT: v_dual_mov_b32 v49, v37 :: v_dual_mov_b32 v48, v36
-; W32-NEXT: v_dual_mov_b32 v47, v35 :: v_dual_mov_b32 v46, v34
-; W32-NEXT: v_dual_mov_b32 v45, v33 :: v_dual_mov_b32 v44, v32
+; W32-NEXT: v_dual_mov_b32 v58, v39 :: v_dual_mov_b32 v57, v38
+; W32-NEXT: v_dual_mov_b32 v56, v37 :: v_dual_mov_b32 v55, v36
+; W32-NEXT: v_dual_mov_b32 v54, v35 :: v_dual_mov_b32 v53, v34
+; W32-NEXT: v_dual_mov_b32 v52, v33 :: v_dual_mov_b32 v51, v32
; W32-NEXT: v_wmma_f16_16x16x16_f16 v[32:39], v[16:23], v[24:31], v[32:39]
; W32-NEXT: s_delay_alu instid0(VALU_DEP_2)
-; W32-NEXT: v_wmma_f16_16x16x16_f16 v[44:51], v[0:7], v[8:15], v[44:51]
+; W32-NEXT: v_wmma_f16_16x16x16_f16 v[51:58], v[0:7], v[8:15], v[51:58]
; W32-NEXT: s_clause 0x1
-; W32-NEXT: global_store_b128 v[40:41], v[44:47], off
-; W32-NEXT: global_store_b128 v[40:41], v[48:51], off offset:16
+; W32-NEXT: global_store_b128 v[40:41], v[51:54], off
+; W32-NEXT: global_store_b128 v[40:41], v[55:58], off offset:16
; W32-NEXT: s_clause 0x1
; W32-NEXT: global_store_b128 v[42:43], v[32:35], off
; W32-NEXT: global_store_b128 v[42:43], v[36:39], off offset:16
@@ -170,16 +170,16 @@ bb:
define amdgpu_ps void @test_wmma_bf16_16x16x16_bf16_tied(<16 x i16> %A.0, <16 x i16> %B.0, <16 x i16> %A.1, <16 x i16> %B.1, <16 x i16> %C, ptr addrspace(1) %out.0, ptr addrspace(1) %out.1) {
; W32-LABEL: test_wmma_bf16_16x16x16_bf16_tied:
; W32: ; %bb.0: ; %bb
-; W32-NEXT: v_dual_mov_b32 v51, v39 :: v_dual_mov_b32 v50, v38
-; W32-NEXT: v_dual_mov_b32 v49, v37 :: v_dual_mov_b32 v48, v36
-; W32-NEXT: v_dual_mov_b32 v47, v35 :: v_dual_mov_b32 v46, v34
-; W32-NEXT: v_dual_mov_b32 v45, v33 :: v_dual_mov_b32 v44, v32
+; W32-NEXT: v_dual_mov_b32 v58, v39 :: v_dual_mov_b32 v57, v38
+; W32-NEXT: v_dual_mov_b32 v56, v37 :: v_dual_mov_b32 v55, v36
+; W32-NEXT: v_dual_mov_b32 v54, v35 :: v_dual_mov_b32 v53, v34
+; W32-NEXT: v_dual_mov_b32 v52, v33 :: v_dual_mov_b32 v51, v32
; W32-NEXT: v_wmma_bf16_16x16x16_bf16 v[32:39], v[16:23], v[24:31], v[32:39]
; W32-NEXT: s_delay_alu instid0(VALU_DEP_2)
-; W32-NEXT: v_wmma_bf16_16x16x16_bf16 v[44:51], v[0:7], v[8:15], v[44:51]
+; W32-NEXT: v_wmma_bf16_16x16x16_bf16 v[51:58], v[0:7], v[8:15], v[51:58]
; W32-NEXT: s_clause 0x1
-; W32-NEXT: global_store_b128 v[40:41], v[44:47], off
-; W32-NEXT: global_store_b128 v[40:41], v[48:51], off offset:16
+; W32-NEXT: global_store_b128 v[40:41], v[51:54], off
+; W32-NEXT: global_store_b128 v[40:41], v[55:58], off offset:16
; W32-NEXT: s_clause 0x1
; W32-NEXT: global_store_b128 v[42:43], v[32:35], off
; W32-NEXT: global_store_b128 v[42:43], v[36:39], off offset:16
diff --git a/llvm/test/CodeGen/AMDGPU/coexec-scheduler.ll b/llvm/test/CodeGen/AMDGPU/coexec-scheduler.ll
index ac121195de432..e3dfb6feab7c1 100644
--- a/llvm/test/CodeGen/AMDGPU/coexec-scheduler.ll
+++ b/llvm/test/CodeGen/AMDGPU/coexec-scheduler.ll
@@ -14,78 +14,74 @@ define amdgpu_kernel void @ds_wmma(ptr addrspace(3) %base, ptr addrspace(1) %out
; COEXEC-NEXT: v_dual_mov_b32 v1, v0 :: v_dual_mov_b32 v2, v0
; COEXEC-NEXT: v_dual_mov_b32 v3, v0 :: v_dual_mov_b32 v4, v0
; COEXEC-NEXT: v_dual_mov_b32 v5, v0 :: v_dual_mov_b32 v6, v0
-; COEXEC-NEXT: v_dual_mov_b32 v7, v0 :: v_dual_mov_b32 v8, v0
+; COEXEC-NEXT: v_dual_mov_b32 v7, v0 :: v_dual_mov_b32 v34, v0
+; COEXEC-NEXT: v_dual_mov_b32 v35, v0 :: v_dual_mov_b32 v36, v0
+; COEXEC-NEXT: v_dual_mov_b32 v37, v0 :: v_dual_mov_b32 v38, v0
+; COEXEC-NEXT: v_dual_mov_b32 v39, v0 :: v_dual_mov_b32 v40, v0
+; COEXEC-NEXT: v_dual_mov_b32 v41, v0 :: v_dual_mov_b32 v8, v0
; COEXEC-NEXT: v_dual_mov_b32 v9, v0 :: v_dual_mov_b32 v10, v0
; COEXEC-NEXT: v_dual_mov_b32 v11, v0 :: v_dual_mov_b32 v12, v0
; COEXEC-NEXT: v_dual_mov_b32 v13, v0 :: v_dual_mov_b32 v14, v0
-; COEXEC-NEXT: v_dual_mov_b32 v15, v0 :: v_dual_mov_b32 v16, v0
-; COEXEC-NEXT: v_dual_mov_b32 v17, v0 :: v_dual_mov_b32 v18, v0
-; COEXEC-NEXT: v_dual_mov_b32 v19, v0 :: v_dual_mov_b32 v20, v0
-; COEXEC-NEXT: v_dual_mov_b32 v21, v0 :: v_dual_mov_b32 v22, v0
-; COEXEC-NEXT: v_dual_mov_b32 v23, v0 :: v_dual_mov_b32 v24, v0
-; COEXEC-NEXT: v_dual_mov_b32 v25, v0 :: v_dual_mov_b32 v26, v0
-; COEXEC-NEXT: v_dual_mov_b32 v27, v0 :: v_dual_mov_b32 v28, v0
+; COEXEC-NEXT: v_dual_mov_b32 v15, v0 :: v_dual_mov_b32 v42, v0
+; COEXEC-NEXT: v_dual_mov_b32 v43, v0 :: v_dual_mov_b32 v44, v0
+; COEXEC-NEXT: v_dual_mov_b32 v45, v0 :: v_dual_mov_b32 v46, v0
; COEXEC-NEXT: s_wait_kmcnt 0x0
; COEXEC-NEXT: s_bitcmp1_b32 s0, 0
-; COEXEC-NEXT: v_mov_b32_e32 v29, v0
+; COEXEC-NEXT: v_mov_b32_e32 v47, v0
; COEXEC-NEXT: s_cselect_b32 s0, -1, 0
-; COEXEC-NEXT: v_mov_b32_e32 v30, v0
+; COEXEC-NEXT: v_mov_b32_e32 v48, v0
; COEXEC-NEXT: s_xor_b32 s0, s0, -1
-; COEXEC-NEXT: v_mov_b32_e32 v31, v0
-; COEXEC-NEXT: v_cndmask_b32_e64 v32, 0, 1, s0
-; COEXEC-NEXT: v_cmp_ne_u32_e64 s0, 1, v32
+; COEXEC-NEXT: v_mov_b32_e32 v49, v0
+; COEXEC-NEXT: v_cndmask_b32_e64 v16, 0, 1, s0
+; COEXEC-NEXT: v_cmp_ne_u32_e64 s0, 1, v16
; COEXEC-NEXT: .LBB0_1: ; %loop
; COEXEC-NEXT: ; =>This Inner Loop Header: Depth=1
; COEXEC-NEXT: s_and_b32 vcc_lo, exec_lo, s0
-; COEXEC-NEXT: v_nop
-; COEXEC-NEXT: v_nop
-; COEXEC-NEXT: v_nop
-; COEXEC-NEXT: v_nop
-; COEXEC-NEXT: v_mov_b32_e32 v88, s2
+; COEXEC-NEXT: v_mov_b32_e32 v32, s2
; COEXEC-NEXT: s_add_co_i32 s2, s2, s1
-; COEXEC-NEXT: ds_load_tr16_b128 v[36:39], v88 offset:192
-; COEXEC-NEXT: ds_load_tr16_b128 v[40:43], v88
-; COEXEC-NEXT: ds_load_tr16_b128 v[44:47], v88 offset:64
-; COEXEC-NEXT: ds_load_tr16_b128 v[32:35], v88 offset:128
-; COEXEC-NEXT: ds_load_tr16_b128 v[52:55], v88 offset:448
-; COEXEC-NEXT: ds_load_tr16_b128 v[48:51], v88 offset:384
-; COEXEC-NEXT: ds_load_tr16_b128 v[56:59], v88 offset:256
-; COEXEC-NEXT: ds_load_tr16_b128 v[60:63], v88 offset:320
-; COEXEC-NEXT: ds_load_tr16_b128 v[68:71], v88 offset:704
-; COEXEC-NEXT: ds_load_tr16_b128 v[64:67], v88 offset:640
-; COEXEC-NEXT: ds_load_tr16_b128 v[76:79], v88 offset:576
-; COEXEC-NEXT: ds_load_tr16_b128 v[72:75], v88 offset:512
-; COEXEC-NEXT: ds_load_tr16_b128 v[84:87], v88 offset:960
-; COEXEC-NEXT: ds_load_tr16_b128 v[80:83], v88 offset:896
-; COEXEC-NEXT: ds_load_tr16_b128 v[92:95], v88 offset:832
-; COEXEC-NEXT: ds_load_tr16_b128 v[88:91], v88 offset:768
+; COEXEC-NEXT: ds_load_tr16_b128 v[20:23], v32 offset:192
+; COEXEC-NEXT: ds_load_tr16_b128 v[24:27], v32
+; COEXEC-NEXT: ds_load_tr16_b128 v[28:31], v32 offset:64
+; COEXEC-NEXT: ds_load_tr16_b128 v[16:19], v32 offset:128
+; COEXEC-NEXT: ...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/205530
More information about the llvm-commits
mailing list