[llvm] [AMDGPU] Balance VM_CNT histories across branches (PR #221115)
Pierre van Houtryve via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 8 01:54:40 PDT 2026
================
@@ -138,6 +146,336 @@ static bool updateVMCntOnly(const MachineInstr &Inst) {
SIInstrInfo::isFLATGlobal(Inst) || SIInstrInfo::isFLATScratch(Inst);
}
+// This optional pre-phase inserts NOPs to balance VM_CNT across branches. It
+// only supports gfx942/gfx950 targets. The NOPs improve performance by allowing
+// more relaxed `s_waitcnt vmcnt` instructions after branches. For example,
+// before padding the CFG might look like:
+//
+// global_load_dword v0
+// / \
+// global_load_dword v1 no VMEM
+// \ /
+// s_waitcnt vmcnt(0)
+// use v0
+//
+// Padding introduces `buffer_inv 0` to the no VMEM branch, allowing a relaxed
+// s_waitcnt:
+//
+// global_load_dword v0
+// / \
+// global_load_dword v1 buffer_inv 0
+// \ /
+// s_waitcnt vmcnt(1)
+// use v0
+//
+// This works because `buffer_inv 0` increments VM_CNT by 1 and is otherwise a
+// NOP.
+
+// Generation identifies the common baseline for EventCount; only generation
+// equality is meaningful.
+//
+// common load G1/E1
+// / \
+// load none G1/E2, G1/E1: comparable
+//
+// common load G1/E1
+// / \
+// vmcnt(0); load load G2/E1, G1/E2: not comparable
+struct WaitcntPaddingState {
+ unsigned Generation = 0;
+ unsigned EventCount = 0;
+};
+
+struct WaitcntEdgePadding {
+ MachineBasicBlock *Pred = nullptr;
+ MachineBasicBlock *Succ = nullptr;
+ unsigned Count = 0;
+};
+
+struct WaitcntJoinPadding {
+ SmallVector<WaitcntEdgePadding, 2> Edges;
+};
+
+class SIWaitcntBranchPadding {
+ MachineFunction &MF;
+ MachineLoopInfo &MLI;
+ const GCNSubtarget &ST;
+ const SIInstrInfo &TII;
+ SmallVector<WaitcntJoinPadding, 4> Padding;
+ unsigned NextGeneration = 1;
+ bool ChangedCFG = false;
+
+ WaitcntPaddingState startNewGeneration() { return {NextGeneration++, 0}; }
+
+ unsigned getCounterMax() const {
+ AMDGPU::HardwareLimits Limits(AMDGPU::getIsaVersion(ST.getCPU()));
+ return Limits.LoadcntMax;
+ }
+
+ bool incrementsCounter(const MachineInstr &MI) const {
+ if (TII.isFLAT(MI))
+ return TII.mayAccessVMEMThroughFlat(MI);
+ if (!SIInstrInfo::isVMEM(MI) || !SIInstrInfo::usesVM_CNT(MI))
+ return false;
+ return !AMDGPU::getMUBUFIsBufferInv(MI.getOpcode()) ||
+ MI.getOpcode() == AMDGPU::BUFFER_INV ||
+ MI.getOpcode() == AMDGPU::BUFFER_WBL2;
+ }
+
+ // Waits establish a new relative baseline; calls and inline asm make the
+ // prior LOAD_CNT history unknowable.
+ bool startsNewGeneration(const MachineInstr &MI) const {
+ if (MI.isCall() || MI.isInlineAsm())
+ return true;
+
+ unsigned Opcode = SIInstrInfo::getNonSoftWaitcntOpcode(MI.getOpcode());
+ if (Opcode == AMDGPU::WAIT_ASYNCMARK)
+ return true;
+ if (Opcode == AMDGPU::S_WAITCNT_lds_direct)
+ return true;
+ if (Opcode == AMDGPU::S_WAITCNT) {
+ AMDGPU::Waitcnt Wait = AMDGPU::decodeWaitcnt(
+ AMDGPU::getIsaVersion(ST.getCPU()), MI.getOperand(0).getImm());
+ return Wait.get(AMDGPU::LOAD_CNT) != ~0u;
+ }
+
+ auto WaitCounter = AMDGPU::counterTypeForInstr(Opcode);
+ return WaitCounter && *WaitCounter == AMDGPU::LOAD_CNT;
+ }
+
+ WaitcntPaddingState transferBlock(MachineBasicBlock &MBB,
+ WaitcntPaddingState State,
+ unsigned MaxEventCount) {
+ for (MachineInstr &MI : MBB) {
+ if (startsNewGeneration(MI)) {
+ State = startNewGeneration();
+ continue;
+ }
+ if (!incrementsCounter(MI))
+ continue;
+ if (State.EventCount == MaxEventCount) {
+ // Make the overflowing event the new generation's implicit baseline.
+ // Its event count remains zero, while paths that did not execute it
+ // retain a different generation.
+ State = startNewGeneration();
+ continue;
+ }
+ ++State.EventCount;
+ }
+ return State;
+ }
+
+ void emitPadding(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsertPt,
+ unsigned Count) const {
+ DebugLoc DL = MBB.findDebugLoc(InsertPt);
+ for (unsigned I = 0; I != Count; ++I)
+ BuildMI(MBB, InsertPt, DL, TII.get(AMDGPU::BUFFER_INV)).addImm(0);
+ }
+
+ bool plan() {
+ const bool IsEnabled =
+ EnableWaitcntBranchPadding.getNumOccurrences()
+ ? EnableWaitcntBranchPadding
+ : MF.getFunction()
+ .getFnAttribute("amdgpu-waitcnt-branch-padding")
+ .getValueAsBool();
+ if (!IsEnabled || !ST.hasGFX940Insts() || ST.hasVscnt() || !ST.isWave64() ||
+ ST.isPreciseMemoryEnabled() || MF.hasBBSections())
+ return false;
+
+ StringRef CPU = ST.getCPU();
+ if (CPU != "gfx942" && CPU != "gfx950")
----------------
Pierre-vh wrote:
I don't think this is how we do CPU generation checks?
https://github.com/llvm/llvm-project/pull/221115
More information about the llvm-commits
mailing list