[llvm] [AMDGPU] Balance VM_CNT histories across branches (PR #221115)

Pierre van Houtryve via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 8 01:54:40 PDT 2026


================
@@ -138,6 +146,336 @@ static bool updateVMCntOnly(const MachineInstr &Inst) {
          SIInstrInfo::isFLATGlobal(Inst) || SIInstrInfo::isFLATScratch(Inst);
 }
 
+// This optional pre-phase inserts NOPs to balance VM_CNT across branches. It
+// only supports gfx942/gfx950 targets. The NOPs improve performance by allowing
+// more relaxed `s_waitcnt vmcnt` instructions after branches. For example,
+// before padding the CFG might look like:
+//
+//             global_load_dword v0
+//                /          \
+//   global_load_dword v1   no VMEM
+//                \          /
+//              s_waitcnt vmcnt(0)
+//                   use v0
+//
+// Padding introduces `buffer_inv 0` to the no VMEM branch, allowing a relaxed
+// s_waitcnt:
+//
+//             global_load_dword v0
+//                /             \
+//   global_load_dword v1   buffer_inv 0
+//                \             /
+//              s_waitcnt vmcnt(1)
+//                   use v0
+//
+// This works because `buffer_inv 0` increments VM_CNT by 1 and is otherwise a
+// NOP.
+
+// Generation identifies the common baseline for EventCount; only generation
+// equality is meaningful.
+//
+//   common load             G1/E1
+//       /    \
+//    load    none           G1/E2, G1/E1: comparable
+//
+//   common load             G1/E1
+//       /    \
+// vmcnt(0); load  load      G2/E1, G1/E2: not comparable
+struct WaitcntPaddingState {
+  unsigned Generation = 0;
+  unsigned EventCount = 0;
+};
+
+struct WaitcntEdgePadding {
+  MachineBasicBlock *Pred = nullptr;
+  MachineBasicBlock *Succ = nullptr;
+  unsigned Count = 0;
+};
+
+struct WaitcntJoinPadding {
+  SmallVector<WaitcntEdgePadding, 2> Edges;
+};
+
+class SIWaitcntBranchPadding {
+  MachineFunction &MF;
+  MachineLoopInfo &MLI;
+  const GCNSubtarget &ST;
+  const SIInstrInfo &TII;
+  SmallVector<WaitcntJoinPadding, 4> Padding;
+  unsigned NextGeneration = 1;
+  bool ChangedCFG = false;
+
+  WaitcntPaddingState startNewGeneration() { return {NextGeneration++, 0}; }
+
+  unsigned getCounterMax() const {
+    AMDGPU::HardwareLimits Limits(AMDGPU::getIsaVersion(ST.getCPU()));
+    return Limits.LoadcntMax;
+  }
+
+  bool incrementsCounter(const MachineInstr &MI) const {
+    if (TII.isFLAT(MI))
+      return TII.mayAccessVMEMThroughFlat(MI);
+    if (!SIInstrInfo::isVMEM(MI) || !SIInstrInfo::usesVM_CNT(MI))
+      return false;
+    return !AMDGPU::getMUBUFIsBufferInv(MI.getOpcode()) ||
+           MI.getOpcode() == AMDGPU::BUFFER_INV ||
+           MI.getOpcode() == AMDGPU::BUFFER_WBL2;
+  }
+
+  // Waits establish a new relative baseline; calls and inline asm make the
+  // prior LOAD_CNT history unknowable.
+  bool startsNewGeneration(const MachineInstr &MI) const {
+    if (MI.isCall() || MI.isInlineAsm())
+      return true;
+
+    unsigned Opcode = SIInstrInfo::getNonSoftWaitcntOpcode(MI.getOpcode());
+    if (Opcode == AMDGPU::WAIT_ASYNCMARK)
+      return true;
+    if (Opcode == AMDGPU::S_WAITCNT_lds_direct)
+      return true;
+    if (Opcode == AMDGPU::S_WAITCNT) {
+      AMDGPU::Waitcnt Wait = AMDGPU::decodeWaitcnt(
+          AMDGPU::getIsaVersion(ST.getCPU()), MI.getOperand(0).getImm());
+      return Wait.get(AMDGPU::LOAD_CNT) != ~0u;
+    }
+
+    auto WaitCounter = AMDGPU::counterTypeForInstr(Opcode);
+    return WaitCounter && *WaitCounter == AMDGPU::LOAD_CNT;
+  }
+
+  WaitcntPaddingState transferBlock(MachineBasicBlock &MBB,
+                                    WaitcntPaddingState State,
+                                    unsigned MaxEventCount) {
+    for (MachineInstr &MI : MBB) {
+      if (startsNewGeneration(MI)) {
+        State = startNewGeneration();
+        continue;
+      }
+      if (!incrementsCounter(MI))
+        continue;
+      if (State.EventCount == MaxEventCount) {
+        // Make the overflowing event the new generation's implicit baseline.
+        // Its event count remains zero, while paths that did not execute it
+        // retain a different generation.
+        State = startNewGeneration();
+        continue;
+      }
+      ++State.EventCount;
+    }
+    return State;
+  }
+
+  void emitPadding(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsertPt,
+                   unsigned Count) const {
+    DebugLoc DL = MBB.findDebugLoc(InsertPt);
+    for (unsigned I = 0; I != Count; ++I)
+      BuildMI(MBB, InsertPt, DL, TII.get(AMDGPU::BUFFER_INV)).addImm(0);
+  }
+
+  bool plan() {
+    const bool IsEnabled =
+        EnableWaitcntBranchPadding.getNumOccurrences()
+            ? EnableWaitcntBranchPadding
+            : MF.getFunction()
+                  .getFnAttribute("amdgpu-waitcnt-branch-padding")
+                  .getValueAsBool();
+    if (!IsEnabled || !ST.hasGFX940Insts() || ST.hasVscnt() || !ST.isWave64() ||
+        ST.isPreciseMemoryEnabled() || MF.hasBBSections())
+      return false;
+
+    StringRef CPU = ST.getCPU();
+    if (CPU != "gfx942" && CPU != "gfx950")
----------------
Pierre-vh wrote:

I don't think this is how we do CPU generation checks?

https://github.com/llvm/llvm-project/pull/221115


More information about the llvm-commits mailing list