[llvm] [AMDGPU][SIInsertWaitcnts][NFC] Drop IsExpertMode and MaxCounter member variables (PR #181092)

via llvm-commits llvm-commits at lists.llvm.org
Wed Feb 11 22:12:32 PST 2026


llvmbot wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-amdgpu

Author: vporpo (vporpo)

<details>
<summary>Changes</summary>

`IsExpertMode` and `MaxCounter` can be trivially computed from MF and ST. So keeping them as member variables is kind of redundant. Replacing them with functions that compute them from MF and ST helps reduce the amount of state in the class.

---
Full diff: https://github.com/llvm/llvm-project/pull/181092.diff


1 Files Affected:

- (modified) llvm/lib/Target/AMDGPU/SIInsertWaitcnts.cpp (+49-38) 


``````````diff
diff --git a/llvm/lib/Target/AMDGPU/SIInsertWaitcnts.cpp b/llvm/lib/Target/AMDGPU/SIInsertWaitcnts.cpp
index 111867583fde3..4251916f94c45 100644
--- a/llvm/lib/Target/AMDGPU/SIInsertWaitcnts.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInsertWaitcnts.cpp
@@ -194,6 +194,22 @@ template <> struct enum_iteration_traits<WaitEventType> {
 
 namespace {
 
+bool isExpertMode(const MachineFunction &MF, const GCNSubtarget &ST) {
+  return ST.hasExpertSchedulingMode() &&
+                   (ExpertSchedulingModeFlag.getNumOccurrences()
+                        ? ExpertSchedulingModeFlag
+                        : MF.getFunction()
+                              .getFnAttribute("amdgpu-expert-scheduling-mode")
+                              .getValueAsBool());
+
+}
+
+InstCounterType getMaxCounter(const GCNSubtarget &ST, bool IsExpertMode) {
+  if (ST.hasExtendedWaitCounts())
+    return IsExpertMode ? NUM_EXPERT_INST_CNTS : NUM_EXTENDED_INST_CNTS;
+  return NUM_NORMAL_INST_CNTS;
+}
+
 /// Return an iterator over all events between VMEM_ACCESS (the first event)
 /// and \c MaxEvent (exclusive, default value yields an enumeration over
 /// all counters).
@@ -354,23 +370,23 @@ class WaitcntGenerator {
   const GCNSubtarget &ST;
   const SIInstrInfo &TII;
   AMDGPU::IsaVersion IV;
-  InstCounterType MaxCounter;
   bool OptNone;
   bool ExpandWaitcntProfiling = false;
   const AMDGPU::HardwareLimits *Limits = nullptr;
+  bool IsExpertMode = false;
 
 public:
   WaitcntGenerator() = delete;
   WaitcntGenerator(const WaitcntGenerator &) = delete;
-  WaitcntGenerator(const MachineFunction &MF, InstCounterType MaxCounter,
+  WaitcntGenerator(const MachineFunction &MF,
                    const AMDGPU::HardwareLimits *Limits)
       : ST(MF.getSubtarget<GCNSubtarget>()), TII(*ST.getInstrInfo()),
-        IV(AMDGPU::getIsaVersion(ST.getCPU())), MaxCounter(MaxCounter),
+        IV(AMDGPU::getIsaVersion(ST.getCPU())),
         OptNone(MF.getFunction().hasOptNone() ||
                 MF.getTarget().getOptLevel() == CodeGenOptLevel::None),
         ExpandWaitcntProfiling(
             MF.getFunction().hasFnAttribute("amdgpu-expand-waitcnt-profiling")),
-        Limits(Limits) {}
+        Limits(Limits), IsExpertMode(isExpertMode(MF, ST)) {}
 
   // Return true if the current function should be compiled with no
   // optimization.
@@ -441,7 +457,9 @@ class WaitcntGeneratorPreGFX12 final : public WaitcntGenerator {
           WaitEventSet()};
 
 public:
-  using WaitcntGenerator::WaitcntGenerator;
+  WaitcntGeneratorPreGFX12(MachineFunction *MF,
+                           const AMDGPU::HardwareLimits *Limits)
+      : WaitcntGenerator(*MF, Limits) {}
   bool
   applyPreexistingWaitcnt(WaitcntBrackets &ScoreBrackets,
                           MachineInstr &OldWaitcntInstr, AMDGPU::Waitcnt &Wait,
@@ -461,7 +479,6 @@ class WaitcntGeneratorPreGFX12 final : public WaitcntGenerator {
 
 class WaitcntGeneratorGFX12Plus final : public WaitcntGenerator {
 protected:
-  bool IsExpertMode;
   static constexpr const WaitEventSet
       WaitEventMaskForInstGFX12Plus[NUM_INST_CNTS] = {
           WaitEventSet({VMEM_ACCESS, GLOBAL_INV_ACCESS}),
@@ -480,10 +497,8 @@ class WaitcntGeneratorGFX12Plus final : public WaitcntGenerator {
 public:
   WaitcntGeneratorGFX12Plus() = delete;
   WaitcntGeneratorGFX12Plus(const MachineFunction &MF,
-                            InstCounterType MaxCounter,
-                            const AMDGPU::HardwareLimits *Limits,
-                            bool IsExpertMode)
-      : WaitcntGenerator(MF, MaxCounter, Limits), IsExpertMode(IsExpertMode) {}
+                            const AMDGPU::HardwareLimits *Limits)
+      : WaitcntGenerator(MF, Limits) {}
 
   bool
   applyPreexistingWaitcnt(WaitcntBrackets &ScoreBrackets,
@@ -515,8 +530,7 @@ class SIInsertWaitcnts {
   const SIRegisterInfo *TRI = nullptr;
   const MachineRegisterInfo *MRI = nullptr;
   InstCounterType SmemAccessCounter;
-  InstCounterType MaxCounter;
-  bool IsExpertMode = false;
+  const MachineFunction *MF = nullptr;
 
 private:
   DenseMap<const Value *, MachineBasicBlock *> SLoadAddresses;
@@ -1067,7 +1081,9 @@ bool WaitcntBrackets::hasPointSamplePendingVmemTypes(const MachineInstr &MI,
 
 void WaitcntBrackets::updateByEvent(WaitEventType E, MachineInstr &Inst) {
   InstCounterType T = Context->getCounterFromEvent(E);
-  assert(T < Context->MaxCounter);
+  assert(T < getMaxCounter(*Context->ST,
+                           isExpertMode(*Context->MF, *Context->ST)) &&
+         "T not supported in current mode!");
 
   unsigned UB = getScoreUB(T);
   unsigned CurrScore = UB + 1;
@@ -1287,7 +1303,8 @@ void WaitcntBrackets::recordAsyncMark(MachineInstr &Inst) {
 void WaitcntBrackets::print(raw_ostream &OS) const {
   const GCNSubtarget *ST = Context->ST;
 
-  for (auto T : inst_counter_types(Context->MaxCounter)) {
+  for (auto T : inst_counter_types(
+           getMaxCounter(*ST, isExpertMode(*Context->MF, *Context->ST)))) {
     unsigned SR = getScoreRange(T);
     switch (T) {
     case LOAD_CNT:
@@ -1745,7 +1762,7 @@ bool WaitcntGenerator::promoteSoftWaitCnt(MachineInstr *Waitcnt) const {
 bool WaitcntGeneratorPreGFX12::applyPreexistingWaitcnt(
     WaitcntBrackets &ScoreBrackets, MachineInstr &OldWaitcntInstr,
     AMDGPU::Waitcnt &Wait, MachineBasicBlock::instr_iterator It) const {
-  assert(isNormalMode(MaxCounter));
+  assert(isNormalMode(getMaxCounter(ST, IsExpertMode)));
 
   bool Modified = false;
   MachineInstr *WaitcntInstr = nullptr;
@@ -1867,7 +1884,7 @@ bool WaitcntGeneratorPreGFX12::applyPreexistingWaitcnt(
 bool WaitcntGeneratorPreGFX12::createNewWaitcnt(
     MachineBasicBlock &Block, MachineBasicBlock::instr_iterator It,
     AMDGPU::Waitcnt Wait, const WaitcntBrackets &ScoreBrackets) {
-  assert(isNormalMode(MaxCounter));
+  assert(isNormalMode(getMaxCounter(ST, IsExpertMode)));
 
   bool Modified = false;
   const DebugLoc &DL = Block.findDebugLoc(It);
@@ -1984,7 +2001,7 @@ WaitcntGeneratorGFX12Plus::getAllZeroWaitcnt(bool IncludeVSCnt) const {
 bool WaitcntGeneratorGFX12Plus::applyPreexistingWaitcnt(
     WaitcntBrackets &ScoreBrackets, MachineInstr &OldWaitcntInstr,
     AMDGPU::Waitcnt &Wait, MachineBasicBlock::instr_iterator It) const {
-  assert(!isNormalMode(MaxCounter));
+  assert(!isNormalMode(getMaxCounter(ST, IsExpertMode)));
 
   bool Modified = false;
   MachineInstr *CombinedLoadDsCntInstr = nullptr;
@@ -2258,7 +2275,7 @@ bool WaitcntGeneratorGFX12Plus::applyPreexistingWaitcnt(
 bool WaitcntGeneratorGFX12Plus::createNewWaitcnt(
     MachineBasicBlock &Block, MachineBasicBlock::instr_iterator It,
     AMDGPU::Waitcnt Wait, const WaitcntBrackets &ScoreBrackets) {
-  assert(!isNormalMode(MaxCounter));
+  assert(!isNormalMode(getMaxCounter(ST, IsExpertMode)));
 
   bool Modified = false;
   const DebugLoc &DL = Block.findDebugLoc(It);
@@ -2825,7 +2842,7 @@ void SIInsertWaitcnts::updateEventWaitcntAfter(MachineInstr &Inst,
   bool IsVMEMAccess = false;
   bool IsSMEMAccess = false;
 
-  if (IsExpertMode) {
+  if (isExpertMode(*MF, *ST)) {
     if (const auto ET = getExpertSchedulingEventType(Inst))
       ScoreBrackets->updateByEvent(*ET, Inst);
   }
@@ -2999,7 +3016,8 @@ bool WaitcntBrackets::mergeAsyncMarks(ArrayRef<MergeInfo> MergeInfos,
   unsigned MergeCount = std::min(OtherSize, OurSize);
   assert(OurSize == MaxSize);
   for (unsigned Idx = 1; Idx <= MergeCount; ++Idx) {
-    for (auto T : inst_counter_types(Context->MaxCounter)) {
+    for (auto T : inst_counter_types(getMaxCounter(
+             *Context->ST, isExpertMode(*Context->MF, *Context->ST)))) {
       StrictDom |= mergeScore(MergeInfos[T], AsyncMarks[OurSize - Idx][T],
                               OtherMarks[OtherSize - Idx][T]);
     }
@@ -3034,7 +3052,8 @@ bool WaitcntBrackets::merge(const WaitcntBrackets &Other) {
   // Array to store MergeInfo for each counter type
   MergeInfo MergeInfos[NUM_INST_CNTS];
 
-  for (auto T : inst_counter_types(Context->MaxCounter)) {
+  for (auto T : inst_counter_types(getMaxCounter(
+           *Context->ST, isExpertMode(*Context->MF, *Context->ST)))) {
     // Merge event flags for this counter
     const WaitEventSet &EventsForT = Context->getWaitEvents(T);
     const WaitEventSet OldEvents = PendingEvents & EventsForT;
@@ -3097,7 +3116,8 @@ bool WaitcntBrackets::merge(const WaitcntBrackets &Other) {
   }
 
   StrictDom |= mergeAsyncMarks(MergeInfos, Other.AsyncMarks);
-  for (auto T : inst_counter_types(Context->MaxCounter))
+  for (auto T : inst_counter_types(getMaxCounter(
+           *Context->ST, isExpertMode(*Context->MF, *Context->ST))))
     StrictDom |= mergeScore(MergeInfos[T], AsyncScore[T], Other.AsyncScore[T]);
 
   purgeEmptyTrackingData();
@@ -3202,8 +3222,8 @@ bool SIInsertWaitcnts::insertWaitcntInBlock(MachineFunction &MF,
 
     // Track pre-existing waitcnts that were added in earlier iterations or by
     // the memory legalizer.
-    if (isWaitInstr(Inst) ||
-        (IsExpertMode && Inst.getOpcode() == AMDGPU::S_WAITCNT_DEPCTR)) {
+    if (isWaitInstr(Inst) || (isExpertMode(MF, *ST) &&
+                              Inst.getOpcode() == AMDGPU::S_WAITCNT_DEPCTR)) {
       ++Iter;
       bool IsSoftXcnt = isSoftXcnt(Inst);
       // The Memory Legalizer conservatively inserts a soft xcnt before each
@@ -3557,6 +3577,7 @@ bool SIInsertWaitcnts::run(MachineFunction &MF) {
   TRI = &TII->getRegisterInfo();
   MRI = &MF.getRegInfo();
   const SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
+  this->MF = &MF;
 
   AMDGPU::IsaVersion IV = AMDGPU::getIsaVersion(ST->getCPU());
 
@@ -3564,21 +3585,11 @@ bool SIInsertWaitcnts::run(MachineFunction &MF) {
   Limits = AMDGPU::HardwareLimits(IV);
 
   if (ST->hasExtendedWaitCounts()) {
-    IsExpertMode = ST->hasExpertSchedulingMode() &&
-                   (ExpertSchedulingModeFlag.getNumOccurrences()
-                        ? ExpertSchedulingModeFlag
-                        : MF.getFunction()
-                              .getFnAttribute("amdgpu-expert-scheduling-mode")
-                              .getValueAsBool());
-    MaxCounter = IsExpertMode ? NUM_EXPERT_INST_CNTS : NUM_EXTENDED_INST_CNTS;
     if (!WCG)
-      WCG = std::make_unique<WaitcntGeneratorGFX12Plus>(MF, MaxCounter, &Limits,
-                                                        IsExpertMode);
+      WCG = std::make_unique<WaitcntGeneratorGFX12Plus>(MF, &Limits);
   } else {
-    MaxCounter = NUM_NORMAL_INST_CNTS;
     if (!WCG)
-      WCG = std::make_unique<WaitcntGeneratorPreGFX12>(MF, NUM_NORMAL_INST_CNTS,
-                                                       &Limits);
+      WCG = std::make_unique<WaitcntGeneratorPreGFX12>(&MF, &Limits);
   }
 
   for (auto T : inst_counter_types())
@@ -3617,7 +3628,7 @@ bool SIInsertWaitcnts::run(MachineFunction &MF) {
                 TII->get(instrsForExtendedCounterTypes[CT]))
             .addImm(0);
       }
-      if (IsExpertMode) {
+      if (isExpertMode(MF, *ST)) {
         unsigned Enc = AMDGPU::DepCtr::encodeFieldVaVdst(0, *ST);
         Enc = AMDGPU::DepCtr::encodeFieldVmVsrc(Enc, 0);
         BuildMI(EntryBB, I, DebugLoc(), TII->get(AMDGPU::S_WAITCNT_DEPCTR))
@@ -3756,7 +3767,7 @@ bool SIInsertWaitcnts::run(MachineFunction &MF) {
     }
   }
 
-  if (IsExpertMode) {
+  if (isExpertMode(MF, *ST)) {
     // Enable expert scheduling on function entry. To satisfy ABI requirements
     // and to allow calls between function with different expert scheduling
     // settings, disable it around calls and before returns.

``````````

</details>


https://github.com/llvm/llvm-project/pull/181092


More information about the llvm-commits mailing list