[llvm] [AMDGPU] Combine redundant ballot intrinsic calls (PR #218357)

Shilei Tian via llvm-commits llvm-commits at lists.llvm.org
Mon Aug 24 08:51:35 PDT 2026


================
@@ -135,8 +145,149 @@ static bool optimizeUniformIntrinsic(IntrinsicInst &II,
   return false;
 }
 
+/// Maximum number of basic blocks inspected while proving that exec is
+/// invariant between two ballots. Keeps the walk below linear-per-pair in
+/// pathological CFGs.
+static constexpr unsigned MaxExecInvarianceBlocks = 100;
+
+/// Returns true if \p I may change exec, i.e. the set of lanes that are active
+/// when the following instructions execute.
+static bool isExecModifyingInst(const Instruction &I) {
+  const auto *CB = dyn_cast<CallBase>(&I);
+  if (!CB)
+    return false;
+
+  switch (CB->getIntrinsicID()) {
+  case Intrinsic::amdgcn_kill:
+  case Intrinsic::amdgcn_wqm_demote:
+  case Intrinsic::amdgcn_init_exec:
+  case Intrinsic::amdgcn_init_exec_from_input:
+  case Intrinsic::amdgcn_init_whole_wave:
+  // The WQM/WWM family makes the exec state at a given point depend on
+  // WQM/Exact decisions that SIWholeQuadMode only makes much later, so treat
+  // any occurrence of it as opaque.
+  case Intrinsic::amdgcn_wqm:
+  case Intrinsic::amdgcn_softwqm:
+  case Intrinsic::amdgcn_strict_wqm:
+  case Intrinsic::amdgcn_wwm:
+  case Intrinsic::amdgcn_strict_wwm:
+  case Intrinsic::amdgcn_set_inactive:
+  case Intrinsic::amdgcn_set_inactive_chain_arg:
+    return true;
+  case Intrinsic::not_intrinsic:
+    // A callee may itself execute llvm.amdgcn.kill.
+    return true;
+  default:
+    return false;
+  }
+}
+
+/// Returns true if exec is provably the same at \p A and at \p B, given that
+/// \p A dominates \p B. That holds when no path from \p A to \p B crosses a
+/// divergent terminator or an instruction that writes exec.
+static bool isExecInvariantBetween(const Instruction *A, const Instruction *B,
+                                   const UniformityInfo &UI) {
+  const BasicBlock *ABB = A->getParent();
+  const BasicBlock *BBB = B->getParent();
+
+  auto hasExecModifier = [](BasicBlock::const_iterator Begin,
+                            BasicBlock::const_iterator End) {
+    return any_of(make_range(Begin, End), isExecModifyingInst);
+  };
+
+  // Straight-line case: only the instructions in between can matter. Note that
+  // if this block is part of a cycle then A re-executes before B does, so the
+  // two still pair up within an iteration.
+  if (ABB == BBB)
+    return !hasExecModifier(std::next(A->getIterator()), B->getIterator());
+
+  // Everything in BBB ahead of B runs between A and B.
+  if (hasExecModifier(BBB->begin(), B->getIterator()))
+    return false;
+
+  // Collect every block on some path ABB ->* BBB that does not re-enter ABB.
+  // Since A dominates B, every such path starts at ABB, and every block found
+  // this way is dominated by ABB. Stopping the walk at ABB is correct: if a
+  // path did revisit ABB, then a later dynamic instance of A would be the one
+  // reaching B, and the path from that instance does not revisit ABB.
+  SmallPtrSet<const BasicBlock *, 8> Region;
+  SmallVector<const BasicBlock *, 8> Worklist;
+  Region.insert(ABB);
+  for (const BasicBlock *Pred : predecessors(BBB))
+    if (Region.insert(Pred).second)
+      Worklist.push_back(Pred);
+
+  while (!Worklist.empty()) {
+    if (Region.size() > MaxExecInvarianceBlocks)
+      return false;
+    const BasicBlock *Cur = Worklist.pop_back_val();
+    if (Cur == ABB)
+      continue;
+    for (const BasicBlock *Pred : predecessors(Cur))
+      if (Region.insert(Pred).second)
+        Worklist.push_back(Pred);
+  }
+
+  // Note that BBB lands in the region only if it can reach itself without
+  // passing through ABB, i.e. B sits in a cycle that A is outside of. In that
+  // case B re-executes and its whole block, terminator included, is checked
+  // below; otherwise BBB is absent and its terminator, which runs after B, is
+  // correctly left out.
+  for (const BasicBlock *Blk : Region) {
+    if (!UI.isUniformTerminator(Blk->getTerminator()))
+      return false;
+
+    // Instructions ahead of A never run between the last A and B.
+    BasicBlock::const_iterator Begin =
+        Blk == ABB ? std::next(A->getIterator()) : Blk->begin();
+    if (hasExecModifier(Begin, Blk->end()))
+      return false;
+  }
+  return true;
+}
+
+/// Removes ballot calls that are made redundant by an earlier identical call
+/// which is guaranteed to have executed with the same exec mask.
+static bool combineRedundantBallots(Function &F, UniformityInfo &UI,
+                                    const DominatorTree &DT) {
+  // Bucket the ballots by (result type, condition); only calls landing in the
+  // same bucket can possibly be identical. Visiting the dominator tree in
+  // pre-order means a dominating call always precedes the calls it dominates.
+  MapVector<std::pair<Type *, Value *>, SmallVector<CallInst *, 4>> Buckets;
+  for (const DomTreeNode *N : depth_first(DT.getRootNode()))
----------------
shiltian wrote:

https://llvm.org/docs/AMDGPU/DeveloperGuideline.html#use-of-braces

https://github.com/llvm/llvm-project/pull/218357


More information about the llvm-commits mailing list