[llvm] [AMDGPU] Combine redundant ballot intrinsic calls (PR #218357)
Shilei Tian via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 24 08:51:35 PDT 2026
================
@@ -135,8 +145,149 @@ static bool optimizeUniformIntrinsic(IntrinsicInst &II,
return false;
}
+/// Maximum number of basic blocks inspected while proving that exec is
+/// invariant between two ballots. Keeps the walk below linear-per-pair in
+/// pathological CFGs.
+static constexpr unsigned MaxExecInvarianceBlocks = 100;
+
+/// Returns true if \p I may change exec, i.e. the set of lanes that are active
+/// when the following instructions execute.
+static bool isExecModifyingInst(const Instruction &I) {
+ const auto *CB = dyn_cast<CallBase>(&I);
+ if (!CB)
+ return false;
+
+ switch (CB->getIntrinsicID()) {
+ case Intrinsic::amdgcn_kill:
+ case Intrinsic::amdgcn_wqm_demote:
+ case Intrinsic::amdgcn_init_exec:
+ case Intrinsic::amdgcn_init_exec_from_input:
+ case Intrinsic::amdgcn_init_whole_wave:
+ // The WQM/WWM family makes the exec state at a given point depend on
+ // WQM/Exact decisions that SIWholeQuadMode only makes much later, so treat
+ // any occurrence of it as opaque.
+ case Intrinsic::amdgcn_wqm:
+ case Intrinsic::amdgcn_softwqm:
+ case Intrinsic::amdgcn_strict_wqm:
+ case Intrinsic::amdgcn_wwm:
+ case Intrinsic::amdgcn_strict_wwm:
+ case Intrinsic::amdgcn_set_inactive:
+ case Intrinsic::amdgcn_set_inactive_chain_arg:
+ return true;
+ case Intrinsic::not_intrinsic:
+ // A callee may itself execute llvm.amdgcn.kill.
+ return true;
+ default:
+ return false;
+ }
+}
+
+/// Returns true if exec is provably the same at \p A and at \p B, given that
+/// \p A dominates \p B. That holds when no path from \p A to \p B crosses a
+/// divergent terminator or an instruction that writes exec.
+static bool isExecInvariantBetween(const Instruction *A, const Instruction *B,
+ const UniformityInfo &UI) {
+ const BasicBlock *ABB = A->getParent();
+ const BasicBlock *BBB = B->getParent();
+
+ auto hasExecModifier = [](BasicBlock::const_iterator Begin,
+ BasicBlock::const_iterator End) {
+ return any_of(make_range(Begin, End), isExecModifyingInst);
+ };
+
+ // Straight-line case: only the instructions in between can matter. Note that
+ // if this block is part of a cycle then A re-executes before B does, so the
+ // two still pair up within an iteration.
+ if (ABB == BBB)
+ return !hasExecModifier(std::next(A->getIterator()), B->getIterator());
+
+ // Everything in BBB ahead of B runs between A and B.
+ if (hasExecModifier(BBB->begin(), B->getIterator()))
+ return false;
+
+ // Collect every block on some path ABB ->* BBB that does not re-enter ABB.
+ // Since A dominates B, every such path starts at ABB, and every block found
+ // this way is dominated by ABB. Stopping the walk at ABB is correct: if a
+ // path did revisit ABB, then a later dynamic instance of A would be the one
+ // reaching B, and the path from that instance does not revisit ABB.
+ SmallPtrSet<const BasicBlock *, 8> Region;
+ SmallVector<const BasicBlock *, 8> Worklist;
+ Region.insert(ABB);
+ for (const BasicBlock *Pred : predecessors(BBB))
+ if (Region.insert(Pred).second)
+ Worklist.push_back(Pred);
+
+ while (!Worklist.empty()) {
+ if (Region.size() > MaxExecInvarianceBlocks)
+ return false;
+ const BasicBlock *Cur = Worklist.pop_back_val();
+ if (Cur == ABB)
+ continue;
+ for (const BasicBlock *Pred : predecessors(Cur))
+ if (Region.insert(Pred).second)
+ Worklist.push_back(Pred);
+ }
+
+ // Note that BBB lands in the region only if it can reach itself without
+ // passing through ABB, i.e. B sits in a cycle that A is outside of. In that
+ // case B re-executes and its whole block, terminator included, is checked
+ // below; otherwise BBB is absent and its terminator, which runs after B, is
+ // correctly left out.
+ for (const BasicBlock *Blk : Region) {
+ if (!UI.isUniformTerminator(Blk->getTerminator()))
+ return false;
+
+ // Instructions ahead of A never run between the last A and B.
+ BasicBlock::const_iterator Begin =
+ Blk == ABB ? std::next(A->getIterator()) : Blk->begin();
+ if (hasExecModifier(Begin, Blk->end()))
+ return false;
+ }
+ return true;
+}
+
+/// Removes ballot calls that are made redundant by an earlier identical call
+/// which is guaranteed to have executed with the same exec mask.
+static bool combineRedundantBallots(Function &F, UniformityInfo &UI,
+ const DominatorTree &DT) {
+ // Bucket the ballots by (result type, condition); only calls landing in the
+ // same bucket can possibly be identical. Visiting the dominator tree in
+ // pre-order means a dominating call always precedes the calls it dominates.
+ MapVector<std::pair<Type *, Value *>, SmallVector<CallInst *, 4>> Buckets;
+ for (const DomTreeNode *N : depth_first(DT.getRootNode()))
----------------
shiltian wrote:
https://llvm.org/docs/AMDGPU/DeveloperGuideline.html#use-of-braces
https://github.com/llvm/llvm-project/pull/218357
More information about the llvm-commits
mailing list