[llvm] [AMDGPU] Model waterfall loop EXEC update as a terminator (PR #219519)
Matt Arsenault via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 13 10:55:17 PDT 2026
================
@@ -604,6 +608,94 @@ bool SIOptimizeExecMasking::optimizeExecSequence() {
return Changed;
}
+// Fold
+//
+// sdst = S_ANDN2_B32 ssrc, exec
+// exec = COPY sdst
+// =>
+// sdst = S_ANDN2_WREXEC_B32 ssrc
+//
+// The waterfall loop emits the two operations separately so that spill code
+// for sdst can be inserted before exec is narrowed.
+bool SIOptimizeExecMasking::optimizeAndN2WrExecSequence(
+ MachineInstr &CopyToExecInst, Register Dst) const {
+ if (!ST->hasNoSdstCMPX() || TII->pseudoToMCOpcode(LMC.AndN2WrExecOpc) == -1)
+ return false;
+
+ MachineBasicBlock &MBB = *CopyToExecInst.getParent();
+
+ // Keep the fused instruction ahead of any trailing meta instructions (e.g.
+ // DBG_VALUEs of Dst), which emit no code, so that the sequence is unchanged
+ // in the common case where the two are already adjacent.
+ MachineBasicBlock::iterator InsertPt = CopyToExecInst.getIterator();
+ while (InsertPt != MBB.begin() && std::prev(InsertPt)->isMetaInstruction())
+ --InsertPt;
+
+ // Scan back for the s_andn2 computing the new exec value. The scheduler may
+ // have moved it away from the exec write, so allow instructions in between
+ // as long as sinking the s_andn2 past them is safe.
+ const unsigned SearchLimit = 20;
+ unsigned Count = 0;
+ MachineInstr *AndN2Inst = nullptr;
+ for (MachineInstr &MI : make_range(
+ std::next(MachineBasicBlock::reverse_iterator(CopyToExecInst)),
+ MBB.rend())) {
+ if (MI.isMetaInstruction())
+ continue;
+ if (++Count > SearchLimit)
+ return false;
+ if (MI.getOpcode() == LMC.AndN2Opc && MI.getOperand(0).getReg() == Dst) {
+ AndN2Inst = &MI;
+ break;
+ }
+ bool ReadsDst = false, ModifiesDst = false, ModifiesExec = false,
+ ReadsSCC = false;
+ for (const MachineOperand &MO : MI.operands()) {
+ if (!MO.isReg())
+ continue;
+ if (TRI->regsOverlap(MO.getReg(), Dst)) {
+ if (MO.isUse())
+ ReadsDst = true;
+ else
+ ModifiesDst = true;
+ }
+ if (!MO.isUse() && TRI->regsOverlap(MO.getReg(), LMC.ExecReg))
+ ModifiesExec = true;
+ if (MO.isUse() && TRI->regsOverlap(MO.getReg(), AMDGPU::SCC))
+ ReadsSCC = true;
+ }
+ if (ReadsDst || ModifiesDst || ModifiesExec || ReadsSCC)
+ return false;
+ }
----------------
arsenm wrote:
This should be its own helper function. You could also use the various MachineInstr reads/writes register functions, the main thing doing it directly here is you can check all 3 cases in one loop
https://github.com/llvm/llvm-project/pull/219519
More information about the llvm-commits
mailing list