[llvm] [AMDGPU] Model waterfall loop EXEC update as a terminator (PR #219519)
Arseniy Obolenskiy via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 14 01:02:35 PDT 2026
================
@@ -604,6 +609,86 @@ bool SIOptimizeExecMasking::optimizeExecSequence() {
return Changed;
}
+bool SIOptimizeExecMasking::blocksAndN2Sink(const MachineInstr &MI,
+ Register Dst) const {
+ for (const MachineOperand &MO : MI.operands()) {
+ if (!MO.isReg())
+ continue;
+ // Any access to Dst would see a stale value once the def sinks past it, a
+ // use of SCC would see the s_andn2 result, and a redefinition of EXEC
+ // would change what the s_andn2 computes.
+ if (TRI->regsOverlap(MO.getReg(), Dst))
+ return true;
+ if (MO.isUse() && TRI->regsOverlap(MO.getReg(), AMDGPU::SCC))
+ return true;
+ if (MO.isDef() && TRI->regsOverlap(MO.getReg(), LMC.ExecReg))
+ return true;
+ }
+ return false;
+}
+
+// Fold
+//
+// sdst = S_ANDN2_B32 ssrc, exec
+// exec = COPY sdst
+// =>
+// sdst = S_ANDN2_WREXEC_B32 ssrc
+//
+// The waterfall loop emits the two operations separately so that spill code
+// for sdst can be inserted before exec is narrowed.
+bool SIOptimizeExecMasking::optimizeAndN2WrExecSequence(
+ MachineInstr &CopyToExecInst, Register Dst) const {
+ if (!ST->hasNoSdstCMPX() || TII->pseudoToMCOpcode(LMC.AndN2WrExecOpc) == -1)
+ return false;
+
+ MachineBasicBlock &MBB = *CopyToExecInst.getParent();
+
+ // Keep the fused instruction ahead of any trailing debug instructions, so
+ // DBG_VALUEs of Dst stay after its def.
+ MachineBasicBlock::iterator InsertPt = CopyToExecInst.getIterator();
+ while (InsertPt != MBB.begin() && std::prev(InsertPt)->isDebugInstr())
+ --InsertPt;
----------------
aobolensk wrote:
done
https://github.com/llvm/llvm-project/pull/219519
More information about the llvm-commits
mailing list