[llvm] [AMDGPU] Add DPP16 Row Share optimization for llvm.amdgcn.wave.shuffle (PR #177470)
Jay Foad via llvm-commits
llvm-commits at lists.llvm.org
Fri Jan 30 01:46:35 PST 2026
================
@@ -553,6 +553,93 @@ static CallInst *rewriteCall(IRBuilderBase &B, CallInst &Old,
return NewCall;
}
+// Return true for sequences of instructions that effectively assign
+// each lane to its thread ID
+static bool isThreadID(const GCNSubtarget &ST, Value *V) {
+ // Case 1:
+ // wave32: mbcnt_lo(-1, 0)
+ // wave64: mbcnt_hi(-1, mbcnt_lo(-1, 0))
+ auto W32Pred = m_Intrinsic<Intrinsic::amdgcn_mbcnt_lo>(m_ConstantInt<-1>(),
+ m_ConstantInt<0>());
+ auto W64Pred = m_Intrinsic<Intrinsic::amdgcn_mbcnt_hi>(
+ m_ConstantInt<-1>(), m_Intrinsic<Intrinsic::amdgcn_mbcnt_lo>(
+ m_ConstantInt<-1>(), m_ConstantInt<0>()));
+ if (ST.isWave32() && match(V, W32Pred))
+ return true;
+ if (ST.isWave64() && match(V, W64Pred))
+ return true;
+
+ // Case 2:
+ // workitem.x()
+ auto WIdXPred = m_Intrinsic<Intrinsic::amdgcn_workitem_id_x>();
+ if (match(V, WIdXPred))
+ return true;
+
+ return false;
+}
+
+// Attempt to capture situations where the index argument matches
+// a DPP pattern, and convert to a DPP-based mov
+static std::optional<Instruction *>
+tryWaveShuffleDPP(const GCNSubtarget &ST, InstCombiner &IC, IntrinsicInst &II) {
+ Value *Val = II.getArgOperand(0);
+ Value *Idx = II.getArgOperand(1);
+ auto &B = IC.Builder;
+
+ // DPP16 Row Share requires GFX10 or later
+ if (ST.getGeneration() >= AMDGPUSubtarget::GFX10) {
+ Value *Tid;
+ uint64_t Mask;
+ uint64_t RowIdx = 0;
+ bool CanDPP16RowShare = false;
+
+ // DPP16 Row Share 0: Idx = Tid & Mask
+ auto RowShare0Pred = m_And(m_Value(Tid), m_ConstantInt(Mask));
+
+ // DPP16 Row Share (0 < Row < 15): Idx = (Tid & Mask) | RowIdx
+ auto RowSharePred =
+ m_Or(m_And(m_Value(Tid), m_ConstantInt(Mask)), m_ConstantInt(RowIdx));
+
+ // DPP16 Row Share 15: Idx = Tid | 0xF
+ auto RowShare15Pred = m_Or(m_Value(Tid), m_ConstantInt(RowIdx));
+
+ if (match(Idx, RowShare0Pred) && isThreadID(ST, Tid)) {
+ // wave32 requires Mask & 0x1F = 0x10
+ if (ST.isWave32() && (Mask & 0x1F) != 0x10)
+ return std::nullopt;
+ // wave64 requires Mask & 0x3F = 0x30
+ if (ST.isWave64() && (Mask & 0x3F) != 0x30)
+ return std::nullopt;
+ CanDPP16RowShare = true;
+ } else if (match(Idx, RowSharePred) && isThreadID(ST, Tid) && RowIdx < 15 &&
+ RowIdx > 0) {
+ // wave32 requires Mask & 0x1F = 0x10
+ if (ST.isWave32() && (Mask & 0x1F) != 0x10)
+ return std::nullopt;
+ // wave64 requires Mask & 0x3F = 0x30
+ if (ST.isWave64() && (Mask & 0x3F) != 0x30)
+ return std::nullopt;
+ CanDPP16RowShare = true;
+ } else if (match(Idx, RowShare15Pred) && isThreadID(ST, Tid) &&
+ RowIdx == 15) {
+ CanDPP16RowShare = true;
+ }
+
+ if (CanDPP16RowShare) {
+ CallInst *UpdateDPP = B.CreateIntrinsic(
+ Intrinsic::amdgcn_update_dpp, Val->getType(),
+ {B.getInt32(0), Val, B.getInt32(AMDGPU::DPP::ROW_SHR0 | RowIdx),
----------------
jayfoad wrote:
Also I think you can use poison instead of 0 for the first argument.
https://github.com/llvm/llvm-project/pull/177470
More information about the llvm-commits
mailing list