[llvm] [AMDGPU] Add DPP16 Row Share optimization for llvm.amdgcn.wave.shuffle (PR #177470)

Jay Foad via llvm-commits llvm-commits at lists.llvm.org
Fri Jan 30 01:46:35 PST 2026


================
@@ -553,6 +553,93 @@ static CallInst *rewriteCall(IRBuilderBase &B, CallInst &Old,
   return NewCall;
 }
 
+// Return true for sequences of instructions that effectively assign
+// each lane to its thread ID
+static bool isThreadID(const GCNSubtarget &ST, Value *V) {
+  // Case 1:
+  //   wave32: mbcnt_lo(-1, 0)
+  //   wave64: mbcnt_hi(-1, mbcnt_lo(-1, 0))
+  auto W32Pred = m_Intrinsic<Intrinsic::amdgcn_mbcnt_lo>(m_ConstantInt<-1>(),
+                                                         m_ConstantInt<0>());
+  auto W64Pred = m_Intrinsic<Intrinsic::amdgcn_mbcnt_hi>(
+      m_ConstantInt<-1>(), m_Intrinsic<Intrinsic::amdgcn_mbcnt_lo>(
+                               m_ConstantInt<-1>(), m_ConstantInt<0>()));
+  if (ST.isWave32() && match(V, W32Pred))
+    return true;
+  if (ST.isWave64() && match(V, W64Pred))
+    return true;
+
+  // Case 2:
+  //   workitem.x()
+  auto WIdXPred = m_Intrinsic<Intrinsic::amdgcn_workitem_id_x>();
+  if (match(V, WIdXPred))
+    return true;
+
+  return false;
+}
+
+// Attempt to capture situations where the index argument matches
+// a DPP pattern, and convert to a DPP-based mov
+static std::optional<Instruction *>
+tryWaveShuffleDPP(const GCNSubtarget &ST, InstCombiner &IC, IntrinsicInst &II) {
+  Value *Val = II.getArgOperand(0);
+  Value *Idx = II.getArgOperand(1);
+  auto &B = IC.Builder;
+
+  // DPP16 Row Share requires GFX10 or later
+  if (ST.getGeneration() >= AMDGPUSubtarget::GFX10) {
+    Value *Tid;
+    uint64_t Mask;
+    uint64_t RowIdx = 0;
+    bool CanDPP16RowShare = false;
+
+    // DPP16 Row Share 0: Idx = Tid & Mask
+    auto RowShare0Pred = m_And(m_Value(Tid), m_ConstantInt(Mask));
+
+    // DPP16 Row Share (0 < Row < 15): Idx = (Tid & Mask) | RowIdx
+    auto RowSharePred =
+        m_Or(m_And(m_Value(Tid), m_ConstantInt(Mask)), m_ConstantInt(RowIdx));
+
+    // DPP16 Row Share 15: Idx = Tid | 0xF
+    auto RowShare15Pred = m_Or(m_Value(Tid), m_ConstantInt(RowIdx));
+
+    if (match(Idx, RowShare0Pred) && isThreadID(ST, Tid)) {
+      // wave32 requires Mask & 0x1F = 0x10
+      if (ST.isWave32() && (Mask & 0x1F) != 0x10)
+        return std::nullopt;
+      // wave64 requires Mask & 0x3F = 0x30
+      if (ST.isWave64() && (Mask & 0x3F) != 0x30)
+        return std::nullopt;
+      CanDPP16RowShare = true;
+    } else if (match(Idx, RowSharePred) && isThreadID(ST, Tid) && RowIdx < 15 &&
+               RowIdx > 0) {
+      // wave32 requires Mask & 0x1F = 0x10
+      if (ST.isWave32() && (Mask & 0x1F) != 0x10)
+        return std::nullopt;
+      // wave64 requires Mask & 0x3F = 0x30
+      if (ST.isWave64() && (Mask & 0x3F) != 0x30)
+        return std::nullopt;
+      CanDPP16RowShare = true;
+    } else if (match(Idx, RowShare15Pred) && isThreadID(ST, Tid) &&
+               RowIdx == 15) {
+      CanDPP16RowShare = true;
+    }
+
+    if (CanDPP16RowShare) {
+      CallInst *UpdateDPP = B.CreateIntrinsic(
+          Intrinsic::amdgcn_update_dpp, Val->getType(),
+          {B.getInt32(0), Val, B.getInt32(AMDGPU::DPP::ROW_SHR0 | RowIdx),
----------------
jayfoad wrote:

Also I think you can use poison instead of 0 for the first argument.

https://github.com/llvm/llvm-project/pull/177470


More information about the llvm-commits mailing list