[llvm] d83abac - [AMDGPU][InstCombine] Match ds_swizzle rotate mode for cyclic lane shuffles (#199004)

via llvm-commits llvm-commits at lists.llvm.org
Thu Jun 4 03:10:12 PDT 2026


Author: Barbara Mitic
Date: 2026-06-04T12:10:07+02:00
New Revision: d83abac82bd290e16a38ab79f6207f0bc3a07b3c

URL: https://github.com/llvm/llvm-project/commit/d83abac82bd290e16a38ab79f6207f0bc3a07b3c
DIFF: https://github.com/llvm/llvm-project/commit/d83abac82bd290e16a38ab79f6207f0bc3a07b3c.diff

LOG: [AMDGPU][InstCombine] Match ds_swizzle rotate mode for cyclic lane shuffles (#199004)

Follow-up to 17cc4f77109d [AMDGPU][InstCombine] Optimize constant
shuffle patterns (#192246).

Adds matchDsSwizzleRotatePattern to recognise shuffles of the form
dst_lane = (src_lane + N) % 32 (N in [1, 31]) and lower them to a single
ds_swizzle with rotate-mode encoding (ROTATE_MODE_ENC | N << 5),
available on GFX9+. The bitmask mode cannot express such rotations since
the carry between bit positions makes the per-bit mapping
non-independent. On wave64 the pattern is accepted only when
hasPeriodicLayout<32> confirms both 32-lane groups rotate by the same
amount. Wave32-only ID forms (mbcnt.lo alone) are correctly rejected on
wave64 targets.

Co-authored-by: Barbara Mitic <Barbara.Mitic at amd.com>

Added: 
    

Modified: 
    llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
    llvm/lib/Target/AMDGPU/GCNSubtarget.h
    llvm/test/Transforms/InstCombine/AMDGPU/wave-shuffle-patterns.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
index 2370c379e75f5..196a164fe602b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
@@ -883,6 +883,29 @@ matchDsSwizzleBitmaskPattern(ArrayRef<uint8_t> Ids) {
          XorMask << AMDGPU::Swizzle::BITMASK_XOR_SHIFT;
 }
 
+/// Match a GFX9+ DS_SWIZZLE rotate-mode permutation: a cyclic left-rotation
+/// of all 32 lanes within each 32-lane group by a constant N in [0, 31],
+/// i.e. dst_lane = (src_lane + N) % 32. On wave64, hasPeriodicLayout<32>
+/// ensures both 32-lane groups rotate by the same amount.
+static std::optional<unsigned>
+matchDsSwizzleRotatePattern(ArrayRef<uint8_t> Ids) {
+  if (!hasPeriodicLayout<32>(Ids))
+    return std::nullopt;
+
+  // Determine the rotation amount from lane 0: every lane must read from
+  // lane (I + N) % 32 where N = Ids[0] and 0 <= N <= 31.
+  unsigned N = Ids[0];
+  if (N >= 32)
+    return std::nullopt;
+
+  for (unsigned I = 0; I < 32; ++I)
+    if (Ids[I] != (I + N) % 32)
+      return std::nullopt;
+
+  return AMDGPU::Swizzle::ROTATE_MODE_ENC |
+         (N << AMDGPU::Swizzle::ROTATE_SIZE_SHIFT);
+}
+
 /// Emit v_mov_b32_dpp with the given control word, row/bank masks 0xF, and
 /// bound_ctrl=1 so out-of-bounds lanes are well-defined and the DPP mov can
 /// be folded into a consuming VALU op by GCNDPPCombine.
@@ -1005,6 +1028,13 @@ static Value *matchShuffleToHWIntrinsic(IRBuilderBase &B, Value *Src,
   if (std::optional<unsigned> Imm = matchDsSwizzleBitmaskPattern(Ids))
     return createDsSwizzle(B, Src, *Imm, DL);
 
+  // DS_SWIZZLE rotate mode (GFX9+): handles cyclic 32-lane rotations that
+  // bitmask mode cannot express (e.g. +1 mod 32 requires inter-bit carry).
+  if (ST.hasDsSwizzleRotateMode()) {
+    if (std::optional<unsigned> Imm = matchDsSwizzleRotatePattern(Ids))
+      return createDsSwizzle(B, Src, *Imm, DL);
+  }
+
   if (ST.hasPermLane64() && matchHalfWaveSwapPattern(Ids))
     return createPermlane64(B, Src);
 
@@ -1048,7 +1078,6 @@ tryOptimizeShufflePattern(InstCombiner &IC, IntrinsicInst &II,
 
   return IC.replaceInstUsesWith(II, Result);
 }
-
 std::optional<Instruction *>
 GCNTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
   Intrinsic::ID IID = II.getIntrinsicID();

diff  --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 082c5c33d6067..af47a8725c2d0 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -513,6 +513,10 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
   /// \returns true if the subtarget has the v_permlane64_b32 instruction.
   bool hasPermLane64() const { return getGeneration() >= GFX11; }
 
+  /// \returns true if the subtarget supports the ds_swizzle rotate and FFT
+  /// swizzle modes (GFX9+).
+  bool hasDsSwizzleRotateMode() const { return getGeneration() >= GFX9; }
+
   bool hasDPPRowShare() const {
     return HasDPP && (HasGFX90AInsts || getGeneration() >= GFX10);
   }

diff  --git a/llvm/test/Transforms/InstCombine/AMDGPU/wave-shuffle-patterns.ll b/llvm/test/Transforms/InstCombine/AMDGPU/wave-shuffle-patterns.ll
index 290c0f39f0b76..30c18ab5f34e9 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/wave-shuffle-patterns.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/wave-shuffle-patterns.ll
@@ -749,3 +749,285 @@ define i32 @test_broadcast_in_rows16_bitmask(i32 %val) {
   %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
   ret i32 %result
 }
+
+; Rotate-left-by-1 using mbcnt.lo only (wave32 ID form); folds to ds_swizzle
+; rotate mode (imm=0xC020) on GFX11 wave32, but not on wave64 targets (GFX11-W64,
+; GFX9) where mbcnt.lo alone is not the full lane ID.
+define i32 @test_rotate_add1_w32(i32 %val) {
+; GFX11-LABEL: @test_rotate_add1_w32(
+; GFX11-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX11-NEXT:    ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_add1_w32(
+; GFX11-W64-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 1
+; GFX11-W64-NEXT:    [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX11-W64-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT:    ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_add1_w32(
+; GFX9-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 1
+; GFX9-NEXT:    [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX9-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT:    ret i32 [[RESULT]]
+;
+  %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+  %add = add i32 %lane, 1
+  %idx = and i32 %add, 31
+  %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+  ret i32 %result
+}
+
+; Rotate-right-by-1 canonicalized to rotate-left-by-31; folds to ds_swizzle
+; rotate mode (imm=0xC3E0) on GFX11 wave32, but not on wave64 targets (GFX11-W64,
+; GFX9) where mbcnt.lo alone is not the full lane ID.
+define i32 @test_rotate_sub1_w32(i32 %val) {
+; GFX11-LABEL: @test_rotate_sub1_w32(
+; GFX11-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 50144)
+; GFX11-NEXT:    ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_sub1_w32(
+; GFX11-W64-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 31
+; GFX11-W64-NEXT:    [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX11-W64-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT:    ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_sub1_w32(
+; GFX9-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 31
+; GFX9-NEXT:    [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX9-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT:    ret i32 [[RESULT]]
+;
+  %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+  %add = add i32 %lane, 31
+  %idx = and i32 %add, 31
+  %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+  ret i32 %result
+}
+
+; Same rotate-left-by-1 via ds_bpermute (byte-addressed); folds to ds_swizzle
+; rotate mode (imm=0xC020) on GFX11 wave32, but not on wave64 targets (GFX11-W64,
+; GFX9) where mbcnt.lo alone is not the full lane ID.
+define i32 @test_bpermute_rotate_add1_w32(i32 %val) {
+; GFX11-LABEL: @test_bpermute_rotate_add1_w32(
+; GFX11-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX11-NEXT:    ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_bpermute_rotate_add1_w32(
+; GFX11-W64-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT:    [[ADD:%.*]] = shl nuw nsw i32 [[LANE]], 2
+; GFX11-W64-NEXT:    [[IDX:%.*]] = add nuw nsw i32 [[ADD]], 4
+; GFX11-W64-NEXT:    [[ADDR:%.*]] = and i32 [[IDX]], 124
+; GFX11-W64-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.bpermute(i32 [[ADDR]], i32 [[VAL:%.*]])
+; GFX11-W64-NEXT:    ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_bpermute_rotate_add1_w32(
+; GFX9-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT:    [[ADD:%.*]] = shl nuw nsw i32 [[LANE]], 2
+; GFX9-NEXT:    [[IDX:%.*]] = add nuw nsw i32 [[ADD]], 4
+; GFX9-NEXT:    [[ADDR:%.*]] = and i32 [[IDX]], 124
+; GFX9-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.bpermute(i32 [[ADDR]], i32 [[VAL:%.*]])
+; GFX9-NEXT:    ret i32 [[RESULT]]
+;
+  %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+  %add = add i32 %lane, 1
+  %idx = and i32 %add, 31
+  %addr = shl i32 %idx, 2
+  %result = call i32 @llvm.amdgcn.ds.bpermute(i32 %addr, i32 %val)
+  ret i32 %result
+}
+
+; Wave64 rotate-left-by-1 with both 32-lane groups rotating independently;
+; folds to ds_swizzle rotate mode on all targets.
+define i32 @test_rotate_add1_w64(i32 %val) {
+; GFX11-LABEL: @test_rotate_add1_w64(
+; GFX11-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX11-NEXT:    ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_add1_w64(
+; GFX11-W64-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX11-W64-NEXT:    ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_add1_w64(
+; GFX9-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX9-NEXT:    ret i32 [[RESULT]]
+;
+  %lo = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+  %lane = call i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 %lo)
+  %low5 = and i32 %lane, 31
+  %add = add i32 %low5, 1
+  %rot = and i32 %add, 31
+  %hi = and i32 %lane, 32
+  %idx = or i32 %rot, %hi
+  %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+  ret i32 %result
+}
+
+; Negative: full-wave64 rotate crosses the 32-lane group boundary, so
+; hasPeriodicLayout<32> rejects it and no fold occurs.
+define i32 @test_rotate_add1_full_w64(i32 %val) {
+; GFX11-LABEL: @test_rotate_add1_full_w64(
+; GFX11-NEXT:    [[LO:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LO]], 1
+; GFX11-NEXT:    [[IDX:%.*]] = and i32 [[ADD]], 63
+; GFX11-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-NEXT:    ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_add1_full_w64(
+; GFX11-W64-NEXT:    [[LO:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT:    [[LANE:%.*]] = call range(i32 0, 65) i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 [[LO]])
+; GFX11-W64-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 1
+; GFX11-W64-NEXT:    [[IDX:%.*]] = and i32 [[ADD]], 63
+; GFX11-W64-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT:    ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_add1_full_w64(
+; GFX9-NEXT:    [[LO:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT:    [[LANE:%.*]] = call range(i32 0, 65) i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 [[LO]])
+; GFX9-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 1
+; GFX9-NEXT:    [[IDX:%.*]] = and i32 [[ADD]], 63
+; GFX9-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT:    ret i32 [[RESULT]]
+;
+  %lo = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+  %lane = call i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 %lo)
+  %add = add i32 %lane, 1
+  %idx = and i32 %add, 63
+  %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+  ret i32 %result
+}
+
+; Identity shuffle (N=0 degenerate case for rotate; add %lane,0 folds to %lane).
+; On GFX11 wave32 this still folds to the identity quad-perm via DPP; on wave64
+; targets (GFX11-W64, GFX9) mbcnt.lo alone is not the full lane ID so no fold.
+define i32 @test_identity_w32(i32 %val) {
+; GFX11-LABEL: @test_identity_w32(
+; GFX11-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.update.dpp.i32(i32 poison, i32 [[VAL:%.*]], i32 228, i32 15, i32 15, i1 true)
+; GFX11-NEXT:    ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_identity_w32(
+; GFX11-W64-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT:    [[IDX:%.*]] = and i32 [[LANE]], 31
+; GFX11-W64-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT:    ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_identity_w32(
+; GFX9-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT:    [[IDX:%.*]] = and i32 [[LANE]], 31
+; GFX9-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT:    ret i32 [[RESULT]]
+;
+  %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+  %add = add i32 %lane, 0
+  %idx = and i32 %add, 31
+  %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+  ret i32 %result
+}
+
+; Two 32-lane halves rotate by 
diff erent amounts (low half by 1, high half by 2).
+; On wave64 hasPeriodicLayout<32> rejects the non-uniform pattern so no fold
+; occurs. On GFX11 wave32 only one 32-lane group exists so the high-half shift
+; is never evaluated; the pattern collapses to a uniform rotate-by-1 and folds.
+define i32 @test_rotate_asymmetric_w64(i32 %val) {
+; GFX11-LABEL: @test_rotate_asymmetric_w64(
+; GFX11-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX11-NEXT:    ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_asymmetric_w64(
+; GFX11-W64-NEXT:    [[LO:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT:    [[LANE:%.*]] = call range(i32 0, 65) i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 [[LO]])
+; GFX11-W64-NEXT:    [[HI:%.*]] = and i32 [[LANE]], 32
+; GFX11-W64-NEXT:    [[HIGH_EXTRA:%.*]] = lshr exact i32 [[HI]], 5
+; GFX11-W64-NEXT:    [[SHIFT:%.*]] = add nuw nsw i32 [[HIGH_EXTRA]], 1
+; GFX11-W64-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], [[SHIFT]]
+; GFX11-W64-NEXT:    [[ROT:%.*]] = and i32 [[ADD]], 31
+; GFX11-W64-NEXT:    [[IDX:%.*]] = or disjoint i32 [[ROT]], [[HI]]
+; GFX11-W64-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT:    ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_asymmetric_w64(
+; GFX9-NEXT:    [[LO:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT:    [[LANE:%.*]] = call range(i32 0, 65) i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 [[LO]])
+; GFX9-NEXT:    [[HI:%.*]] = and i32 [[LANE]], 32
+; GFX9-NEXT:    [[HIGH_EXTRA:%.*]] = lshr exact i32 [[HI]], 5
+; GFX9-NEXT:    [[SHIFT:%.*]] = add nuw nsw i32 [[HIGH_EXTRA]], 1
+; GFX9-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], [[SHIFT]]
+; GFX9-NEXT:    [[ROT:%.*]] = and i32 [[ADD]], 31
+; GFX9-NEXT:    [[IDX:%.*]] = or disjoint i32 [[ROT]], [[HI]]
+; GFX9-NEXT:    [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT:    ret i32 [[RESULT]]
+;
+  %lo = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+  %lane = call i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 %lo)
+  %low5 = and i32 %lane, 31
+  %hi = and i32 %lane, 32
+  %high_extra = lshr i32 %hi, 5
+  %shift = add i32 %high_extra, 1
+  %add = add i32 %low5, %shift
+  %rot = and i32 %add, 31
+  %idx = or i32 %rot, %hi
+  %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+  ret i32 %result
+}
+
+; Rotate-left-by-3 using an f32 source; createDsSwizzle bitcasts to i32.
+define float @test_rotate_add3_f32_w32(float %val) {
+; GFX11-LABEL: @test_rotate_add3_f32_w32(
+; GFX11-NEXT:    [[TMP1:%.*]] = bitcast float [[VAL:%.*]] to i32
+; GFX11-NEXT:    [[TMP2:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[TMP1]], i32 49248)
+; GFX11-NEXT:    [[RESULT:%.*]] = bitcast i32 [[TMP2]] to float
+; GFX11-NEXT:    ret float [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_add3_f32_w32(
+; GFX11-W64-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 3
+; GFX11-W64-NEXT:    [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX11-W64-NEXT:    [[RESULT:%.*]] = call float @llvm.amdgcn.wave.shuffle.f32(float [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT:    ret float [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_add3_f32_w32(
+; GFX9-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 3
+; GFX9-NEXT:    [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX9-NEXT:    [[RESULT:%.*]] = call float @llvm.amdgcn.wave.shuffle.f32(float [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT:    ret float [[RESULT]]
+;
+  %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+  %add = add i32 %lane, 3
+  %idx = and i32 %add, 31
+  %result = call float @llvm.amdgcn.wave.shuffle.f32(float %val, i32 %idx)
+  ret float %result
+}
+
+; Rotate-left-by-3 using a local-memory pointer source; createDsSwizzle
+; ptrtoint/inttoptr-converts around the i32 ds_swizzle.
+define ptr addrspace(3) @test_rotate_add3_ptr_w32(ptr addrspace(3) %val) {
+; GFX11-LABEL: @test_rotate_add3_ptr_w32(
+; GFX11-NEXT:    [[TMP1:%.*]] = ptrtoint ptr addrspace(3) [[VAL:%.*]] to i32
+; GFX11-NEXT:    [[TMP2:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[TMP1]], i32 49248)
+; GFX11-NEXT:    [[RESULT:%.*]] = inttoptr i32 [[TMP2]] to ptr addrspace(3)
+; GFX11-NEXT:    ret ptr addrspace(3) [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_add3_ptr_w32(
+; GFX11-W64-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 3
+; GFX11-W64-NEXT:    [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX11-W64-NEXT:    [[RESULT:%.*]] = call ptr addrspace(3) @llvm.amdgcn.wave.shuffle.p3(ptr addrspace(3) [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT:    ret ptr addrspace(3) [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_add3_ptr_w32(
+; GFX9-NEXT:    [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT:    [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 3
+; GFX9-NEXT:    [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX9-NEXT:    [[RESULT:%.*]] = call ptr addrspace(3) @llvm.amdgcn.wave.shuffle.p3(ptr addrspace(3) [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT:    ret ptr addrspace(3) [[RESULT]]
+;
+  %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+  %add = add i32 %lane, 3
+  %idx = and i32 %add, 31
+  %result = call ptr addrspace(3) @llvm.amdgcn.wave.shuffle.p3(ptr addrspace(3) %val, i32 %idx)
+  ret ptr addrspace(3) %result
+}


        


More information about the llvm-commits mailing list