[llvm] d83abac - [AMDGPU][InstCombine] Match ds_swizzle rotate mode for cyclic lane shuffles (#199004)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Jun 4 03:10:12 PDT 2026
Author: Barbara Mitic
Date: 2026-06-04T12:10:07+02:00
New Revision: d83abac82bd290e16a38ab79f6207f0bc3a07b3c
URL: https://github.com/llvm/llvm-project/commit/d83abac82bd290e16a38ab79f6207f0bc3a07b3c
DIFF: https://github.com/llvm/llvm-project/commit/d83abac82bd290e16a38ab79f6207f0bc3a07b3c.diff
LOG: [AMDGPU][InstCombine] Match ds_swizzle rotate mode for cyclic lane shuffles (#199004)
Follow-up to 17cc4f77109d [AMDGPU][InstCombine] Optimize constant
shuffle patterns (#192246).
Adds matchDsSwizzleRotatePattern to recognise shuffles of the form
dst_lane = (src_lane + N) % 32 (N in [1, 31]) and lower them to a single
ds_swizzle with rotate-mode encoding (ROTATE_MODE_ENC | N << 5),
available on GFX9+. The bitmask mode cannot express such rotations since
the carry between bit positions makes the per-bit mapping
non-independent. On wave64 the pattern is accepted only when
hasPeriodicLayout<32> confirms both 32-lane groups rotate by the same
amount. Wave32-only ID forms (mbcnt.lo alone) are correctly rejected on
wave64 targets.
Co-authored-by: Barbara Mitic <Barbara.Mitic at amd.com>
Added:
Modified:
llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
llvm/lib/Target/AMDGPU/GCNSubtarget.h
llvm/test/Transforms/InstCombine/AMDGPU/wave-shuffle-patterns.ll
Removed:
################################################################################
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
index 2370c379e75f5..196a164fe602b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
@@ -883,6 +883,29 @@ matchDsSwizzleBitmaskPattern(ArrayRef<uint8_t> Ids) {
XorMask << AMDGPU::Swizzle::BITMASK_XOR_SHIFT;
}
+/// Match a GFX9+ DS_SWIZZLE rotate-mode permutation: a cyclic left-rotation
+/// of all 32 lanes within each 32-lane group by a constant N in [0, 31],
+/// i.e. dst_lane = (src_lane + N) % 32. On wave64, hasPeriodicLayout<32>
+/// ensures both 32-lane groups rotate by the same amount.
+static std::optional<unsigned>
+matchDsSwizzleRotatePattern(ArrayRef<uint8_t> Ids) {
+ if (!hasPeriodicLayout<32>(Ids))
+ return std::nullopt;
+
+ // Determine the rotation amount from lane 0: every lane must read from
+ // lane (I + N) % 32 where N = Ids[0] and 0 <= N <= 31.
+ unsigned N = Ids[0];
+ if (N >= 32)
+ return std::nullopt;
+
+ for (unsigned I = 0; I < 32; ++I)
+ if (Ids[I] != (I + N) % 32)
+ return std::nullopt;
+
+ return AMDGPU::Swizzle::ROTATE_MODE_ENC |
+ (N << AMDGPU::Swizzle::ROTATE_SIZE_SHIFT);
+}
+
/// Emit v_mov_b32_dpp with the given control word, row/bank masks 0xF, and
/// bound_ctrl=1 so out-of-bounds lanes are well-defined and the DPP mov can
/// be folded into a consuming VALU op by GCNDPPCombine.
@@ -1005,6 +1028,13 @@ static Value *matchShuffleToHWIntrinsic(IRBuilderBase &B, Value *Src,
if (std::optional<unsigned> Imm = matchDsSwizzleBitmaskPattern(Ids))
return createDsSwizzle(B, Src, *Imm, DL);
+ // DS_SWIZZLE rotate mode (GFX9+): handles cyclic 32-lane rotations that
+ // bitmask mode cannot express (e.g. +1 mod 32 requires inter-bit carry).
+ if (ST.hasDsSwizzleRotateMode()) {
+ if (std::optional<unsigned> Imm = matchDsSwizzleRotatePattern(Ids))
+ return createDsSwizzle(B, Src, *Imm, DL);
+ }
+
if (ST.hasPermLane64() && matchHalfWaveSwapPattern(Ids))
return createPermlane64(B, Src);
@@ -1048,7 +1078,6 @@ tryOptimizeShufflePattern(InstCombiner &IC, IntrinsicInst &II,
return IC.replaceInstUsesWith(II, Result);
}
-
std::optional<Instruction *>
GCNTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
Intrinsic::ID IID = II.getIntrinsicID();
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 082c5c33d6067..af47a8725c2d0 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -513,6 +513,10 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
/// \returns true if the subtarget has the v_permlane64_b32 instruction.
bool hasPermLane64() const { return getGeneration() >= GFX11; }
+ /// \returns true if the subtarget supports the ds_swizzle rotate and FFT
+ /// swizzle modes (GFX9+).
+ bool hasDsSwizzleRotateMode() const { return getGeneration() >= GFX9; }
+
bool hasDPPRowShare() const {
return HasDPP && (HasGFX90AInsts || getGeneration() >= GFX10);
}
diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/wave-shuffle-patterns.ll b/llvm/test/Transforms/InstCombine/AMDGPU/wave-shuffle-patterns.ll
index 290c0f39f0b76..30c18ab5f34e9 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/wave-shuffle-patterns.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/wave-shuffle-patterns.ll
@@ -749,3 +749,285 @@ define i32 @test_broadcast_in_rows16_bitmask(i32 %val) {
%result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
ret i32 %result
}
+
+; Rotate-left-by-1 using mbcnt.lo only (wave32 ID form); folds to ds_swizzle
+; rotate mode (imm=0xC020) on GFX11 wave32, but not on wave64 targets (GFX11-W64,
+; GFX9) where mbcnt.lo alone is not the full lane ID.
+define i32 @test_rotate_add1_w32(i32 %val) {
+; GFX11-LABEL: @test_rotate_add1_w32(
+; GFX11-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX11-NEXT: ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_add1_w32(
+; GFX11-W64-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 1
+; GFX11-W64-NEXT: [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX11-W64-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT: ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_add1_w32(
+; GFX9-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 1
+; GFX9-NEXT: [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX9-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT: ret i32 [[RESULT]]
+;
+ %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+ %add = add i32 %lane, 1
+ %idx = and i32 %add, 31
+ %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+ ret i32 %result
+}
+
+; Rotate-right-by-1 canonicalized to rotate-left-by-31; folds to ds_swizzle
+; rotate mode (imm=0xC3E0) on GFX11 wave32, but not on wave64 targets (GFX11-W64,
+; GFX9) where mbcnt.lo alone is not the full lane ID.
+define i32 @test_rotate_sub1_w32(i32 %val) {
+; GFX11-LABEL: @test_rotate_sub1_w32(
+; GFX11-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 50144)
+; GFX11-NEXT: ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_sub1_w32(
+; GFX11-W64-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 31
+; GFX11-W64-NEXT: [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX11-W64-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT: ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_sub1_w32(
+; GFX9-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 31
+; GFX9-NEXT: [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX9-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT: ret i32 [[RESULT]]
+;
+ %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+ %add = add i32 %lane, 31
+ %idx = and i32 %add, 31
+ %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+ ret i32 %result
+}
+
+; Same rotate-left-by-1 via ds_bpermute (byte-addressed); folds to ds_swizzle
+; rotate mode (imm=0xC020) on GFX11 wave32, but not on wave64 targets (GFX11-W64,
+; GFX9) where mbcnt.lo alone is not the full lane ID.
+define i32 @test_bpermute_rotate_add1_w32(i32 %val) {
+; GFX11-LABEL: @test_bpermute_rotate_add1_w32(
+; GFX11-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX11-NEXT: ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_bpermute_rotate_add1_w32(
+; GFX11-W64-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT: [[ADD:%.*]] = shl nuw nsw i32 [[LANE]], 2
+; GFX11-W64-NEXT: [[IDX:%.*]] = add nuw nsw i32 [[ADD]], 4
+; GFX11-W64-NEXT: [[ADDR:%.*]] = and i32 [[IDX]], 124
+; GFX11-W64-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.bpermute(i32 [[ADDR]], i32 [[VAL:%.*]])
+; GFX11-W64-NEXT: ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_bpermute_rotate_add1_w32(
+; GFX9-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT: [[ADD:%.*]] = shl nuw nsw i32 [[LANE]], 2
+; GFX9-NEXT: [[IDX:%.*]] = add nuw nsw i32 [[ADD]], 4
+; GFX9-NEXT: [[ADDR:%.*]] = and i32 [[IDX]], 124
+; GFX9-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.bpermute(i32 [[ADDR]], i32 [[VAL:%.*]])
+; GFX9-NEXT: ret i32 [[RESULT]]
+;
+ %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+ %add = add i32 %lane, 1
+ %idx = and i32 %add, 31
+ %addr = shl i32 %idx, 2
+ %result = call i32 @llvm.amdgcn.ds.bpermute(i32 %addr, i32 %val)
+ ret i32 %result
+}
+
+; Wave64 rotate-left-by-1 with both 32-lane groups rotating independently;
+; folds to ds_swizzle rotate mode on all targets.
+define i32 @test_rotate_add1_w64(i32 %val) {
+; GFX11-LABEL: @test_rotate_add1_w64(
+; GFX11-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX11-NEXT: ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_add1_w64(
+; GFX11-W64-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX11-W64-NEXT: ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_add1_w64(
+; GFX9-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX9-NEXT: ret i32 [[RESULT]]
+;
+ %lo = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+ %lane = call i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 %lo)
+ %low5 = and i32 %lane, 31
+ %add = add i32 %low5, 1
+ %rot = and i32 %add, 31
+ %hi = and i32 %lane, 32
+ %idx = or i32 %rot, %hi
+ %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+ ret i32 %result
+}
+
+; Negative: full-wave64 rotate crosses the 32-lane group boundary, so
+; hasPeriodicLayout<32> rejects it and no fold occurs.
+define i32 @test_rotate_add1_full_w64(i32 %val) {
+; GFX11-LABEL: @test_rotate_add1_full_w64(
+; GFX11-NEXT: [[LO:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LO]], 1
+; GFX11-NEXT: [[IDX:%.*]] = and i32 [[ADD]], 63
+; GFX11-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-NEXT: ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_add1_full_w64(
+; GFX11-W64-NEXT: [[LO:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT: [[LANE:%.*]] = call range(i32 0, 65) i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 [[LO]])
+; GFX11-W64-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 1
+; GFX11-W64-NEXT: [[IDX:%.*]] = and i32 [[ADD]], 63
+; GFX11-W64-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT: ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_add1_full_w64(
+; GFX9-NEXT: [[LO:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT: [[LANE:%.*]] = call range(i32 0, 65) i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 [[LO]])
+; GFX9-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 1
+; GFX9-NEXT: [[IDX:%.*]] = and i32 [[ADD]], 63
+; GFX9-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT: ret i32 [[RESULT]]
+;
+ %lo = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+ %lane = call i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 %lo)
+ %add = add i32 %lane, 1
+ %idx = and i32 %add, 63
+ %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+ ret i32 %result
+}
+
+; Identity shuffle (N=0 degenerate case for rotate; add %lane,0 folds to %lane).
+; On GFX11 wave32 this still folds to the identity quad-perm via DPP; on wave64
+; targets (GFX11-W64, GFX9) mbcnt.lo alone is not the full lane ID so no fold.
+define i32 @test_identity_w32(i32 %val) {
+; GFX11-LABEL: @test_identity_w32(
+; GFX11-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.update.dpp.i32(i32 poison, i32 [[VAL:%.*]], i32 228, i32 15, i32 15, i1 true)
+; GFX11-NEXT: ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_identity_w32(
+; GFX11-W64-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT: [[IDX:%.*]] = and i32 [[LANE]], 31
+; GFX11-W64-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT: ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_identity_w32(
+; GFX9-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT: [[IDX:%.*]] = and i32 [[LANE]], 31
+; GFX9-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT: ret i32 [[RESULT]]
+;
+ %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+ %add = add i32 %lane, 0
+ %idx = and i32 %add, 31
+ %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+ ret i32 %result
+}
+
+; Two 32-lane halves rotate by
diff erent amounts (low half by 1, high half by 2).
+; On wave64 hasPeriodicLayout<32> rejects the non-uniform pattern so no fold
+; occurs. On GFX11 wave32 only one 32-lane group exists so the high-half shift
+; is never evaluated; the pattern collapses to a uniform rotate-by-1 and folds.
+define i32 @test_rotate_asymmetric_w64(i32 %val) {
+; GFX11-LABEL: @test_rotate_asymmetric_w64(
+; GFX11-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[VAL:%.*]], i32 49184)
+; GFX11-NEXT: ret i32 [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_asymmetric_w64(
+; GFX11-W64-NEXT: [[LO:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT: [[LANE:%.*]] = call range(i32 0, 65) i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 [[LO]])
+; GFX11-W64-NEXT: [[HI:%.*]] = and i32 [[LANE]], 32
+; GFX11-W64-NEXT: [[HIGH_EXTRA:%.*]] = lshr exact i32 [[HI]], 5
+; GFX11-W64-NEXT: [[SHIFT:%.*]] = add nuw nsw i32 [[HIGH_EXTRA]], 1
+; GFX11-W64-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], [[SHIFT]]
+; GFX11-W64-NEXT: [[ROT:%.*]] = and i32 [[ADD]], 31
+; GFX11-W64-NEXT: [[IDX:%.*]] = or disjoint i32 [[ROT]], [[HI]]
+; GFX11-W64-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT: ret i32 [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_asymmetric_w64(
+; GFX9-NEXT: [[LO:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT: [[LANE:%.*]] = call range(i32 0, 65) i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 [[LO]])
+; GFX9-NEXT: [[HI:%.*]] = and i32 [[LANE]], 32
+; GFX9-NEXT: [[HIGH_EXTRA:%.*]] = lshr exact i32 [[HI]], 5
+; GFX9-NEXT: [[SHIFT:%.*]] = add nuw nsw i32 [[HIGH_EXTRA]], 1
+; GFX9-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], [[SHIFT]]
+; GFX9-NEXT: [[ROT:%.*]] = and i32 [[ADD]], 31
+; GFX9-NEXT: [[IDX:%.*]] = or disjoint i32 [[ROT]], [[HI]]
+; GFX9-NEXT: [[RESULT:%.*]] = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT: ret i32 [[RESULT]]
+;
+ %lo = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+ %lane = call i32 @llvm.amdgcn.mbcnt.hi(i32 -1, i32 %lo)
+ %low5 = and i32 %lane, 31
+ %hi = and i32 %lane, 32
+ %high_extra = lshr i32 %hi, 5
+ %shift = add i32 %high_extra, 1
+ %add = add i32 %low5, %shift
+ %rot = and i32 %add, 31
+ %idx = or i32 %rot, %hi
+ %result = call i32 @llvm.amdgcn.wave.shuffle.i32(i32 %val, i32 %idx)
+ ret i32 %result
+}
+
+; Rotate-left-by-3 using an f32 source; createDsSwizzle bitcasts to i32.
+define float @test_rotate_add3_f32_w32(float %val) {
+; GFX11-LABEL: @test_rotate_add3_f32_w32(
+; GFX11-NEXT: [[TMP1:%.*]] = bitcast float [[VAL:%.*]] to i32
+; GFX11-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[TMP1]], i32 49248)
+; GFX11-NEXT: [[RESULT:%.*]] = bitcast i32 [[TMP2]] to float
+; GFX11-NEXT: ret float [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_add3_f32_w32(
+; GFX11-W64-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 3
+; GFX11-W64-NEXT: [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX11-W64-NEXT: [[RESULT:%.*]] = call float @llvm.amdgcn.wave.shuffle.f32(float [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT: ret float [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_add3_f32_w32(
+; GFX9-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 3
+; GFX9-NEXT: [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX9-NEXT: [[RESULT:%.*]] = call float @llvm.amdgcn.wave.shuffle.f32(float [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT: ret float [[RESULT]]
+;
+ %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+ %add = add i32 %lane, 3
+ %idx = and i32 %add, 31
+ %result = call float @llvm.amdgcn.wave.shuffle.f32(float %val, i32 %idx)
+ ret float %result
+}
+
+; Rotate-left-by-3 using a local-memory pointer source; createDsSwizzle
+; ptrtoint/inttoptr-converts around the i32 ds_swizzle.
+define ptr addrspace(3) @test_rotate_add3_ptr_w32(ptr addrspace(3) %val) {
+; GFX11-LABEL: @test_rotate_add3_ptr_w32(
+; GFX11-NEXT: [[TMP1:%.*]] = ptrtoint ptr addrspace(3) [[VAL:%.*]] to i32
+; GFX11-NEXT: [[TMP2:%.*]] = call i32 @llvm.amdgcn.ds.swizzle(i32 [[TMP1]], i32 49248)
+; GFX11-NEXT: [[RESULT:%.*]] = inttoptr i32 [[TMP2]] to ptr addrspace(3)
+; GFX11-NEXT: ret ptr addrspace(3) [[RESULT]]
+;
+; GFX11-W64-LABEL: @test_rotate_add3_ptr_w32(
+; GFX11-W64-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX11-W64-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 3
+; GFX11-W64-NEXT: [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX11-W64-NEXT: [[RESULT:%.*]] = call ptr addrspace(3) @llvm.amdgcn.wave.shuffle.p3(ptr addrspace(3) [[VAL:%.*]], i32 [[IDX]])
+; GFX11-W64-NEXT: ret ptr addrspace(3) [[RESULT]]
+;
+; GFX9-LABEL: @test_rotate_add3_ptr_w32(
+; GFX9-NEXT: [[LANE:%.*]] = call range(i32 0, 33) i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+; GFX9-NEXT: [[ADD:%.*]] = add nuw nsw i32 [[LANE]], 3
+; GFX9-NEXT: [[IDX:%.*]] = and i32 [[ADD]], 31
+; GFX9-NEXT: [[RESULT:%.*]] = call ptr addrspace(3) @llvm.amdgcn.wave.shuffle.p3(ptr addrspace(3) [[VAL:%.*]], i32 [[IDX]])
+; GFX9-NEXT: ret ptr addrspace(3) [[RESULT]]
+;
+ %lane = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0)
+ %add = add i32 %lane, 3
+ %idx = and i32 %add, 31
+ %result = call ptr addrspace(3) @llvm.amdgcn.wave.shuffle.p3(ptr addrspace(3) %val, i32 %idx)
+ ret ptr addrspace(3) %result
+}
More information about the llvm-commits
mailing list