[llvm] [WIP][SLP] Use costing to decide between Strided Loads and Compressed Loads (PR #226540)
Ryan Buchner via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 28 07:55:54 PDT 2026
https://github.com/bababuck updated https://github.com/llvm/llvm-project/pull/226540
>From f29c29bdedbad9a358e81c1302b99d063f261698 Mon Sep 17 00:00:00 2001
From: bababuck <buchner.ryan at gmail.com>
Date: Thu, 24 Sep 2026 10:25:16 -0700
Subject: [PATCH 1/2] [SLP] Use costing to decide between Strided Loads and
Compressed Loads
For small vectors will small strides, the vectorized load can be
represented as eithe a stride or a masked compress. Rather than
preferentially choose compression, decide based on the relative costs.
@alexey-bataev I'm running into an issue with this MR where during the
tree building phase (i.e. in `canVectorizeLoads`) we are missing
information about if the `VL` were uniquified from duplicates.
Normally, the costs are:
CompressedCost = MaskedLoadCost + ShuffleCost
StridedCost = StridedLoadCost
In this case, because there will need to be a shuffle to move the duplicated
value, but that shuffle can be folded with the CompressedVectorize shuffle (I
included an IR example below):
CompressedCost = MaskedLoadCost + ShuffleCost
StridedCost = StridedLoadCost + ShuffleCost
However, I don't think it would be easy to pass this information around everywhere to
make it available in `canVectorizeLoads`. So I'm thinking we should keep preferring
compressed loads in `canVectorizeLoads` and instead make this a `TransformNodes`
operation. Thoughts?
i.e.
```
Compressed (note the two shuffle can be folded):
%tmp0 = call <7 x i16> @llvm.masked.load.v7i16.p0(ptr align 2 null, <7 x i1> <i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true>, <7 x i16> poison)
%tmp1 = shufflevector <7 x i16> %tmp0, <7 x i16> poison, <3 x i32> <i32 0, i32 3, i32 6>
%tmp2 = shufflevector <3 x i16> %tmp1, <3 x i16> poison, <4 x i32> <i32 0, i32 0, i32 1, i32 2>
Strided:
%tmp0 = call <3 x i16> @llvm.experimental.vp.strided.load.v3i16.p0.i64(ptr align 2 null, i64 6, <3 x i1> splat (i1 true), i32 3)
%tmp1 = shufflevector <3 x i16> %tmp0, <3 x i16> poison, <4 x i32> <i32 0, i32 0, i32 1, i32 2>
```
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 59 ++++++++++++---
.../SLPVectorizer/RISCV/reductions.ll | 72 ++-----------------
.../RISCV/segmented-loads-simple.ll | 11 ++-
3 files changed, 60 insertions(+), 82 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index db949de5f91b8..79e4fec2df1ac 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -6241,19 +6241,60 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
// Check that the sorted loads are consecutive.
if (static_cast<uint64_t>(Diff) == Sz - 1)
return LoadsState::Vectorize;
- if (isMaskedLoadCompress(
- VL, PointerOps, Order, *TTI, *DL, *SE, *AC, *DT, *TLI, CostKind,
- [&](Value *V) {
- return areAllUsersVectorized(cast<Instruction>(V),
- UserIgnoreList);
- },
- SLPReVec))
+
+ bool IsMasked;
+ unsigned InterleaveFactor;
+ SmallVector<int> CompressMask;
+ VectorType *LoadVecTy;
+ bool IsMaskedLoadCompress = isMaskedLoadCompress(
+ VL, PointerOps, Order, *TTI, *DL, *SE, *AC, *DT, *TLI, CostKind,
+ [&](Value *V) {
+ return areAllUsersVectorized(cast<Instruction>(V), UserIgnoreList);
+ },
+ SLPReVec, IsMasked, InterleaveFactor, CompressMask, LoadVecTy);
+ // If needs re-ordered, then prefer compressed over strided since will have
+ // to shuffle either way. Interleaved loads are cheaper so prefer compressed
+ // in this case.
+ if (IsMaskedLoadCompress && (!Order.empty() || InterleaveFactor || !IsMasked))
return LoadsState::CompressVectorize;
+
// Widened strided loads must be legal for every group, not just the first
// pointer, which may have a stronger alignment than the remaining loads.
- if (analyzeConstantStrideCandidate(PointerOps, ScalarTy, CommonAlignment,
- Order, Diff, Ptr0, SPtrInfo))
+ bool IsStridedLoad = analyzeConstantStrideCandidate(
+ PointerOps, ScalarTy, CommonAlignment, Order, Diff, Ptr0, SPtrInfo);
+
+ if (IsMaskedLoadCompress && IsStridedLoad) {
+ const auto *LI0 = cast<LoadInst>(VL0);
+ InstructionCost StridedLoadCost =
+ TTI->getMemIntrinsicInstrCost(
+ MemIntrinsicCostAttributes(
+ Intrinsic::experimental_vp_strided_load, VecTy,
+ LI0->getPointerOperand(),
+ /*VariableMask=*/false, CommonAlignment),
+ CostKind);
+ FixedVectorType *StridedLoadTy = SPtrInfo.Ty;
+ if (StridedLoadTy != VecTy)
+ StridedLoadCost +=
+ TTI->getCastInstrCost(Instruction::BitCast, VecTy, StridedLoadTy,
+ TTI::CastContextHint::None, CostKind);
+
+ InstructionCost MaskedLoadCompressCost;
+ MaskedLoadCompressCost = TTI->getMemIntrinsicInstrCost(
+ MemIntrinsicCostAttributes(Intrinsic::masked_load, LoadVecTy,
+ LI0->getAlign(),
+ LI0->getPointerAddressSpace()),
+ CostKind);
+ MaskedLoadCompressCost +=
+ getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, LoadVecTy, CostKind,
+ CompressMask);
+ if (MaskedLoadCompressCost > StridedLoadCost)
+ return LoadsState::StridedVectorize;
+ return LoadsState::CompressVectorize;
+ } else if (IsMaskedLoadCompress) {
+ return LoadsState::CompressVectorize;
+ } else if (IsStridedLoad) {
return LoadsState::StridedVectorize;
+ }
}
if (!IsMaskedGatherLegal())
return LoadsState::Gather;
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/reductions.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/reductions.ll
index 6c1c3141cf2cc..d68ac4744293b 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/reductions.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/reductions.ll
@@ -181,73 +181,11 @@ entry:
define i64 @red_strided_ld_16xi64(ptr %ptr) {
-; ZVFHMIN-LABEL: @red_strided_ld_16xi64(
-; ZVFHMIN-NEXT: entry:
-; ZVFHMIN-NEXT: [[TMP0:%.*]] = call <16 x i64> @llvm.experimental.vp.strided.load.v16i64.p0.i64(ptr align 8 [[PTR:%.*]], i64 16, <16 x i1> splat (i1 true), i32 16)
-; ZVFHMIN-NEXT: [[TMP1:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP0]])
-; ZVFHMIN-NEXT: ret i64 [[TMP1]]
-;
-; ZVFHDEFAULT-LABEL: @red_strided_ld_16xi64(
-; ZVFHDEFAULT-NEXT: entry:
-; ZVFHDEFAULT-NEXT: [[TMP0:%.*]] = call <16 x i64> @llvm.experimental.vp.strided.load.v16i64.p0.i64(ptr align 8 [[PTR:%.*]], i64 16, <16 x i1> splat (i1 true), i32 16)
-; ZVFHDEFAULT-NEXT: [[TMP1:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP0]])
-; ZVFHDEFAULT-NEXT: ret i64 [[TMP1]]
-;
-; ZVFH256-LABEL: @red_strided_ld_16xi64(
-; ZVFH256-NEXT: entry:
-; ZVFH256-NEXT: [[TMP0:%.*]] = call <16 x i64> @llvm.experimental.vp.strided.load.v16i64.p0.i64(ptr align 8 [[PTR:%.*]], i64 16, <16 x i1> splat (i1 true), i32 16)
-; ZVFH256-NEXT: [[TMP1:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP0]])
-; ZVFH256-NEXT: ret i64 [[TMP1]]
-;
-; ZVFH512-LABEL: @red_strided_ld_16xi64(
-; ZVFH512-NEXT: entry:
-; ZVFH512-NEXT: [[LD0:%.*]] = load i64, ptr [[PTR:%.*]], align 8
-; ZVFH512-NEXT: [[GEP:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 2
-; ZVFH512-NEXT: [[LD1:%.*]] = load i64, ptr [[GEP]], align 8
-; ZVFH512-NEXT: [[ADD_1:%.*]] = add nuw nsw i64 [[LD0]], [[LD1]]
-; ZVFH512-NEXT: [[GEP_1:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 4
-; ZVFH512-NEXT: [[LD2:%.*]] = load i64, ptr [[GEP_1]], align 8
-; ZVFH512-NEXT: [[ADD_2:%.*]] = add nuw nsw i64 [[ADD_1]], [[LD2]]
-; ZVFH512-NEXT: [[GEP_2:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 6
-; ZVFH512-NEXT: [[LD3:%.*]] = load i64, ptr [[GEP_2]], align 8
-; ZVFH512-NEXT: [[ADD_3:%.*]] = add nuw nsw i64 [[ADD_2]], [[LD3]]
-; ZVFH512-NEXT: [[GEP_3:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 8
-; ZVFH512-NEXT: [[LD4:%.*]] = load i64, ptr [[GEP_3]], align 8
-; ZVFH512-NEXT: [[ADD_4:%.*]] = add nuw nsw i64 [[ADD_3]], [[LD4]]
-; ZVFH512-NEXT: [[GEP_4:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 10
-; ZVFH512-NEXT: [[LD5:%.*]] = load i64, ptr [[GEP_4]], align 8
-; ZVFH512-NEXT: [[ADD_5:%.*]] = add nuw nsw i64 [[ADD_4]], [[LD5]]
-; ZVFH512-NEXT: [[GEP_5:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 12
-; ZVFH512-NEXT: [[LD6:%.*]] = load i64, ptr [[GEP_5]], align 8
-; ZVFH512-NEXT: [[ADD_6:%.*]] = add nuw nsw i64 [[ADD_5]], [[LD6]]
-; ZVFH512-NEXT: [[GEP_6:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 14
-; ZVFH512-NEXT: [[LD7:%.*]] = load i64, ptr [[GEP_6]], align 8
-; ZVFH512-NEXT: [[ADD_7:%.*]] = add nuw nsw i64 [[ADD_6]], [[LD7]]
-; ZVFH512-NEXT: [[GEP_7:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 16
-; ZVFH512-NEXT: [[LD8:%.*]] = load i64, ptr [[GEP_7]], align 8
-; ZVFH512-NEXT: [[ADD_8:%.*]] = add nuw nsw i64 [[ADD_7]], [[LD8]]
-; ZVFH512-NEXT: [[GEP_8:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 18
-; ZVFH512-NEXT: [[LD9:%.*]] = load i64, ptr [[GEP_8]], align 8
-; ZVFH512-NEXT: [[ADD_9:%.*]] = add nuw nsw i64 [[ADD_8]], [[LD9]]
-; ZVFH512-NEXT: [[GEP_9:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 20
-; ZVFH512-NEXT: [[LD10:%.*]] = load i64, ptr [[GEP_9]], align 8
-; ZVFH512-NEXT: [[ADD_10:%.*]] = add nuw nsw i64 [[ADD_9]], [[LD10]]
-; ZVFH512-NEXT: [[GEP_10:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 22
-; ZVFH512-NEXT: [[LD11:%.*]] = load i64, ptr [[GEP_10]], align 8
-; ZVFH512-NEXT: [[ADD_11:%.*]] = add nuw nsw i64 [[ADD_10]], [[LD11]]
-; ZVFH512-NEXT: [[GEP_11:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 24
-; ZVFH512-NEXT: [[LD12:%.*]] = load i64, ptr [[GEP_11]], align 8
-; ZVFH512-NEXT: [[ADD_12:%.*]] = add nuw nsw i64 [[ADD_11]], [[LD12]]
-; ZVFH512-NEXT: [[GEP_12:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 26
-; ZVFH512-NEXT: [[LD13:%.*]] = load i64, ptr [[GEP_12]], align 8
-; ZVFH512-NEXT: [[ADD_13:%.*]] = add nuw nsw i64 [[ADD_12]], [[LD13]]
-; ZVFH512-NEXT: [[GEP_13:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 28
-; ZVFH512-NEXT: [[LD14:%.*]] = load i64, ptr [[GEP_13]], align 8
-; ZVFH512-NEXT: [[ADD_14:%.*]] = add nuw nsw i64 [[ADD_13]], [[LD14]]
-; ZVFH512-NEXT: [[GEP_14:%.*]] = getelementptr inbounds i64, ptr [[PTR]], i64 30
-; ZVFH512-NEXT: [[LD15:%.*]] = load i64, ptr [[GEP_14]], align 8
-; ZVFH512-NEXT: [[ADD_15:%.*]] = add nuw nsw i64 [[ADD_14]], [[LD15]]
-; ZVFH512-NEXT: ret i64 [[ADD_15]]
+; CHECK-LABEL: @red_strided_ld_16xi64(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = call <16 x i64> @llvm.experimental.vp.strided.load.v16i64.p0.i64(ptr align 8 [[PTR:%.*]], i64 16, <16 x i1> splat (i1 true), i32 16)
+; CHECK-NEXT: [[TMP1:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP0]])
+; CHECK-NEXT: ret i64 [[TMP1]]
;
entry:
%ld0 = load i64, ptr %ptr
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/segmented-loads-simple.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/segmented-loads-simple.ll
index 8497db2bd4344..cb1326610aae9 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/segmented-loads-simple.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/segmented-loads-simple.ll
@@ -58,12 +58,11 @@ define i32 @sum_of_abs_stride_3(ptr noalias %a, ptr noalias %b) {
; CHECK-LABEL: define i32 @sum_of_abs_stride_3
; CHECK-SAME: (ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = call <22 x i8> @llvm.masked.load.v22i8.p0(ptr align 1 [[A]], <22 x i1> <i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true>, <22 x i8> poison)
-; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <22 x i8> [[TMP0]], <22 x i8> poison, <8 x i32> <i32 0, i32 3, i32 6, i32 9, i32 12, i32 15, i32 18, i32 21>
-; CHECK-NEXT: [[TMP2:%.*]] = call <8 x i8> @llvm.abs.v8i8(<8 x i8> [[TMP1]], i1 false)
-; CHECK-NEXT: [[TMP3:%.*]] = sext <8 x i8> [[TMP2]] to <8 x i32>
-; CHECK-NEXT: [[TMP4:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP3]])
-; CHECK-NEXT: ret i32 [[TMP4]]
+; CHECK-NEXT: [[TMP0:%.*]] = call <8 x i8> @llvm.experimental.vp.strided.load.v8i8.p0.i64(ptr align 1 [[A]], i64 3, <8 x i1> splat (i1 true), i32 8)
+; CHECK-NEXT: [[TMP1:%.*]] = call <8 x i8> @llvm.abs.v8i8(<8 x i8> [[TMP0]], i1 false)
+; CHECK-NEXT: [[TMP2:%.*]] = sext <8 x i8> [[TMP1]] to <8 x i32>
+; CHECK-NEXT: [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP2]])
+; CHECK-NEXT: ret i32 [[TMP3]]
;
entry:
%0 = load i8, ptr %a, align 1
>From 9cc6365401b844c7b4d0d22c1af1bf38df8dcef8 Mon Sep 17 00:00:00 2001
From: bababuck <buchner.ryan at gmail.com>
Date: Fri, 25 Sep 2026 21:03:51 -0700
Subject: [PATCH 2/2] [SLP] Make compressed vs strided decision in
transformNodes()
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 192 ++++++++++--------
.../RISCV/basic-strided-loads.ll | 4 +-
.../SLPVectorizer/RISCV/revec-strided-load.ll | 4 +-
.../RISCV/segmented-loads-simple.ll | 11 +-
4 files changed, 119 insertions(+), 92 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 79e4fec2df1ac..4a67c63ce9e28 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -2524,6 +2524,11 @@ class slpvectorizer::BoUpSLP {
ArrayRef<Value *> VectorizedVals,
SmallPtrSetImpl<Value *> &CheckedExtracts);
+ InstructionCost
+ getCompressedLoadCost(ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
+ ArrayRef<unsigned> Order, const LoadInst *LI0,
+ const TreeEntry *CompressEntry = nullptr);
+
/// Estimates spill/reload cost from vector register pressure for \p E at the
/// point of emitting its vector result type \p FinalVecTy. \p ScalarTy is the
/// scalar/slot type used to widen into \p VecTy/\p FinalVecTy and may itself
@@ -6241,60 +6246,19 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
// Check that the sorted loads are consecutive.
if (static_cast<uint64_t>(Diff) == Sz - 1)
return LoadsState::Vectorize;
-
- bool IsMasked;
- unsigned InterleaveFactor;
- SmallVector<int> CompressMask;
- VectorType *LoadVecTy;
- bool IsMaskedLoadCompress = isMaskedLoadCompress(
- VL, PointerOps, Order, *TTI, *DL, *SE, *AC, *DT, *TLI, CostKind,
- [&](Value *V) {
- return areAllUsersVectorized(cast<Instruction>(V), UserIgnoreList);
- },
- SLPReVec, IsMasked, InterleaveFactor, CompressMask, LoadVecTy);
- // If needs re-ordered, then prefer compressed over strided since will have
- // to shuffle either way. Interleaved loads are cheaper so prefer compressed
- // in this case.
- if (IsMaskedLoadCompress && (!Order.empty() || InterleaveFactor || !IsMasked))
+ if (isMaskedLoadCompress(
+ VL, PointerOps, Order, *TTI, *DL, *SE, *AC, *DT, *TLI, CostKind,
+ [&](Value *V) {
+ return areAllUsersVectorized(cast<Instruction>(V),
+ UserIgnoreList);
+ },
+ SLPReVec))
return LoadsState::CompressVectorize;
-
// Widened strided loads must be legal for every group, not just the first
// pointer, which may have a stronger alignment than the remaining loads.
- bool IsStridedLoad = analyzeConstantStrideCandidate(
- PointerOps, ScalarTy, CommonAlignment, Order, Diff, Ptr0, SPtrInfo);
-
- if (IsMaskedLoadCompress && IsStridedLoad) {
- const auto *LI0 = cast<LoadInst>(VL0);
- InstructionCost StridedLoadCost =
- TTI->getMemIntrinsicInstrCost(
- MemIntrinsicCostAttributes(
- Intrinsic::experimental_vp_strided_load, VecTy,
- LI0->getPointerOperand(),
- /*VariableMask=*/false, CommonAlignment),
- CostKind);
- FixedVectorType *StridedLoadTy = SPtrInfo.Ty;
- if (StridedLoadTy != VecTy)
- StridedLoadCost +=
- TTI->getCastInstrCost(Instruction::BitCast, VecTy, StridedLoadTy,
- TTI::CastContextHint::None, CostKind);
-
- InstructionCost MaskedLoadCompressCost;
- MaskedLoadCompressCost = TTI->getMemIntrinsicInstrCost(
- MemIntrinsicCostAttributes(Intrinsic::masked_load, LoadVecTy,
- LI0->getAlign(),
- LI0->getPointerAddressSpace()),
- CostKind);
- MaskedLoadCompressCost +=
- getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, LoadVecTy, CostKind,
- CompressMask);
- if (MaskedLoadCompressCost > StridedLoadCost)
- return LoadsState::StridedVectorize;
- return LoadsState::CompressVectorize;
- } else if (IsMaskedLoadCompress) {
- return LoadsState::CompressVectorize;
- } else if (IsStridedLoad) {
+ if (analyzeConstantStrideCandidate(PointerOps, ScalarTy, CommonAlignment,
+ Order, Diff, Ptr0, SPtrInfo))
return LoadsState::StridedVectorize;
- }
}
if (!IsMaskedGatherLegal())
return LoadsState::Gather;
@@ -9340,6 +9304,47 @@ static bool allStructUsersAreExtractValueInsts(ArrayRef<Value *> VL) {
});
}
+InstructionCost BoUpSLP::getCompressedLoadCost(ArrayRef<Value *> VL,
+ ArrayRef<Value *> PointerOps,
+ ArrayRef<unsigned> Order,
+ const LoadInst *LI0,
+ const TreeEntry *CompressEntry) {
+ bool IsMasked;
+ unsigned InterleaveFactor;
+ SmallVector<int> CompressMask;
+ VectorType *LoadVecTy;
+ if (!isMaskedLoadCompress(
+ VL, PointerOps, Order, *TTI, *DL, *SE, *AC, *DT, *TLI, CostKind,
+ [](Value *) { return true; }, SLPReVec, IsMasked, InterleaveFactor,
+ CompressMask, LoadVecTy))
+ return InstructionCost::getInvalid();
+
+ if (CompressEntry)
+ CompressEntryToData.try_emplace(CompressEntry, CompressMask, LoadVecTy,
+ InterleaveFactor, IsMasked);
+
+ if (InterleaveFactor)
+ return TTI->getInterleavedMemoryOpCost(
+ Instruction::Load, LoadVecTy, InterleaveFactor, {}, LI0->getAlign(),
+ LI0->getPointerAddressSpace(), CostKind);
+
+ InstructionCost Cost;
+ if (IsMasked) {
+ Cost = TTI->getMemIntrinsicInstrCost(
+ MemIntrinsicCostAttributes(Intrinsic::masked_load, LoadVecTy,
+ LI0->getAlign(),
+ LI0->getPointerAddressSpace()),
+ CostKind);
+ } else {
+ Cost = TTI->getMemoryOpCost(Instruction::Load, LoadVecTy, LI0->getAlign(),
+ LI0->getPointerAddressSpace(), CostKind,
+ TTI::getOperandInfo(LI0->getPointerOperand()));
+ }
+ Cost += getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, LoadVecTy, CostKind,
+ CompressMask);
+ return Cost;
+}
+
BoUpSLP::TreeEntry::EntryState BoUpSLP::getScalarsVectorizationState(
const InstructionsState &S, ArrayRef<Value *> VL,
bool IsScatterVectorizeUserTE, OrdersType &CurrentOrder,
@@ -14403,6 +14408,58 @@ void BoUpSLP::transformNodes() {
continue;
switch (E.getOpcode()) {
case Instruction::Load: {
+ if (E.State == TreeEntry::CompressVectorize) {
+ StridedPtrInfo SPtrInfo;
+ auto PreferStridedOverCompressed = [&]() -> bool {
+ // In cases where a shuffle is mandatory regarless of load type,
+ // prefer Compressed since the reorder/reuse shuffle can be merged
+ // with the compress shuffle.
+ if (!E.ReorderIndices.empty() || !E.ReuseShuffleIndices.empty())
+ return false;
+ SmallVector<Value *> PointerOps(E.Scalars.size());
+ transform(E.Scalars, PointerOps.begin(), [](Value *V) {
+ return cast<LoadInst>(V)->getPointerOperand();
+ });
+
+ Type *ScalarTy = E.getMainOp()->getType();
+ Align CommonAlignment = computeCommonAlignment<LoadInst>(E.Scalars);
+ SmallVector<unsigned> Order;
+ std::optional<int64_t> Diff =
+ getPointersDiff(ScalarTy, PointerOps.front(), ScalarTy,
+ PointerOps.back(), *DL, *SE);
+ if (!Diff || !analyzeConstantStrideCandidate(
+ PointerOps, ScalarTy, CommonAlignment, Order, *Diff,
+ PointerOps.front(), SPtrInfo))
+ return false;
+
+ auto *LI0 = cast<LoadInst>(E.Scalars.front());
+ InstructionCost CompressedCost =
+ getCompressedLoadCost(E.Scalars, PointerOps, {}, LI0);
+ if (!CompressedCost.isValid())
+ return false;
+
+ auto *VecTy = cast<FixedVectorType>(
+ getWidenedType(ScalarTy, E.getVectorFactor()));
+ FixedVectorType *StridedLoadTy = SPtrInfo.Ty;
+ InstructionCost StridedCost = TTI->getMemIntrinsicInstrCost(
+ MemIntrinsicCostAttributes(
+ Intrinsic::experimental_vp_strided_load, StridedLoadTy,
+ LI0->getPointerOperand(), /*VariableMask=*/false,
+ CommonAlignment),
+ CostKind);
+ if (StridedLoadTy != VecTy)
+ StridedCost += TTI->getCastInstrCost(
+ Instruction::BitCast, VecTy, StridedLoadTy,
+ TTI::CastContextHint::None, CostKind);
+ return StridedCost < CompressedCost;
+ };
+
+ if (PreferStridedOverCompressed()) {
+ E.State = TreeEntry::StridedVectorize;
+ TreeEntryToStridedPtrInfoMap[&E] = SPtrInfo;
+ }
+ }
+
// No need to reorder masked gather loads, just reorder the scalar
// operands.
if (E.State != TreeEntry::Vectorize)
@@ -17166,10 +17223,6 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
break;
}
case TreeEntry::CompressVectorize: {
- bool IsMasked;
- unsigned InterleaveFactor;
- SmallVector<int> CompressMask;
- VectorType *LoadVecTy;
SmallVector<Value *> Scalars(VL);
if (!E->ReorderIndices.empty()) {
SmallVector<int> Mask(E->ReorderIndices.begin(),
@@ -17179,35 +17232,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
SmallVector<Value *> PointerOps(Scalars.size());
for (auto [I, V] : enumerate(Scalars))
PointerOps[I] = cast<LoadInst>(V)->getPointerOperand();
- [[maybe_unused]] bool IsVectorized = isMaskedLoadCompress(
- Scalars, PointerOps, E->ReorderIndices, *TTI, *DL, *SE, *AC, *DT,
- *TLI, CostKind, [](Value *) { return true; }, SLPReVec, IsMasked,
- InterleaveFactor, CompressMask, LoadVecTy);
- CompressEntryToData.try_emplace(E, CompressMask, LoadVecTy,
- InterleaveFactor, IsMasked);
- Align CommonAlignment = LI0->getAlign();
- if (InterleaveFactor) {
- VecLdCost = TTI->getInterleavedMemoryOpCost(
- Instruction::Load, LoadVecTy, InterleaveFactor, {},
- CommonAlignment, LI0->getPointerAddressSpace(), CostKind);
- } else if (IsMasked) {
- VecLdCost = TTI->getMemIntrinsicInstrCost(
- MemIntrinsicCostAttributes(Intrinsic::masked_load, LoadVecTy,
- CommonAlignment,
- LI0->getPointerAddressSpace()),
- CostKind);
- // TODO: include this cost into CommonCost.
- VecLdCost += getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, LoadVecTy,
- CostKind, CompressMask);
- } else {
- VecLdCost = TTI->getMemoryOpCost(
- Instruction::Load, LoadVecTy, CommonAlignment,
- LI0->getPointerAddressSpace(), CostKind,
- TTI::getOperandInfo(LI0->getPointerOperand()));
- // TODO: include this cost into CommonCost.
- VecLdCost += getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, LoadVecTy,
- CostKind, CompressMask);
- }
+ VecLdCost = getCompressedLoadCost(Scalars, PointerOps,
+ E->ReorderIndices, LI0, E);
break;
}
case TreeEntry::ScatterVectorize: {
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/basic-strided-loads.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/basic-strided-loads.ll
index b2729cba17a01..89992413b6ec9 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/basic-strided-loads.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/basic-strided-loads.ll
@@ -630,8 +630,8 @@ define void @constant_stride_masked_no_reordering(ptr %pl, i64 %stride, ptr %ps)
; CHECK-SAME: ptr [[PL:%.*]], i64 [[STRIDE:%.*]], ptr [[PS:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[GEP_L0:%.*]] = getelementptr inbounds i8, ptr [[PL]], i64 0
; CHECK-NEXT: [[GEP_S0:%.*]] = getelementptr inbounds i8, ptr [[PS]], i64 0
-; CHECK-NEXT: [[TMP1:%.*]] = call <28 x i8> @llvm.masked.load.v28i8.p0(ptr align 1 [[GEP_L0]], <28 x i1> <i1 true, i1 true, i1 true, i1 true, i1 false, i1 false, i1 false, i1 false, i1 true, i1 true, i1 true, i1 true, i1 false, i1 false, i1 false, i1 false, i1 true, i1 true, i1 true, i1 true, i1 false, i1 false, i1 false, i1 false, i1 true, i1 true, i1 true, i1 true>, <28 x i8> poison)
-; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <28 x i8> [[TMP1]], <28 x i8> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11, i32 16, i32 17, i32 18, i32 19, i32 24, i32 25, i32 26, i32 27>
+; CHECK-NEXT: [[TMP1:%.*]] = call <4 x i32> @llvm.experimental.vp.strided.load.v4i32.p0.i64(ptr align 1 [[GEP_L0]], i64 8, <4 x i1> splat (i1 true), i32 4)
+; CHECK-NEXT: [[TMP2:%.*]] = bitcast <4 x i32> [[TMP1]] to <16 x i8>
; CHECK-NEXT: store <16 x i8> [[TMP2]], ptr [[GEP_S0]], align 1
; CHECK-NEXT: ret void
;
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/revec-strided-load.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/revec-strided-load.ll
index 920f39c1cb762..f3080fb5fb523 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/revec-strided-load.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/revec-strided-load.ll
@@ -194,8 +194,8 @@ entry:
define void @non_aligned_stride_scalar(ptr %in0, ptr %out0) {
; CHECK-LABEL: @non_aligned_stride_scalar(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = call <5 x i8> @llvm.masked.load.v5i8.p0(ptr align 2 [[IN0:%.*]], <5 x i1> <i1 true, i1 true, i1 false, i1 true, i1 true>, <5 x i8> poison)
-; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <5 x i8> [[TMP0]], <5 x i8> poison, <4 x i32> <i32 0, i32 1, i32 3, i32 4>
+; CHECK-NEXT: [[TMP0:%.*]] = call <2 x i16> @llvm.experimental.vp.strided.load.v2i16.p0.i64(ptr align 2 [[IN0:%.*]], i64 3, <2 x i1> splat (i1 true), i32 2)
+; CHECK-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[TMP0]] to <4 x i8>
; CHECK-NEXT: store <4 x i8> [[TMP1]], ptr [[OUT0:%.*]], align 2
; CHECK-NEXT: ret void
;
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/segmented-loads-simple.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/segmented-loads-simple.ll
index cb1326610aae9..8497db2bd4344 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/segmented-loads-simple.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/segmented-loads-simple.ll
@@ -58,11 +58,12 @@ define i32 @sum_of_abs_stride_3(ptr noalias %a, ptr noalias %b) {
; CHECK-LABEL: define i32 @sum_of_abs_stride_3
; CHECK-SAME: (ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = call <8 x i8> @llvm.experimental.vp.strided.load.v8i8.p0.i64(ptr align 1 [[A]], i64 3, <8 x i1> splat (i1 true), i32 8)
-; CHECK-NEXT: [[TMP1:%.*]] = call <8 x i8> @llvm.abs.v8i8(<8 x i8> [[TMP0]], i1 false)
-; CHECK-NEXT: [[TMP2:%.*]] = sext <8 x i8> [[TMP1]] to <8 x i32>
-; CHECK-NEXT: [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP2]])
-; CHECK-NEXT: ret i32 [[TMP3]]
+; CHECK-NEXT: [[TMP0:%.*]] = call <22 x i8> @llvm.masked.load.v22i8.p0(ptr align 1 [[A]], <22 x i1> <i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true>, <22 x i8> poison)
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <22 x i8> [[TMP0]], <22 x i8> poison, <8 x i32> <i32 0, i32 3, i32 6, i32 9, i32 12, i32 15, i32 18, i32 21>
+; CHECK-NEXT: [[TMP2:%.*]] = call <8 x i8> @llvm.abs.v8i8(<8 x i8> [[TMP1]], i1 false)
+; CHECK-NEXT: [[TMP3:%.*]] = sext <8 x i8> [[TMP2]] to <8 x i32>
+; CHECK-NEXT: [[TMP4:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP3]])
+; CHECK-NEXT: ret i32 [[TMP4]]
;
entry:
%0 = load i8, ptr %a, align 1
More information about the llvm-commits
mailing list