[llvm] [SLP][Modularisation][NFC] Extract free cost helpers into SLPCostAnalysis (PR #210278)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Jul 17 02:56:48 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-vectorizers
Author: Madhur Amilkanthwar (madhur13490)
<details>
<summary>Changes</summary>
Move the BoUpSLP-independent cost helpers out of SLPVectorizer.cpp into SLPVectorizer/SLPCostAnalysis.{h,cpp} (namespace llvm::slpvectorizer).
Moved:
* getShuffleCost
* getGEPCosts
RFC: https://discourse.llvm.org/t/modularizing-slpvectorizer-cpp/90922
---
Patch is 43.39 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/210278.diff
4 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/CMakeLists.txt (+1)
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp (+87-206)
- (added) llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.cpp (+131)
- (added) llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCostAnalysis.h (+53)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/CMakeLists.txt b/llvm/lib/Transforms/Vectorize/CMakeLists.txt
index 6e26203d957cb..ac91401736283 100644
--- a/llvm/lib/Transforms/Vectorize/CMakeLists.txt
+++ b/llvm/lib/Transforms/Vectorize/CMakeLists.txt
@@ -23,6 +23,7 @@ add_llvm_component_library(LLVMVectorize
SandboxVectorizer/Scheduler.cpp
SandboxVectorizer/SeedCollector.cpp
SandboxVectorizer/VecUtils.cpp
+ SLPVectorizer/SLPCostAnalysis.cpp
SLPVectorizer/SLPUtils.cpp
SLPVectorizer.cpp
Vectorize.cpp
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index b76cb80426456..fd37429ec6ef8 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -17,6 +17,7 @@
//===----------------------------------------------------------------------===//
#include "llvm/Transforms/Vectorize/SLPVectorizer.h"
+#include "SLPVectorizer/SLPCostAnalysis.h"
#include "SLPVectorizer/SLPUtils.h"
#include "llvm/ADT/DenseMap.h"
#include "llvm/ADT/DenseSet.h"
@@ -6990,40 +6991,6 @@ static const SCEV *calculateRtStride(ArrayRef<Value *> PointerOps, Type *ElemTy,
return Stride;
}
-static std::pair<InstructionCost, InstructionCost>
-getGEPCosts(const TargetTransformInfo &TTI, ArrayRef<Value *> Ptrs,
- Value *BasePtr, unsigned Opcode, TTI::TargetCostKind CostKind,
- Type *ScalarTy, VectorType *VecTy);
-
-/// Returns the cost of the shuffle instructions with the given \p Kind, vector
-/// type \p Tp and optional \p Mask. Adds SLP-specifc cost estimation for insert
-/// subvector pattern.
-static InstructionCost
-getShuffleCost(const TargetTransformInfo &TTI, TTI::ShuffleKind Kind,
- VectorType *Tp, ArrayRef<int> Mask = {},
- TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput,
- int Index = 0, VectorType *SubTp = nullptr,
- ArrayRef<const Value *> Args = {}) {
- VectorType *DstTy = Tp;
- if (!Mask.empty())
- DstTy = FixedVectorType::get(Tp->getScalarType(), Mask.size());
-
- if (Kind != TTI::SK_PermuteTwoSrc)
- return TTI.getShuffleCost(Kind, DstTy, Tp, Mask, CostKind, Index, SubTp,
- Args);
- int NumSrcElts = Tp->getElementCount().getKnownMinValue();
- int NumSubElts;
- if (Mask.size() > 2 && ShuffleVectorInst::isInsertSubvectorMask(
- Mask, NumSrcElts, NumSubElts, Index)) {
- if (Index + NumSubElts > NumSrcElts &&
- Index + NumSrcElts <= static_cast<int>(Mask.size()))
- return TTI.getShuffleCost(TTI::SK_InsertSubvector, DstTy, Tp, Mask,
- TTI::TCK_RecipThroughput, Index, Tp);
- }
- return TTI.getShuffleCost(Kind, DstTy, Tp, Mask, CostKind, Index, SubTp,
- Args);
-}
-
/// This is similar to TargetTransformInfo::getScalarizationOverhead, but if
/// ScalarTy is a FixedVectorType, a vector will be inserted or extracted
/// instead of a scalar.
@@ -7297,7 +7264,7 @@ static bool isMaskedLoadCompress(
// shuffle cost estimation in that case.
if (VectorGEPCost + LoadCost >= GatherCost)
return false;
- InstructionCost CompressCost = ::getShuffleCost(
+ InstructionCost CompressCost = getShuffleCost(
TTI, TTI::SK_PermuteSingleSrc, LoadVecTy, CompressMask, CostKind);
if (!Order.empty()) {
SmallVector<int> NewMask(Sz, PoisonMaskElem);
@@ -7867,7 +7834,7 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
getScalarizationOverhead(
TTI, PtrScalarTy, PtrVecTy, APInt::getOneBitSet(Sz, 0),
/*Insert=*/true, /*Extract=*/false, CostKind) +
- ::getShuffleCost(TTI, TTI::SK_Broadcast, PtrVecTy, {}, CostKind);
+ getShuffleCost(TTI, TTI::SK_Broadcast, PtrVecTy, {}, CostKind);
// The cost of scalar loads.
InstructionCost ScalarLoadsCost =
accumulate(VL, InstructionCost(),
@@ -7972,8 +7939,7 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
getScalarizationOverhead(
TTI, ScalarTy, SubVecTy, APInt::getOneBitSet(SliceVF, 0),
/*Insert=*/true, /*Extract=*/false, CostKind) +
- ::getShuffleCost(TTI, TTI::SK_Broadcast, SubVecTy, {},
- CostKind);
+ getShuffleCost(TTI, TTI::SK_Broadcast, SubVecTy, {}, CostKind);
}
switch (LS) {
case LoadsState::Vectorize:
@@ -7998,8 +7964,8 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
Intrinsic::masked_load, SubVecTy,
CommonAlignment, LI0->getPointerAddressSpace()),
CostKind) +
- ::getShuffleCost(TTI, TTI::SK_PermuteSingleSrc, SubVecTy,
- {}, CostKind);
+ getShuffleCost(TTI, TTI::SK_PermuteSingleSrc, SubVecTy,
+ {}, CostKind);
break;
case LoadsState::ScatterVectorize:
VecLdCost += TTI.getMemIntrinsicInstrCost(
@@ -8019,8 +7985,8 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
ShuffleMask[Idx] = Idx / VF == SliceIdx ? VL.size() + Idx % VF : Idx;
if (SliceStart > 0)
VecLdCost +=
- ::getShuffleCost(TTI, TTI::SK_InsertSubvector, VecTy, ShuffleMask,
- CostKind, SliceStart, SubVecTy);
+ getShuffleCost(TTI, TTI::SK_InsertSubvector, VecTy, ShuffleMask,
+ CostKind, SliceStart, SubVecTy);
}
// If masked gather cost is higher - better to vectorize, so
// consider it as a gather node. It will be better estimated
@@ -8566,7 +8532,7 @@ BoUpSLP::getReorderingData(const TreeEntry &TE, bool TopToBottom,
InstructionCost PermuteCost =
TopToBottom
? 0
- : ::getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, Ty, Mask);
+ : getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, Ty, Mask);
InstructionCost InsertFirstCost = TTI->getVectorInstrCost(
Instruction::InsertElement, Ty, TTI::TCK_RecipThroughput, 0,
PoisonValue::get(Ty), *It);
@@ -11641,7 +11607,7 @@ static bool tryToFindDuplicates(SmallVectorImpl<Value *> &VL,
return std::make_pair(true, false);
}
constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
- InstructionCost ReusesCost = ::getShuffleCost(
+ InstructionCost ReusesCost = getShuffleCost(
TTI, TTI::SK_PermuteSingleSrc, VecTy,
NumUniqueScalarValues > VL.size() / 2 ? ArrayRef<int>()
: ArrayRef(ReuseShuffleIndices),
@@ -11822,12 +11788,12 @@ bool BoUpSLP::canBuildSplitNode(ArrayRef<Value *> VL,
if (NumParts >= VL.size())
return false;
constexpr TTI::TargetCostKind Kind = TTI::TCK_RecipThroughput;
- InstructionCost InsertCost = ::getShuffleCost(
+ InstructionCost InsertCost = getShuffleCost(
*TTI, TTI::SK_InsertSubvector, VecTy, {}, Kind, Op1.size(), Op2VecTy);
auto *SubVecTy = cast<VectorType>(
getWidenedType(ScalarTy, std::max(Op1.size(), Op2.size())));
InstructionCost NewShuffleCost =
- ::getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, SubVecTy, Mask, Kind);
+ getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, SubVecTy, Mask, Kind);
if (!LocalState.isCmpOp() && NumParts <= 1 &&
(Mask.empty() || InsertCost >= NewShuffleCost))
return false;
@@ -11848,8 +11814,8 @@ bool BoUpSLP::canBuildSplitNode(ArrayRef<Value *> VL,
OriginalMask[Idx] = Idx + (Op1Indices.test(Idx) ? 0 : VL.size());
}
InstructionCost OriginalCost =
- OriginalVecOpsCost + ::getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc,
- VecTy, OriginalMask, Kind);
+ OriginalVecOpsCost +
+ getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy, OriginalMask, Kind);
InstructionCost NewVecOpsCost =
TTI->getArithmeticInstrCost(Opcode0, Op1VecTy, Kind) +
TTI->getArithmeticInstrCost(Opcode1, Op2VecTy, Kind);
@@ -12934,7 +12900,7 @@ BoUpSLP::getScalarsVectorizationLegality(ArrayRef<Value *> VL, unsigned Depth,
Type *ScalarTy = VL.front()->getType();
auto *VecTy = cast<VectorType>(getWidenedType(ScalarTy, VL.size()));
InstructionCost VectorizeCostEstimate =
- ::getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy, {}, Kind) +
+ getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy, {}, Kind) +
::getScalarizationOverhead(*TTI, ScalarTy, VecTy, Extracted,
/*Insert=*/false, /*Extract=*/true, Kind);
InstructionCost ScalarizeCostEstimate = ::getScalarizationOverhead(
@@ -14449,88 +14415,6 @@ class BaseShuffleAnalysis {
};
} // namespace
-/// Calculate the scalar and the vector costs from vectorizing set of GEPs.
-static std::pair<InstructionCost, InstructionCost>
-getGEPCosts(const TargetTransformInfo &TTI, ArrayRef<Value *> Ptrs,
- Value *BasePtr, unsigned Opcode, TTI::TargetCostKind CostKind,
- Type *ScalarTy, VectorType *VecTy) {
- InstructionCost ScalarCost = 0;
- InstructionCost VecCost = 0;
- // Here we differentiate two cases: (1) when Ptrs represent a regular
- // vectorization tree node (as they are pointer arguments of scattered
- // loads) or (2) when Ptrs are the arguments of loads or stores being
- // vectorized as plane wide unit-stride load/store since all the
- // loads/stores are known to be from/to adjacent locations.
- if (Opcode == Instruction::Load || Opcode == Instruction::Store) {
- // Case 2: estimate costs for pointer related costs when vectorizing to
- // a wide load/store.
- // Scalar cost is estimated as a set of pointers with known relationship
- // between them.
- // For vector code we will use BasePtr as argument for the wide load/store
- // but we also need to account all the instructions which are going to
- // stay in vectorized code due to uses outside of these scalar
- // loads/stores.
- ScalarCost = TTI.getPointersChainCost(
- Ptrs, BasePtr, TTI::PointersChainInfo::getUnitStride(), ScalarTy,
- CostKind);
-
- SmallVector<const Value *> PtrsRetainedInVecCode;
- for (Value *V : Ptrs) {
- if (V == BasePtr) {
- PtrsRetainedInVecCode.push_back(V);
- continue;
- }
- auto *Ptr = dyn_cast<GetElementPtrInst>(V);
- // For simplicity assume Ptr to stay in vectorized code if it's not a
- // GEP instruction. We don't care since it's cost considered free.
- // TODO: We should check for any uses outside of vectorizable tree
- // rather than just single use.
- if (!Ptr || !Ptr->hasOneUse())
- PtrsRetainedInVecCode.push_back(V);
- }
-
- if (PtrsRetainedInVecCode.size() == Ptrs.size()) {
- // If all pointers stay in vectorized code then we don't have
- // any savings on that.
- return std::make_pair(TTI::TCC_Free, TTI::TCC_Free);
- }
- VecCost = TTI.getPointersChainCost(PtrsRetainedInVecCode, BasePtr,
- TTI::PointersChainInfo::getKnownStride(),
- VecTy, CostKind);
- } else {
- // Case 1: Ptrs are the arguments of loads that we are going to transform
- // into masked gather load intrinsic.
- // All the scalar GEPs will be removed as a result of vectorization.
- // For any external uses of some lanes extract element instructions will
- // be generated (which cost is estimated separately).
- TTI::PointersChainInfo PtrsInfo =
- all_of(Ptrs,
- [](const Value *V) {
- auto *Ptr = dyn_cast<GetElementPtrInst>(V);
- return Ptr && !Ptr->hasAllConstantIndices();
- })
- ? TTI::PointersChainInfo::getUnknownStride()
- : TTI::PointersChainInfo::getKnownStride();
-
- ScalarCost =
- TTI.getPointersChainCost(Ptrs, BasePtr, PtrsInfo, ScalarTy, CostKind);
- auto *BaseGEP = dyn_cast<GEPOperator>(BasePtr);
- if (!BaseGEP) {
- auto *It = find_if(Ptrs, IsaPred<GEPOperator>);
- if (It != Ptrs.end())
- BaseGEP = cast<GEPOperator>(*It);
- }
- if (BaseGEP) {
- SmallVector<const Value *> Indices(BaseGEP->indices());
- VecCost = TTI.getGEPCost(BaseGEP->getSourceElementType(),
- BaseGEP->getPointerOperand(), Indices, VecTy,
- CostKind);
- }
- }
-
- return std::make_pair(ScalarCost, VecCost);
-}
-
void BoUpSLP::reorderGatherNode(TreeEntry &TE) {
assert(TE.isGather() && TE.ReorderIndices.empty() &&
"Expected gather node without reordering.");
@@ -14647,9 +14531,8 @@ void BoUpSLP::reorderGatherNode(TreeEntry &TE) {
auto *ScalarTy = TE.Scalars.front()->getType();
auto *VecTy = cast<VectorType>(getWidenedType(ScalarTy, TE.Scalars.size()));
for (auto [Idx, Sz] : SubVectors) {
- Cost +=
- ::getShuffleCost(*TTI, TTI::SK_InsertSubvector, VecTy, {}, CostKind,
- Idx, cast<VectorType>(getWidenedType(ScalarTy, Sz)));
+ Cost += getShuffleCost(*TTI, TTI::SK_InsertSubvector, VecTy, {}, CostKind,
+ Idx, cast<VectorType>(getWidenedType(ScalarTy, Sz)));
}
Cost += getScalarizationOverhead(*TTI, ScalarTy, VecTy, DemandedElts,
/*Insert=*/true,
@@ -14665,11 +14548,11 @@ void BoUpSLP::reorderGatherNode(TreeEntry &TE) {
ReorderMask[I] = I + TE.ReorderIndices.size();
}
}
- Cost += ::getShuffleCost(*TTI,
- any_of(ReorderMask, [&](int I) { return I >= Sz; })
- ? TTI::SK_PermuteTwoSrc
- : TTI::SK_PermuteSingleSrc,
- VecTy, ReorderMask);
+ Cost += getShuffleCost(*TTI,
+ any_of(ReorderMask, [&](int I) { return I >= Sz; })
+ ? TTI::SK_PermuteTwoSrc
+ : TTI::SK_PermuteSingleSrc,
+ VecTy, ReorderMask);
DemandedElts = APInt::getAllOnes(TE.Scalars.size());
ReorderMask.assign(Sz, PoisonMaskElem);
for (unsigned I : seq<unsigned>(Sz)) {
@@ -14686,7 +14569,7 @@ void BoUpSLP::reorderGatherNode(TreeEntry &TE) {
getScalarizationOverhead(*TTI, ScalarTy, VecTy, DemandedElts,
/*Insert=*/true, /*Extract=*/false, CostKind);
if (!DemandedElts.isAllOnes())
- BVCost += ::getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy, ReorderMask);
+ BVCost += getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy, ReorderMask);
if (Cost >= BVCost) {
SmallVector<int> Mask(TE.ReorderIndices.begin(), TE.ReorderIndices.end());
reorderScalars(TE.Scalars, Mask);
@@ -14882,8 +14765,8 @@ bool BoUpSLP::matchesShlZExt(const TreeEntry &TE, OrdersType &Order,
fixupOrderingIndices(Order);
SmallVector<int> Mask;
inversePermutation(Order, Mask);
- BitcastCost += ::getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, SrcVecTy,
- Mask, CostKind);
+ BitcastCost += getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, SrcVecTy,
+ Mask, CostKind);
}
// Check if the combination can be modeled as a bitcast+byteswap operation.
constexpr unsigned ByteSize = 8;
@@ -15394,7 +15277,7 @@ void BoUpSLP::transformNodes() {
TTI->getMemoryOpCost(Instruction::Load, VecTy, BaseLI->getAlign(),
BaseLI->getPointerAddressSpace(), CostKind,
TTI::OperandValueInfo()) +
- ::getShuffleCost(*TTI, TTI::SK_Reverse, VecTy, Mask, CostKind);
+ getShuffleCost(*TTI, TTI::SK_Reverse, VecTy, Mask, CostKind);
InstructionCost StridedCost = TTI->getMemIntrinsicInstrCost(
MemIntrinsicCostAttributes(Intrinsic::experimental_vp_strided_load,
VecTy, BaseLI->getPointerOperand(),
@@ -15435,7 +15318,7 @@ void BoUpSLP::transformNodes() {
TTI->getMemoryOpCost(Instruction::Store, VecTy, BaseSI->getAlign(),
BaseSI->getPointerAddressSpace(), CostKind,
TTI::OperandValueInfo()) +
- ::getShuffleCost(*TTI, TTI::SK_Reverse, VecTy, Mask, CostKind);
+ getShuffleCost(*TTI, TTI::SK_Reverse, VecTy, Mask, CostKind);
InstructionCost StridedCost = TTI->getMemIntrinsicInstrCost(
MemIntrinsicCostAttributes(Intrinsic::experimental_vp_strided_store,
VecTy, BaseSI->getPointerOperand(),
@@ -15781,11 +15664,10 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
InstructionCost InsertCost =
TTI.getVectorInstrCost(Instruction::InsertElement, VecTy, CostKind, 0,
PoisonValue::get(VecTy), *It);
- return InsertCost + ::getShuffleCost(TTI,
- TargetTransformInfo::SK_Broadcast,
- VecTy, ShuffleMask, CostKind,
- /*Index=*/0, /*SubTp=*/nullptr,
- /*Args=*/*It);
+ return InsertCost + getShuffleCost(TTI, TargetTransformInfo::SK_Broadcast,
+ VecTy, ShuffleMask, CostKind,
+ /*Index=*/0, /*SubTp=*/nullptr,
+ /*Args=*/*It);
}
return GatherCost +
(all_of(Gathers, IsaPred<UndefValue>)
@@ -15889,14 +15771,14 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
if (*ShuffleKinds[Part] != TTI::SK_PermuteSingleSrc ||
!ShuffleVectorInst::isIdentityMask(
MaskSlice, std::max<unsigned>(NumElts, MaskSlice.size())))
- Cost += ::getShuffleCost(
+ Cost += getShuffleCost(
TTI, *ShuffleKinds[Part],
cast<VectorType>(getWidenedType(ScalarTy, NumElts)), MaskSlice);
continue;
}
if (*RegShuffleKind != TTI::SK_PermuteSingleSrc ||
!ShuffleVectorInst::isIdentityMask(SubMask, EltsPerVector)) {
- Cost += ::getShuffleCost(
+ Cost += getShuffleCost(
TTI, *RegShuffleKind,
cast<VectorType>(getWidenedType(ScalarTy, EltsPerVector)), SubMask);
}
@@ -15905,7 +15787,7 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
for (const auto [Idx, SubVecSize] : zip(Indices, SubVecSizes)) {
assert((Idx + SubVecSize) <= BaseVF &&
"SK_ExtractSubvector index out of range");
- Cost += ::getShuffleCost(
+ Cost += getShuffleCost(
TTI, TTI::SK_ExtractSubvector,
cast<VectorType>(getWidenedType(ScalarTy, BaseVF)), {}, CostKind,
Idx, cast<VectorType>(getWidenedType(ScalarTy, SubVecSize)));
@@ -15914,7 +15796,7 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
// subvector extract.
SubMask.assign(NumElts, PoisonMaskElem);
copy(MaskSlice, SubMask.begin());
- InstructionCost OriginalCost = ::getShuffleCost(
+ InstructionCost OriginalCost = getShuffleCost(
TTI, *ShuffleKinds[Part],
cast<VectorType>(getWidenedType(ScalarTy, NumElts)), SubMask);
if (OriginalCost < Cost)
@@ -16011,8 +15893,8 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
cast<VectorType>(V1->getType())->getElementCount().getKnownMinValue();
if (isEmptyOrIdentity(Mask, VF))
return TTI::TCC_Free;
- return ::getShuffleCost(TTI, TTI::SK_PermuteTwoSrc,
- cast<VectorType>(V1->getType()), Mask);
+ return getShuffleCost(TTI, TTI::SK_PermuteTwoSrc,
+ cast<VectorType>(V1->getType()), Mask);
}
InstructionCost createShuffleVector(Value *V1, ArrayRef<int> Mask,
ArrayRef<Value *> VL) const {
@@ -16021,7 +15903,7 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
cast<VectorType>(V1->getType())->getElementCount().getKnownMinValue();
if (is...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/210278
More information about the llvm-commits
mailing list