[llvm] [SLP] Cost using TCK_CodeSize under -Os and -Oz (PR #217398)
Ryan Buchner via llvm-commits
llvm-commits at lists.llvm.org
Fri Aug 21 09:58:41 PDT 2026
https://github.com/bababuck updated https://github.com/llvm/llvm-project/pull/217398
>From dcef9db0c71f03624471967fb0ab18289226fb75 Mon Sep 17 00:00:00 2001
From: bababuck <buchner.ryan at gmail.com>
Date: Wed, 19 Aug 2026 07:43:34 -0700
Subject: [PATCH 1/2] [SLP] Test for code size costing
---
.../Transforms/SLPVectorizer/X86/cost-size.ll | 70 +++++++++++++++++++
1 file changed, 70 insertions(+)
create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/cost-size.ll
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/cost-size.ll b/llvm/test/Transforms/SLPVectorizer/X86/cost-size.ll
new file mode 100644
index 0000000000000..6e15f2f001065
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/cost-size.ll
@@ -0,0 +1,70 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -slp-threshold=0 -passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s --check-prefixes=CHECK
+
+; Profitable to vectorize based on size, but not based on reciprocal throughput
+
+define i16 @test_optsize(ptr %p, ptr %inc) #0 {
+; CHECK-LABEL: define i16 @test_optsize(
+; CHECK-SAME: ptr [[P:%.*]], ptr [[INC:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[E0:%.*]] = load i16, ptr [[P]], align 4
+; CHECK-NEXT: [[E1:%.*]] = load i16, ptr [[INC]], align 2
+; CHECK-NEXT: [[TMP3:%.*]] = udiv i16 [[E0]], 13
+; CHECK-NEXT: [[TMP4:%.*]] = udiv i16 [[E1]], 14
+; CHECK-NEXT: [[A:%.*]] = add i16 [[TMP3]], [[TMP4]]
+; CHECK-NEXT: ret i16 [[A]]
+;
+entry:
+ %e0 = load i16, ptr %p, align 4
+ %e1 = load i16, ptr %inc, align 2
+
+ %d0 = udiv i16 %e0, 13
+ %d1 = udiv i16 %e1, 14
+ %a = add i16 %d0, %d1
+ ret i16 %a
+}
+
+define i16 @testc_optsize(ptr %p, ptr %inc) #1 {
+; CHECK-LABEL: define i16 @testc_optsize(
+; CHECK-SAME: ptr [[P:%.*]], ptr [[INC:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[E0:%.*]] = load i16, ptr [[P]], align 4
+; CHECK-NEXT: [[E1:%.*]] = load i16, ptr [[INC]], align 2
+; CHECK-NEXT: [[TMP3:%.*]] = udiv i16 [[E0]], 13
+; CHECK-NEXT: [[TMP4:%.*]] = udiv i16 [[E1]], 14
+; CHECK-NEXT: [[A:%.*]] = add i16 [[TMP3]], [[TMP4]]
+; CHECK-NEXT: ret i16 [[A]]
+;
+entry:
+ %e0 = load i16, ptr %p, align 4
+ %e1 = load i16, ptr %inc, align 2
+
+ %d0 = udiv i16 %e0, 13
+ %d1 = udiv i16 %e1, 14
+ %a = add i16 %d0, %d1
+ ret i16 %a
+}
+
+define i16 @testsf_optsize(ptr %p, ptr %inc) {
+; CHECK-LABEL: define i16 @testsf_optsize(
+; CHECK-SAME: ptr [[P:%.*]], ptr [[INC:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[E0:%.*]] = load i16, ptr [[P]], align 4
+; CHECK-NEXT: [[E1:%.*]] = load i16, ptr [[INC]], align 2
+; CHECK-NEXT: [[D0:%.*]] = udiv i16 [[E0]], 13
+; CHECK-NEXT: [[D1:%.*]] = udiv i16 [[E1]], 14
+; CHECK-NEXT: [[A:%.*]] = add i16 [[D0]], [[D1]]
+; CHECK-NEXT: ret i16 [[A]]
+;
+entry:
+ %e0 = load i16, ptr %p, align 4
+ %e1 = load i16, ptr %inc, align 2
+
+ %d0 = udiv i16 %e0, 13
+ %d1 = udiv i16 %e1, 14
+ %a = add i16 %d0, %d1
+ ret i16 %a
+}
+
+attributes #0 = { optsize }
+attributes #1 = { minsize optsize }
>From 8a7a32ceb02e002bba6ecc4650f153f73384f65a Mon Sep 17 00:00:00 2001
From: bababuck <buchner.ryan at gmail.com>
Date: Tue, 18 Aug 2026 16:25:29 -0700
Subject: [PATCH 2/2] [SLP] Cost using CodeSize for -Os and -Oz
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 248 +++++++++---------
.../Transforms/SLPVectorizer/X86/cost-size.ll | 14 +-
2 files changed, 134 insertions(+), 128 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 166554b922572..7ec2ad730e7de 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -254,6 +254,10 @@ static cl::opt<bool> ForcePostProcessStoresOperands(
"slp-postprocess-stores-operands", cl::init(false), cl::Hidden,
cl::desc("Force vectorization of non-vectorizable stores operands."));
+static TTI::TargetCostKind getSLPCostKind(const Function *F) {
+ return F && F->hasOptSize() ? TTI::TCK_CodeSize : TTI::TCK_RecipThroughput;
+}
+
static cl::opt<bool> NonVectReductions(
"slp-non-vectorizables-as-reductions", cl::init(false), cl::Hidden,
cl::desc(
@@ -683,7 +687,7 @@ class slpvectorizer::BoUpSLP {
DominatorTree *Dt, AssumptionCache *AC, DemandedBits *DB,
const DataLayout *DL, OptimizationRemarkEmitter *ORE)
: BatchAA(*Aa), F(Func), SE(Se), TTI(Tti), TLI(TLi), LI(Li), DT(Dt),
- AC(AC), DB(DB), DL(DL), ORE(ORE),
+ AC(AC), DB(DB), DL(DL), ORE(ORE), CostKind(getSLPCostKind(Func)),
Builder(Se->getContext(), TargetFolder(*DL)) {
CodeMetrics::collectEphemeralValues(F, AC, EphValues);
// Use the vector register size specified by the target unless overridden
@@ -722,6 +726,8 @@ class slpvectorizer::BoUpSLP {
/// holding live values over call sites.
InstructionCost getSpillCost();
+ TargetTransformInfo::TargetCostKind getCostKind() const { return CostKind; }
+
/// Calculates the cost of the subtrees, trims non-profitable ones and returns
/// final cost.
InstructionCost
@@ -5543,6 +5549,8 @@ class slpvectorizer::BoUpSLP {
DemandedBits *DB;
const DataLayout *DL;
OptimizationRemarkEmitter *ORE;
+ /// Cached cost-model mode for this function (-Os/-Oz => CodeSize).
+ const TargetTransformInfo::TargetCostKind CostKind;
unsigned MaxVecRegSize; // This is set by TTI or overridden by cl::opt.
unsigned MinVecRegSize; // Set by cl::opt (default: 128).
@@ -6185,10 +6193,11 @@ static InstructionCost getVectorInstrCost(
/// This is similar to TargetTransformInfo::getExtractWithExtendCost, but if Dst
/// is a FixedVectorType, a vector will be extracted instead of a scalar.
-static InstructionCost getExtractWithExtendCost(
- const TargetTransformInfo &TTI, unsigned Opcode, Type *Dst,
- VectorType *VecTy, unsigned Index,
- TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput) {
+static InstructionCost getExtractWithExtendCost(const TargetTransformInfo &TTI,
+ unsigned Opcode, Type *Dst,
+ VectorType *VecTy,
+ unsigned Index,
+ TTI::TargetCostKind CostKind) {
if (isVectorizedTy(Dst)) {
assert(SLPReVec && "Only supported by REVEC.");
auto *SubTp = cast<FixedVectorType>(
@@ -6276,19 +6285,20 @@ static bool buildCompressMask(ArrayRef<Value *> PointerOps,
/// Checks if the \p VL can be transformed to a (masked)load + compress or
/// (masked) interleaved load.
-static bool isMaskedLoadCompress(
- ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
- ArrayRef<unsigned> Order, const TargetTransformInfo &TTI,
- const DataLayout &DL, ScalarEvolution &SE, AssumptionCache &AC,
- const DominatorTree &DT, const TargetLibraryInfo &TLI,
- const function_ref<bool(Value *)> AreAllUsersVectorized, bool &IsMasked,
- unsigned &InterleaveFactor, SmallVectorImpl<int> &CompressMask,
- VectorType *&LoadVecTy) {
+static bool
+isMaskedLoadCompress(ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
+ ArrayRef<unsigned> Order, const TargetTransformInfo &TTI,
+ const DataLayout &DL, ScalarEvolution &SE,
+ AssumptionCache &AC, const DominatorTree &DT,
+ const TargetLibraryInfo &TLI, TTI::TargetCostKind CostKind,
+ const function_ref<bool(Value *)> AreAllUsersVectorized,
+ bool &IsMasked, unsigned &InterleaveFactor,
+ SmallVectorImpl<int> &CompressMask,
+ VectorType *&LoadVecTy) {
InterleaveFactor = 0;
Type *ScalarTy = VL.front()->getType();
const size_t Sz = VL.size();
auto *VecTy = cast<VectorType>(getWidenedType(ScalarTy, Sz));
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
SmallVector<int> Mask;
if (!Order.empty())
inversePermutation(Order, Mask);
@@ -6421,15 +6431,15 @@ isMaskedLoadCompress(ArrayRef<Value *> VL, ArrayRef<Value *> PointerOps,
ArrayRef<unsigned> Order, const TargetTransformInfo &TTI,
const DataLayout &DL, ScalarEvolution &SE,
AssumptionCache &AC, const DominatorTree &DT,
- const TargetLibraryInfo &TLI,
+ const TargetLibraryInfo &TLI, TTI::TargetCostKind CostKind,
const function_ref<bool(Value *)> AreAllUsersVectorized) {
bool IsMasked;
unsigned InterleaveFactor;
SmallVector<int> CompressMask;
VectorType *LoadVecTy;
return isMaskedLoadCompress(VL, PointerOps, Order, TTI, DL, SE, AC, DT, TLI,
- AreAllUsersVectorized, IsMasked, InterleaveFactor,
- CompressMask, LoadVecTy);
+ CostKind, AreAllUsersVectorized, IsMasked,
+ InterleaveFactor, CompressMask, LoadVecTy);
}
/// Checks if the stores \p VL with pointers \p PointerOps can be lowered as a
@@ -6938,7 +6948,7 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
if (static_cast<uint64_t>(Diff) == Sz - 1)
return LoadsState::Vectorize;
if (isMaskedLoadCompress(VL, PointerOps, Order, *TTI, *DL, *SE, *AC, *DT,
- *TLI, [&](Value *V) {
+ *TLI, CostKind, [&](Value *V) {
return areAllUsersVectorized(
cast<Instruction>(V), UserIgnoreList);
}))
@@ -6961,7 +6971,6 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
if (BestVF)
*BestVF = 0;
// Compare masked gather cost and loads + insert subvector costs.
- TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
auto [ScalarGEPCost, VectorGEPCost] =
getGEPCosts(TTI, PointerOps, PointerOps.front(), Instruction::Load,
CostKind, ScalarTy, VecTy);
@@ -7687,12 +7696,12 @@ BoUpSLP::getReorderingData(const TreeEntry &TE, bool TopToBottom,
TopToBottom
? 0
: getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, Ty, Mask);
- InstructionCost InsertFirstCost = TTI->getVectorInstrCost(
- Instruction::InsertElement, Ty, TTI::TCK_RecipThroughput, 0,
- PoisonValue::get(Ty), *It);
- InstructionCost InsertIdxCost = TTI->getVectorInstrCost(
- Instruction::InsertElement, Ty, TTI::TCK_RecipThroughput, Idx,
- PoisonValue::get(Ty), *It);
+ InstructionCost InsertFirstCost =
+ TTI->getVectorInstrCost(Instruction::InsertElement, Ty, CostKind, 0,
+ PoisonValue::get(Ty), *It);
+ InstructionCost InsertIdxCost =
+ TTI->getVectorInstrCost(Instruction::InsertElement, Ty, CostKind,
+ Idx, PoisonValue::get(Ty), *It);
if (InsertFirstCost + PermuteCost < InsertIdxCost) {
OrdersType Order(Sz, Sz);
Order[Idx] = 0;
@@ -9862,7 +9871,8 @@ buildIntrinsicArgTypes(const CallInst *CI, const Intrinsic::ID ID,
/// calls, if they cannot be vectorized/will be scalarized.
static std::pair<InstructionCost, InstructionCost>
getVectorCallCosts(CallInst *CI, Type *VecTy, const TargetTransformInfo *TTI,
- const TargetLibraryInfo *TLI, ArrayRef<Type *> ArgTys) {
+ const TargetLibraryInfo *TLI, ArrayRef<Type *> ArgTys,
+ TTI::TargetCostKind CostKind) {
auto Shape = VFShape::get(CI->getFunctionType(),
ElementCount::getFixed(getNumElements(VecTy)),
false /*HasGlobalPred*/);
@@ -9871,8 +9881,7 @@ getVectorCallCosts(CallInst *CI, Type *VecTy, const TargetTransformInfo *TTI,
if (!CI->isNoBuiltin() && VecFunc) {
// Calculate the cost of the vector library call.
// If the corresponding vector call is cheaper, return its cost.
- LibCost =
- TTI->getCallInstrCost(nullptr, VecTy, ArgTys, TTI::TCK_RecipThroughput);
+ LibCost = TTI->getCallInstrCost(nullptr, VecTy, ArgTys, CostKind);
}
Intrinsic::ID ID = getVectorIntrinsicIDForCall(CI, TLI);
@@ -9883,8 +9892,7 @@ getVectorCallCosts(CallInst *CI, Type *VecTy, const TargetTransformInfo *TTI,
const InstructionCost ScalarLimit = 10000;
IntrinsicCostAttributes CostAttrs(ID, VecTy, ArgTys, FMF, nullptr,
LibCost.isValid() ? LibCost : ScalarLimit);
- auto IntrinsicCost =
- TTI->getIntrinsicInstrCost(CostAttrs, TTI::TCK_RecipThroughput);
+ auto IntrinsicCost = TTI->getIntrinsicInstrCost(CostAttrs, CostKind);
if (LibCost.isValid()) {
if (IntrinsicCost > LibCost)
IntrinsicCost = InstructionCost::getInvalid();
@@ -9896,8 +9904,7 @@ getVectorCallCosts(CallInst *CI, Type *VecTy, const TargetTransformInfo *TTI,
SmallVector<const Value *> Args(CI->args());
IntrinsicCostAttributes ArgAwareAttrs(
ID, VecTy, Args, ArgTys, FMF, dyn_cast<IntrinsicInst>(CI), ScalarLimit);
- IntrinsicCost =
- TTI->getIntrinsicInstrCost(ArgAwareAttrs, TTI::TCK_RecipThroughput);
+ IntrinsicCost = TTI->getIntrinsicInstrCost(ArgAwareAttrs, CostKind);
if (IntrinsicCost > ScalarLimit)
IntrinsicCost = InstructionCost::getInvalid();
}
@@ -9909,16 +9916,16 @@ getVectorCallCosts(CallInst *CI, Type *VecTy, const TargetTransformInfo *TTI,
/// arithmetic op or a vectorizable call).
static InstructionCost getVectorOpCost(Instruction *I, unsigned VF,
const TargetTransformInfo &TTI,
- const TargetLibraryInfo &TLI) {
+ const TargetLibraryInfo &TLI,
+ const TTI::TargetCostKind CostKind) {
assert((isa<BinaryOperator, CallInst>(I)) &&
"getVectorOpCost expects an arithmetic op or a vectorizable call.");
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
Type *VecTy = getWidenedType(I->getType(), VF);
if (auto *CI = dyn_cast<CallInst>(I)) {
Intrinsic::ID ID = getVectorIntrinsicIDForCall(CI, &TLI);
SmallVector<Type *> ArgTys = buildIntrinsicArgTypes(CI, ID, VF, 0, &TTI);
auto [IntrCost, LibCost] =
- getVectorCallCosts(CI, VecTy, &TTI, &TLI, ArgTys);
+ getVectorCallCosts(CI, VecTy, &TTI, &TLI, ArgTys, CostKind);
return std::min(IntrCost, LibCost);
}
return TTI.getArithmeticInstrCost(I->getOpcode(), VecTy, CostKind);
@@ -9974,7 +9981,8 @@ static SeedGroupKey getSeedGroupKey(const Instruction *I,
/// per lane (e.g. fdiv, frem, fsqrt).
static bool isPoorThroughputOp(Instruction *I, const TargetTransformInfo &TTI,
const TargetLibraryInfo &TLI,
- PoorThroughputOpCache &Cache) {
+ PoorThroughputOpCache &Cache,
+ const TTI::TargetCostKind CostKind) {
if (!isa<BinaryOperator, CallInst>(I))
return false;
Type *Ty = I->getType();
@@ -9982,12 +9990,11 @@ static bool isPoorThroughputOp(Instruction *I, const TargetTransformInfo &TTI,
!isValidElementType(Ty))
return false;
auto Analyze = [&]() {
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
InstructionCost ScalarCost = TTI.getInstructionCost(I, CostKind);
if (ScalarCost < TTI::TCC_Expensive)
return false;
constexpr unsigned MinVF = 2;
- return getVectorOpCost(I, MinVF, TTI, TLI) < ScalarCost * MinVF;
+ return getVectorOpCost(I, MinVF, TTI, TLI, CostKind) < ScalarCost * MinVF;
};
auto CheckCached = [&](bool IsCheap, llvm::function_ref<void()> MarkCheap) {
if (IsCheap)
@@ -10585,7 +10592,8 @@ BoUpSLP::TreeEntry::EntryState BoUpSLP::getScalarsVectorizationState(
SmallVector<Type *> ArgTys =
buildIntrinsicArgTypes(CI, ID, VL.size(), 0, TTI);
auto *VecTy = getWidenedType(S.getMainOp()->getType(), VL.size());
- auto VecCallCosts = getVectorCallCosts(CI, VecTy, TTI, TLI, ArgTys);
+ auto VecCallCosts =
+ getVectorCallCosts(CI, VecTy, TTI, TLI, ArgTys, CostKind);
if (!VecCallCosts.first.isValid() && !VecCallCosts.second.isValid())
return TreeEntry::NeedToGather;
@@ -10893,7 +10901,7 @@ static bool tryToFindDuplicates(SmallVectorImpl<Value *> &VL,
UniquesNumParts <= NumParts)
return std::make_pair(true, false);
}
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
+ const TTI::TargetCostKind CostKind = R.getCostKind();
InstructionCost ReusesCost = getShuffleCost(
TTI, TTI::SK_PermuteSingleSrc, VecTy,
NumUniqueScalarValues > VL.size() / 2 ? ArrayRef<int>()
@@ -11074,13 +11082,12 @@ bool BoUpSLP::canBuildSplitNode(ArrayRef<Value *> VL,
// as alternate ops.
if (NumParts >= VL.size())
return false;
- constexpr TTI::TargetCostKind Kind = TTI::TCK_RecipThroughput;
InstructionCost InsertCost = getShuffleCost(
- *TTI, TTI::SK_InsertSubvector, VecTy, {}, Kind, Op1.size(), Op2VecTy);
+ *TTI, TTI::SK_InsertSubvector, VecTy, {}, CostKind, Op1.size(), Op2VecTy);
auto *SubVecTy = cast<VectorType>(
getWidenedType(ScalarTy, std::max(Op1.size(), Op2.size())));
InstructionCost NewShuffleCost =
- getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, SubVecTy, Mask, Kind);
+ getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, SubVecTy, Mask, CostKind);
if (!LocalState.isCmpOp() && NumParts <= 1 &&
(Mask.empty() || InsertCost >= NewShuffleCost))
return false;
@@ -11092,8 +11099,8 @@ bool BoUpSLP::canBuildSplitNode(ArrayRef<Value *> VL,
(LocalState.getMainOp()->isUnaryOp() &&
LocalState.getAltOp()->isUnaryOp())) {
InstructionCost OriginalVecOpsCost =
- TTI->getArithmeticInstrCost(Opcode0, VecTy, Kind) +
- TTI->getArithmeticInstrCost(Opcode1, VecTy, Kind);
+ TTI->getArithmeticInstrCost(Opcode0, VecTy, CostKind) +
+ TTI->getArithmeticInstrCost(Opcode1, VecTy, CostKind);
SmallVector<int> OriginalMask(VL.size(), PoisonMaskElem);
for (unsigned Idx : seq<unsigned>(VL.size())) {
if (isa<PoisonValue>(VL[Idx]))
@@ -11101,11 +11108,11 @@ bool BoUpSLP::canBuildSplitNode(ArrayRef<Value *> VL,
OriginalMask[Idx] = Idx + (Op1Indices.test(Idx) ? 0 : VL.size());
}
InstructionCost OriginalCost =
- OriginalVecOpsCost +
- getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy, OriginalMask, Kind);
+ OriginalVecOpsCost + getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy,
+ OriginalMask, CostKind);
InstructionCost NewVecOpsCost =
- TTI->getArithmeticInstrCost(Opcode0, Op1VecTy, Kind) +
- TTI->getArithmeticInstrCost(Opcode1, Op2VecTy, Kind);
+ TTI->getArithmeticInstrCost(Opcode0, Op1VecTy, CostKind) +
+ TTI->getArithmeticInstrCost(Opcode1, Op2VecTy, CostKind);
InstructionCost NewCost =
NewVecOpsCost + InsertCost +
(!VectorizableTree.empty() && VectorizableTree.front()->hasState() &&
@@ -11813,7 +11820,7 @@ class InstructionsCompatibilityAnalysis {
}
if (!Res)
return OrigS;
- constexpr TTI::TargetCostKind Kind = TTI::TCK_RecipThroughput;
+ const TTI::TargetCostKind Kind = R.getCostKind();
InstructionCost ScalarCost = TTI.getInstructionCost(S.getMainOp(), Kind);
InstructionCost VectorCost;
auto *VecTy = getWidenedType(S.getMainOp()->getType(), VL.size());
@@ -12303,7 +12310,6 @@ BoUpSLP::getScalarsVectorizationLegality(ArrayRef<Value *> VL, unsigned Depth,
return std::make_pair(Vectorized, Extracted);
};
auto [Vectorized, Extracted] = GetNumVectorizedExtracted();
- constexpr TTI::TargetCostKind Kind = TTI::TCK_RecipThroughput;
bool PreferScalarize = !Vectorized.isAllOnes() && VL.size() == 2;
if (!Vectorized.isAllOnes() && !PreferScalarize) {
// Rough cost estimation, if the vector code (+ potential extracts) is
@@ -12311,12 +12317,13 @@ BoUpSLP::getScalarsVectorizationLegality(ArrayRef<Value *> VL, unsigned Depth,
Type *ScalarTy = VL.front()->getType();
auto *VecTy = cast<VectorType>(getWidenedType(ScalarTy, VL.size()));
InstructionCost VectorizeCostEstimate =
- getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy, {}, Kind) +
+ getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy, {}, CostKind) +
::getScalarizationOverhead(*TTI, ScalarTy, VecTy, Extracted,
- /*Insert=*/false, /*Extract=*/true, Kind);
+ /*Insert=*/false, /*Extract=*/true,
+ CostKind);
InstructionCost ScalarizeCostEstimate = ::getScalarizationOverhead(
*TTI, ScalarTy, VecTy, Vectorized,
- /*Insert=*/true, /*Extract=*/false, Kind, /*ForPoisonSrc=*/false);
+ /*Insert=*/true, /*Extract=*/false, CostKind, /*ForPoisonSrc=*/false);
PreferScalarize = VectorizeCostEstimate > ScalarizeCostEstimate;
}
if (PreferScalarize) {
@@ -13596,7 +13603,8 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
const InstructionsState &S,
DominatorTree &DT, const DataLayout &DL,
TargetTransformInfo &TTI,
- const TargetLibraryInfo &TLI);
+ const TargetLibraryInfo &TLI,
+ TTI::TargetCostKind CostKind);
uint64_t BoUpSLP::getNumScalarInsts(bool HasTreeLoop) {
uint64_t Total = 0;
@@ -13676,7 +13684,8 @@ uint64_t BoUpSLP::getNumScalarInsts(bool HasTreeLoop) {
if (!I || (TE.isAltShuffle() && I->getOpcode() != Instruction::FAdd &&
I->getOpcode() != Instruction::FSub))
continue;
- if (canConvertToFMA(I, InstructionsState(I, I), *DT, *DL, *TTI, *TLI)
+ if (canConvertToFMA(I, InstructionsState(I, I), *DT, *DL, *TTI, *TLI,
+ CostKind)
.isValid()) {
assert(Count > 0 && "Underflow in scalar inst count (fma)");
--Count;
@@ -14335,7 +14344,6 @@ void BoUpSLP::reorderGatherNode(TreeEntry &TE) {
if (!TE.ReuseShuffleIndices.empty() || TE.ReorderIndices.empty())
return;
// Do simple cost estimation.
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
InstructionCost Cost = 0;
auto *ScalarTy = TE.Scalars.front()->getType();
auto *VecTy = cast<VectorType>(getWidenedType(ScalarTy, TE.Scalars.size()));
@@ -14392,7 +14400,8 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
const InstructionsState &S,
DominatorTree &DT, const DataLayout &DL,
TargetTransformInfo &TTI,
- const TargetLibraryInfo &TLI) {
+ const TargetLibraryInfo &TLI,
+ TTI::TargetCostKind CostKind) {
assert(all_of(VL,
[](Value *V) {
return V->getType()->getScalarType()->isFloatingPointTy();
@@ -14435,7 +14444,6 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
// Compare the costs.
InstructionCost FMulPlusFAddCost = 0;
InstructionCost FMACost = 0;
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
FastMathFlags FMF;
FMF.set();
for (Value *V : VL) {
@@ -14550,7 +14558,6 @@ bool BoUpSLP::matchesShlZExt(const TreeEntry &TE, OrdersType &Order,
if (is_contained(Order, VF))
return false;
}
- TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
auto *SrcType = IntegerType::getIntNTy(ScalarTy->getContext(),
Stride * LhsTE->getVectorFactor());
FastMathFlags FMF;
@@ -14693,7 +14700,6 @@ bool BoUpSLP::matchesInversedZExtSelect(
getWidenedType(Cmp->getOperand(0)->getType(), CmpTE->getVectorFactor());
Type *CmpTy = CmpInst::makeCmpResultType(VecTy);
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
InstructionCost VecCost =
TTI->getCmpSelInstrCost(CmpTE->getOpcode(), VecTy, CmpTy, MainPred,
CostKind, getOperandInfo(CmpTE->getOperand(0)),
@@ -14761,7 +14767,6 @@ bool BoUpSLP::matchesSelectOfBits(const TreeEntry &SelectTE) const {
VecTy = cast<VectorType>(
getWidenedType(EffectiveScalarTy, SelectTE.getVectorFactor()));
}
- TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
InstructionCost BitcastCost = TTI->getCastInstrCost(
Instruction::BitCast, DstTy, CmpTy, TTI::CastContextHint::None, CostKind);
if (DstTy != ScalarTy) {
@@ -14779,7 +14784,6 @@ bool BoUpSLP::matchesSelectOfBits(const TreeEntry &SelectTE) const {
}
void BoUpSLP::transformNodes() {
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
BaseGraphSize = VectorizableTree.size();
// Turn graph transforming mode on and off, when done.
class GraphTransformModeRAAI {
@@ -15262,7 +15266,8 @@ void BoUpSLP::transformNodes() {
(E.getOpcode() == Instruction::FSub ||
!IsOneUseVectorFMulOperand(RHS)))
break;
- if (!canConvertToFMA(E.Scalars, E.getOperations(), *DT, *DL, *TTI, *TLI)
+ if (!canConvertToFMA(E.Scalars, E.getOperations(), *DT, *DL, *TTI, *TLI,
+ CostKind)
.isValid())
break;
// This node is a fmuladd node.
@@ -15422,7 +15427,7 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
SmallDenseSet<Value *> VectorizedVals;
BoUpSLP &R;
SmallPtrSetImpl<Value *> &CheckedExtracts;
- constexpr static TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
+ const TTI::TargetCostKind CostKind;
/// While set, still trying to estimate the cost for the same nodes and we
/// can delay actual cost estimation (virtual shuffle instruction emission).
/// May help better estimate the cost if same nodes must be permuted + allows
@@ -15687,6 +15692,7 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
class ShuffleCostBuilder {
const TargetTransformInfo &TTI;
+ const TTI::TargetCostKind CostKind;
static bool isEmptyOrIdentity(ArrayRef<int> Mask, unsigned VF) {
int Index = -1;
@@ -15698,7 +15704,9 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
}
public:
- ShuffleCostBuilder(const TargetTransformInfo &TTI) : TTI(TTI) {}
+ ShuffleCostBuilder(const TargetTransformInfo &TTI,
+ TTI::TargetCostKind CostKind)
+ : TTI(TTI), CostKind(CostKind) {}
~ShuffleCostBuilder() = default;
InstructionCost createShuffleVector(Value *V1, Value *,
ArrayRef<int> Mask) const {
@@ -15717,9 +15725,9 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
cast<VectorType>(V1->getType())->getElementCount().getKnownMinValue();
if (isEmptyOrIdentity(Mask, VF))
return TTI::TCC_Free;
- return getShuffleCost(
- TTI, TTI::SK_PermuteSingleSrc, cast<VectorType>(V1->getType()), Mask,
- TTI::TCK_RecipThroughput, /*Index=*/0, /*SubTp=*/nullptr, VL);
+ return getShuffleCost(TTI, TTI::SK_PermuteSingleSrc,
+ cast<VectorType>(V1->getType()), Mask, CostKind,
+ /*Index=*/0, /*SubTp=*/nullptr, VL);
}
InstructionCost createIdentity(Value *) const { return TTI::TCC_Free; }
InstructionCost createPoison(Type *Ty, unsigned VF) const {
@@ -15735,7 +15743,7 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
createShuffle(const PointerUnion<Value *, const TreeEntry *> &P1,
const PointerUnion<Value *, const TreeEntry *> &P2,
ArrayRef<int> Mask, ArrayRef<Value *> VL = {}) {
- ShuffleCostBuilder Builder(TTI);
+ ShuffleCostBuilder Builder(TTI, CostKind);
SmallVector<int> CommonMask(Mask);
Value *V1 = P1.dyn_cast<Value *>(), *V2 = P2.dyn_cast<Value *>();
unsigned CommonVF = Mask.size();
@@ -15951,7 +15959,7 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
SmallPtrSetImpl<Value *> &CheckedExtracts)
: BaseShuffleAnalysis(ScalarTy), TTI(TTI),
VectorizedVals(VectorizedVals.begin(), VectorizedVals.end()), R(R),
- CheckedExtracts(CheckedExtracts) {}
+ CheckedExtracts(CheckedExtracts), CostKind(R.getCostKind()) {}
Value *adjustExtracts(const TreeEntry *E, MutableArrayRef<int> Mask,
ArrayRef<std::optional<TTI::ShuffleKind>> ShuffleKinds,
unsigned NumParts, bool &UseVecBaseAsInput) {
@@ -16716,7 +16724,6 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
ScalarTy = ScalarTy->getScalarType();
if (!isValidElementType(ScalarTy))
return InstructionCost::getInvalid();
- TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
// If we have computed a smaller type for the expression, update VecTy so
// that the costs will be accurate.
@@ -16979,7 +16986,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
};
auto GetFMulAddCost = [&, &TTI = *TTI](const InstructionsState &S,
Instruction *VI) {
- InstructionCost Cost = canConvertToFMA(VI, S, *DT, *DL, TTI, *TLI);
+ InstructionCost Cost =
+ canConvertToFMA(VI, S, *DT, *DL, TTI, *TLI, CostKind);
return Cost;
};
switch (ShuffleOrOp) {
@@ -17706,8 +17714,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
PointerOps[I] = cast<LoadInst>(V)->getPointerOperand();
[[maybe_unused]] bool IsVectorized = isMaskedLoadCompress(
Scalars, PointerOps, E->ReorderIndices, *TTI, *DL, *SE, *AC, *DT,
- *TLI, [](Value *) { return true; }, IsMasked, InterleaveFactor,
- CompressMask, LoadVecTy);
+ *TLI, CostKind, [](Value *) { return true; }, IsMasked,
+ InterleaveFactor, CompressMask, LoadVecTy);
CompressEntryToData.try_emplace(E, CompressMask, LoadVecTy,
InterleaveFactor, IsMasked);
Align CommonAlignment = LI0->getAlign();
@@ -17868,7 +17876,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
SmallVector<Type *> ArgTys = buildIntrinsicArgTypes(
CI, ID, getNumElements(VecTy),
It != MinBWs.end() ? It->second.first : 0, TTI);
- auto VecCallCosts = getVectorCallCosts(CI, VecTy, TTI, TLI, ArgTys);
+ auto VecCallCosts =
+ getVectorCallCosts(CI, VecTy, TTI, TLI, ArgTys, CostKind);
return std::min(VecCallCosts.first, VecCallCosts.second) + CommonCost;
};
return GetCostDiff(GetScalarCost, GetVectorCost);
@@ -18495,8 +18504,7 @@ bool BoUpSLP::isTreeTinyAndNotFullyVectorizable(bool ForReduction) const {
cast<VectorType>(
getWidenedType(Back.Scalars.front()->getType(), BackVF)),
APInt::getAllOnes(BackVF),
- /*Insert=*/true, /*Extract=*/false,
- TTI::TCK_RecipThroughput) > -SLPCostThreshold)
+ /*Insert=*/true, /*Extract=*/false, CostKind) > -SLPCostThreshold)
return false;
}
@@ -18596,10 +18604,9 @@ InstructionCost BoUpSLP::getSpillCost() {
if (!Inserted)
return It->second;
IntrinsicCostAttributes ICA(II->getIntrinsicID(), *II);
- InstructionCost IntrCost =
- TTI->getIntrinsicInstrCost(ICA, TTI::TCK_RecipThroughput);
+ InstructionCost IntrCost = TTI->getIntrinsicInstrCost(ICA, CostKind);
InstructionCost CallCost = TTI->getCallInstrCost(
- nullptr, II->getType(), ICA.getArgTypes(), TTI::TCK_RecipThroughput);
+ nullptr, II->getType(), ICA.getArgTypes(), CostKind);
bool Res = IntrCost < CallCost;
It->second = Res;
return Res;
@@ -19186,7 +19193,6 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
return false;
return IsExternallyUsedV(V);
};
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
InstructionCost Cost = 0;
SmallDenseMap<const TreeEntry *, uint64_t> EntryToScale;
uint64_t PrevScale = 0;
@@ -19691,7 +19697,7 @@ InstructionCost BoUpSLP::getTreeCost(InstructionCost TreeCost,
return TE.hasState() && !DeletedNodes.contains(&TE) && !TE.isGather() &&
!TransformedToGatherNodes.contains(&TE) &&
TE.State != TreeEntry::CombinedVectorize &&
- isPoorThroughputOp(TE.getMainOp(), *TTI, *TLI, Cache);
+ isPoorThroughputOp(TE.getMainOp(), *TTI, *TLI, Cache, CostKind);
});
};
// Reject vectorization if the vector code would produce more instructions
@@ -19941,7 +19947,6 @@ InstructionCost BoUpSLP::getTreeCost(InstructionCost TreeCost,
else
VecOpcode =
It->second.second ? Instruction::SExt : Instruction::ZExt;
- TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
InstructionCost C = TTI->getCastInstrCost(
VecOpcode, FTy,
getWidenedType(IntegerType::get(FTy->getContext(), BWSz),
@@ -19969,7 +19974,6 @@ InstructionCost BoUpSLP::getTreeCost(InstructionCost TreeCost,
}
}
- TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
// If we plan to rewrite the tree in a smaller type, we will need to sign
// extend the extracted value back to the original type. Here, we account
// for the extract and the added cost of the sign extend if needed.
@@ -19999,8 +20003,8 @@ InstructionCost BoUpSLP::getTreeCost(InstructionCost TreeCost,
? Instruction::ZExt
: Instruction::SExt;
VecTy = getWidenedType(MinTy, BundleWidth);
- ExtraCost = getExtractWithExtendCost(*TTI, Extend, ScalarTy,
- cast<VectorType>(VecTy), EU.Lane);
+ ExtraCost = getExtractWithExtendCost(
+ *TTI, Extend, ScalarTy, cast<VectorType>(VecTy), EU.Lane, CostKind);
LLVM_DEBUG(dbgs() << " ExtractExtend or ExtractSubvec cost: "
<< ExtraCost << "\n");
} else {
@@ -20211,7 +20215,6 @@ InstructionCost BoUpSLP::getTreeCost(InstructionCost TreeCost,
}
}
if (!AnyRootKeptAsScalar && HaveCommonBase) {
- TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
auto *VecTy = getWidenedType(UserScalarTy, RootEntry.Scalars.size());
InstructionCost ScalarGEPCost = TTI->getPointersChainCost(
Pointers, CommonBase, TTI::PointersChainInfo::getUnitStride(),
@@ -20256,10 +20259,8 @@ InstructionCost BoUpSLP::getTreeCost(InstructionCost TreeCost,
assert(SLPReVec && "Only supported by REVEC.");
SrcTy = getWidenedType(SrcTy, VecTy->getNumElements());
}
- InstructionCost CastCost =
- TTI->getCastInstrCost(Opcode, DstTy, SrcTy,
- TTI::CastContextHint::None,
- TTI::TCK_RecipThroughput);
+ InstructionCost CastCost = TTI->getCastInstrCost(
+ Opcode, DstTy, SrcTy, TTI::CastContextHint::None, CostKind);
CastCost = ScaleCost(CastCost, Root, /*Scalar=*/nullptr, ReductionRoot);
Cost += CastCost;
}
@@ -20384,7 +20385,7 @@ InstructionCost BoUpSLP::getTreeCost(InstructionCost TreeCost,
cast<FixedVectorType>(
ShuffledInserts[I].InsertElements.front()->getType()),
DemandedElts[I],
- /*Insert*/ true, /*Extract*/ false, TTI::TCK_RecipThroughput);
+ /*Insert*/ true, /*Extract*/ false, CostKind);
Cost -= InsertCost;
}
@@ -20431,8 +20432,7 @@ InstructionCost BoUpSLP::getTreeCost(InstructionCost TreeCost,
break;
}
InstructionCost CastCost =
- TTI->getCastInstrCost(Opcode, DstVecTy, SrcVecTy, CCH,
- TTI::TCK_RecipThroughput);
+ TTI->getCastInstrCost(Opcode, DstVecTy, SrcVecTy, CCH, CostKind);
CastCost = ScaleCost(CastCost, *VectorizableTree.front().get(),
/*Scalar=*/nullptr, ReductionRoot);
Cost += CastCost;
@@ -21257,7 +21257,6 @@ BoUpSLP::isGatherShuffledSingleRegisterEntry(
NewVF = VF;
}
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
auto *VecTy =
cast<VectorType>(getWidenedType(VL.front()->getType(), NewVF));
auto *MaskVecTy =
@@ -21443,7 +21442,6 @@ InstructionCost BoUpSLP::getGatherCost(ArrayRef<Value *> VL, bool ForPoisonSrc,
// Check if the same elements are inserted several times and count them as
// shuffle candidates.
APInt DemandedElements = APInt::getZero(VF);
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
InstructionCost Cost;
auto EstimateInsertCost = [&](unsigned I, Value *V) {
DemandedElements.setBit(I);
@@ -22990,7 +22988,6 @@ ResTy BoUpSLP::processBuildVector(const TreeEntry *E, Type *ScalarTy,
auto CheckIfSplatIsProfitable = [&]() {
// Estimate the cost of splatting + shuffle and compare with
// insert + shuffle.
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
Value *V = *find_if_not(NonConstants, IsaPred<UndefValue>);
if (isa<ExtractElementInst>(V) || isVectorized(V))
return false;
@@ -24121,8 +24118,8 @@ Value *BoUpSLP::vectorizeTree(TreeEntry *E) {
Value *V = nullptr;
unsigned NumElts = E->Scalars.size();
FixedVectorType *PaddedVecTy = nullptr;
- if (getMaskedDivRemCost(*TTI, ShuffleOrOp, ScalarTy, NumElts,
- TTI::TCK_RecipThroughput, &PaddedVecTy)
+ if (getMaskedDivRemCost(*TTI, ShuffleOrOp, ScalarTy, NumElts, CostKind,
+ &PaddedVecTy)
.isValid()) {
assert(PaddedVecTy && "Expected padded type for masked div/rem.");
// Scale the lane count up to elements for REVEC, where each lane
@@ -24412,7 +24409,8 @@ Value *BoUpSLP::vectorizeTree(TreeEntry *E) {
SmallVector<Type *> ArgTys = buildIntrinsicArgTypes(
CI, ID, getNumElements(VecTy),
It != MinBWs.end() ? It->second.first : 0, TTI);
- auto VecCallCosts = getVectorCallCosts(CI, VecTy, TTI, TLI, ArgTys);
+ auto VecCallCosts =
+ getVectorCallCosts(CI, VecTy, TTI, TLI, ArgTys, CostKind);
bool UseIntrinsic = ID != Intrinsic::not_intrinsic &&
VecCallCosts.first <= VecCallCosts.second;
@@ -25041,7 +25039,6 @@ bool BoUpSLP::canVersionForRuntimeChecks() {
// relative to the guarded scalar region to avoid pessimizing that path too
// much. The scalar region cost is the cost of the (current, still scalar)
// block body; no IR is emitted here.
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
InstructionCost ScalarCost = 0;
for (Instruction &I : *BB) {
if (isa<PHINode>(&I) || I.isTerminator())
@@ -25067,27 +25064,26 @@ InstructionCost BoUpSLP::getRuntimeChecksCost() const {
LLVMContext &Ctx = F->getContext();
Type *IntTy = DL->getIntPtrType(Ctx);
Type *I1Ty = Type::getInt1Ty(Ctx);
- constexpr TTI::TargetCostKind Kind = TTI::TCK_RecipThroughput;
unsigned NumBases = RTChecks.Bounds.size();
unsigned NumPairs = RTChecks.BasePairs.size();
InstructionCost Cost = 0;
// Per base: a [Low, High) address pair. canVersionForRuntimeChecks() folds
// each bound to a base-plus-constant offset, so it expands to two integer
// adds (one per bound) with no runtime umin/umax reduction.
- Cost += TTI->getArithmeticInstrCost(Instruction::Add, IntTy, Kind) *
+ Cost += TTI->getArithmeticInstrCost(Instruction::Add, IntTy, CostKind) *
(2 * NumBases);
// Two integer compares and one logical and per checked pair.
InstructionCost CmpCost = TTI->getCmpSelInstrCost(
- Instruction::ICmp, IntTy, I1Ty, CmpInst::ICMP_ULT, Kind);
+ Instruction::ICmp, IntTy, I1Ty, CmpInst::ICMP_ULT, CostKind);
InstructionCost AndCost =
- TTI->getArithmeticInstrCost(Instruction::And, I1Ty, Kind);
+ TTI->getArithmeticInstrCost(Instruction::And, I1Ty, CostKind);
Cost += (CmpCost * 2 + AndCost) * NumPairs;
// Or-reduction of the per-pair conflicts.
if (NumPairs > 1)
- Cost += TTI->getArithmeticInstrCost(Instruction::Or, I1Ty, Kind) *
+ Cost += TTI->getArithmeticInstrCost(Instruction::Or, I1Ty, CostKind) *
(NumPairs - 1);
// The guard branch.
- Cost += TTI->getCFInstrCost(Instruction::CondBr, Kind);
+ Cost += TTI->getCFInstrCost(Instruction::CondBr, CostKind);
return Cost;
}
@@ -28093,7 +28089,7 @@ bool BoUpSLP::collectValuesToDemote(
buildIntrinsicArgTypes(IC, ID, VF, MinBW, TTI);
auto VecCallCosts = getVectorCallCosts(
IC, getWidenedType(IntegerType::get(IC->getContext(), MinBW), VF),
- TTI, TLI, ArgTys);
+ TTI, TLI, ArgTys, CostKind);
InstructionCost Cost = std::min(VecCallCosts.first, VecCallCosts.second);
if (Cost < BestCost) {
BestCost = Cost;
@@ -31806,7 +31802,7 @@ class HorizontalReduction {
const SmallMapVector<Value *, unsigned, 16> SameValuesCounter,
bool IsCmpSelMinMax, FastMathFlags FMF, const BoUpSLP &R,
DominatorTree &DT, const DataLayout &DL, const TargetLibraryInfo &TLI) {
- TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
+ TTI::TargetCostKind CostKind = R.getCostKind();
Type *ScalarTy = ReducedVals.front()->getType();
unsigned ReduxWidth = ReducedVals.size();
FixedVectorType *VectorTy = R.getReductionType();
@@ -31853,8 +31849,9 @@ class HorizontalReduction {
auto *RdxOp = cast<Instruction>(U);
if (hasRequiredNumberOfUses(IsCmpSelMinMax, RdxOp)) {
if (RdxKind == RecurKind::FAdd) {
- InstructionCost FMACost = canConvertToFMA(
- RdxOp, getSameOpcode(RdxOp, TLI), DT, DL, *TTI, TLI);
+ InstructionCost FMACost =
+ canConvertToFMA(RdxOp, getSameOpcode(RdxOp, TLI), DT, DL,
+ *TTI, TLI, CostKind);
if (FMACost.isValid()) {
LLVM_DEBUG(dbgs() << "FMA cost: " << FMACost << "\n");
if (auto *I = dyn_cast<Instruction>(RdxVal)) {
@@ -31954,7 +31951,7 @@ class HorizontalReduction {
}
if (!Ops.empty()) {
FMACost = canConvertToFMA(Ops, getSameOpcode(Ops, TLI), DT, DL,
- *TTI, TLI);
+ *TTI, TLI, CostKind);
if (FMACost.isValid()) {
// Calculate actual FMAD cost.
IntrinsicCostAttributes ICA(Intrinsic::fmuladd, RVecTy,
@@ -32759,7 +32756,8 @@ bool SLPVectorizerPass::tryToVectorize(
if (!AllowFMACandidates &&
(I->getOpcode() == Instruction::FAdd ||
I->getOpcode() == Instruction::FSub) &&
- canConvertToFMA(I, getSameOpcode(I, *TLI), *DT, *DL, *TTI, *TLI)
+ canConvertToFMA(I, getSameOpcode(I, *TLI), *DT, *DL, *TTI, *TLI,
+ R.getCostKind())
.isValid()) {
FMACandidates.insert(I);
return false;
@@ -32811,7 +32809,7 @@ bool SLPVectorizerPass::tryToVectorize(
return false;
// Check the cost of operations.
auto *VecTy = cast<VectorType>(getWidenedType(Ty, Ops.size()));
- constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
+ const TTI::TargetCostKind CostKind = R.getCostKind();
InstructionCost ScalarCost =
TTI.getScalarizationOverhead(
VecTy, APInt::getAllOnes(getNumElements(VecTy)), /*Insert=*/false,
@@ -33888,7 +33886,8 @@ bool SLPVectorizerPass::vectorizeChainsInBlock(BasicBlock *BB, BoUpSLP &R) {
else if (isNonVectorizableInst(&*It, TLI))
PostProcessInsts.insert(&*It);
else if (VectorizePoorThroughput &&
- isPoorThroughputOp(&*It, *TTI, *TLI, PoorThroughputCache))
+ isPoorThroughputOp(&*It, *TTI, *TLI, PoorThroughputCache,
+ R.getCostKind()))
PoorThroughputSeeds.insert(&*It);
}
@@ -33919,12 +33918,11 @@ bool SLPVectorizerPass::vectorizeChainsInBlock(BasicBlock *BB, BoUpSLP &R) {
ScalarTy = ::getWidenedType(ScalarTy, getNumElements(ValTy));
auto *VecTy = cast<VectorType>(
::getWidenedType(ScalarTy, PostProcessStores.size()));
- InstructionCost ExtractsCost = ::getScalarizationOverhead(
- *TTI, ScalarTy, VecTy,
- APInt::getAllOnes(PostProcessStores.size()),
- /*Insert=*/false, /*Extract=*/true, TTI::TCK_RecipThroughput,
- /*ForPoisonSrc=*/true, {}, TTI::VectorInstrContext::Store);
- TryVectorize = ExtractsCost <= PostProcessStores.size() + 1;
+ InstructionCost ExtractsCost = ::getScalarizationOverhead(
+ *TTI, ScalarTy, VecTy, APInt::getAllOnes(PostProcessStores.size()),
+ /*Insert=*/false, /*Extract=*/true, R.getCostKind(),
+ /*ForPoisonSrc=*/true, {}, TTI::VectorInstrContext::Store);
+ TryVectorize = ExtractsCost <= PostProcessStores.size() + 1;
}
}
}
@@ -34026,7 +34024,8 @@ bool SLPVectorizerPass::vectorizeOnceUsedSeeds(BasicBlock *BB, BoUpSLP &R) {
// The poor-throughput ops are seeded on their own, with the different
// grouping.
if (VectorizePoorThroughput &&
- isPoorThroughputOp(&I, *TTI, *TLI, PoorThroughputCache))
+ isPoorThroughputOp(&I, *TTI, *TLI, PoorThroughputCache,
+ R.getCostKind()))
continue;
// The multiplication is contracted into the scalar FMA with its user, the
// vector node breaks the contraction.
@@ -34034,7 +34033,8 @@ bool SLPVectorizerPass::vectorizeOnceUsedSeeds(BasicBlock *BB, BoUpSLP &R) {
auto *U = cast<Instruction>(I.user_back());
if (InstructionsState S = getSameOpcode(U, *TLI);
S && S.isAddSubLikeOp() &&
- canConvertToFMA(U, S, *DT, *DL, *TTI, *TLI).isValid())
+ canConvertToFMA(U, S, *DT, *DL, *TTI, *TLI, R.getCostKind())
+ .isValid())
continue;
}
// The keys are hashes, so the groups are numbered by the first seed to
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/cost-size.ll b/llvm/test/Transforms/SLPVectorizer/X86/cost-size.ll
index 6e15f2f001065..d8c4c2a9a7f39 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/cost-size.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/cost-size.ll
@@ -9,8 +9,11 @@ define i16 @test_optsize(ptr %p, ptr %inc) #0 {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[E0:%.*]] = load i16, ptr [[P]], align 4
; CHECK-NEXT: [[E1:%.*]] = load i16, ptr [[INC]], align 2
-; CHECK-NEXT: [[TMP3:%.*]] = udiv i16 [[E0]], 13
-; CHECK-NEXT: [[TMP4:%.*]] = udiv i16 [[E1]], 14
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x i16> poison, i16 [[E0]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x i16> [[TMP0]], i16 [[E1]], i64 1
+; CHECK-NEXT: [[TMP2:%.*]] = udiv <2 x i16> [[TMP1]], <i16 13, i16 14>
+; CHECK-NEXT: [[TMP3:%.*]] = extractelement <2 x i16> [[TMP2]], i64 0
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x i16> [[TMP2]], i64 1
; CHECK-NEXT: [[A:%.*]] = add i16 [[TMP3]], [[TMP4]]
; CHECK-NEXT: ret i16 [[A]]
;
@@ -30,8 +33,11 @@ define i16 @testc_optsize(ptr %p, ptr %inc) #1 {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[E0:%.*]] = load i16, ptr [[P]], align 4
; CHECK-NEXT: [[E1:%.*]] = load i16, ptr [[INC]], align 2
-; CHECK-NEXT: [[TMP3:%.*]] = udiv i16 [[E0]], 13
-; CHECK-NEXT: [[TMP4:%.*]] = udiv i16 [[E1]], 14
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x i16> poison, i16 [[E0]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x i16> [[TMP0]], i16 [[E1]], i64 1
+; CHECK-NEXT: [[TMP2:%.*]] = udiv <2 x i16> [[TMP1]], <i16 13, i16 14>
+; CHECK-NEXT: [[TMP3:%.*]] = extractelement <2 x i16> [[TMP2]], i64 0
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x i16> [[TMP2]], i64 1
; CHECK-NEXT: [[A:%.*]] = add i16 [[TMP3]], [[TMP4]]
; CHECK-NEXT: ret i16 [[A]]
;
More information about the llvm-commits
mailing list