[llvm] 875d2c9 - [LV][NFC] Factor out MinBWs of values from the cost model (#194492)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Apr 30 05:16:08 PDT 2026
Author: Hassnaa Hamdi
Date: 2026-04-30T13:16:03+01:00
New Revision: 875d2c9fbc1431fe4e70166d9fcb4be4c1d190ad
URL: https://github.com/llvm/llvm-project/commit/875d2c9fbc1431fe4e70166d9fcb4be4c1d190ad
DIFF: https://github.com/llvm/llvm-project/commit/875d2c9fbc1431fe4e70166d9fcb4be4c1d190ad.diff
LOG: [LV][NFC] Factor out MinBWs of values from the cost model (#194492)
Move MinBWs out of the CM to the planner, as it doesn't depend on the
CM.
Added:
Modified:
llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index 0e847f4767a8b..98e4da75b04fe 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -563,6 +563,10 @@ bool VFSelectionContext::runtimeChecksRequired() {
return false;
}
+void VFSelectionContext::computeMinimalBitwidths() {
+ MinBWs = computeMinimumValueSizes(TheLoop->getBlocks(), *DB, &TTI);
+}
+
void VFSelectionContext::collectInLoopReductions() {
// Avoid duplicating work finding in-loop reductions.
if (!InLoopReductions.empty())
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 29ebf3b2147f6..7bdd5a3b34e8e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -552,6 +552,7 @@ class VFSelectionContext {
const Loop *TheLoop;
const Function &F;
PredicatedScalarEvolution &PSE;
+ DemandedBits *DB;
OptimizationRemarkEmitter *ORE;
const LoopVectorizeHints *Hints;
@@ -585,6 +586,11 @@ class VFSelectionContext {
/// computeFeasibleMaxVF.
std::optional<unsigned> MaxSafeElements;
+ /// Map of scalar integer values to the smallest bitwidth they can be legally
+ /// represented as. The vector equivalents of these values should be truncated
+ /// to this type.
+ MapVector<Instruction *, uint64_t> MinBWs;
+
public:
/// The kind of cost that we are calculating.
const TTI::TargetCostKind CostKind;
@@ -596,11 +602,11 @@ class VFSelectionContext {
VFSelectionContext(const TargetTransformInfo &TTI,
const LoopVectorizationLegality *Legal,
const Loop *TheLoop, const Function &F,
- PredicatedScalarEvolution &PSE,
+ PredicatedScalarEvolution &PSE, DemandedBits *DB,
OptimizationRemarkEmitter *ORE,
const LoopVectorizeHints *Hints, bool OptForSize)
- : TTI(TTI), Legal(Legal), TheLoop(TheLoop), F(F), PSE(PSE), ORE(ORE),
- Hints(Hints),
+ : TTI(TTI), Legal(Legal), TheLoop(TheLoop), F(F), PSE(PSE), DB(DB),
+ ORE(ORE), Hints(Hints),
CostKind(F.hasMinSize() ? TTI::TCK_CodeSize : TTI::TCK_RecipThroughput),
OptForSize(OptForSize) {
initializeVScaleForTuning();
@@ -688,6 +694,16 @@ class VFSelectionContext {
/// Check whether vectorization would require runtime checks. When optimizing
/// for size, returning true here aborts vectorization.
bool runtimeChecksRequired();
+
+ /// Compute smallest bitwidth each instruction can be represented with.
+ /// The vector equivalents of these instructions should be truncated to this
+ /// type.
+ void computeMinimalBitwidths();
+
+ /// \returns The smallest bitwidth each instruction can be represented with.
+ const MapVector<Instruction *, uint64_t> &getMinimalBitwidths() const {
+ return MinBWs;
+ }
};
/// Planner drives the vectorization process after having passed
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 6c9298a2cf98d..fdc2cfe34d47c 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -819,16 +819,18 @@ class LoopVectorizationCostModel {
friend class LoopVectorizationPlanner;
public:
- LoopVectorizationCostModel(
- EpilogueLowering SEL, Loop *L, PredicatedScalarEvolution &PSE,
- LoopInfo *LI, LoopVectorizationLegality *Legal,
- const TargetTransformInfo &TTI, const TargetLibraryInfo *TLI,
- DemandedBits *DB, AssumptionCache *AC, OptimizationRemarkEmitter *ORE,
- std::function<BlockFrequencyInfo &()> GetBFI, const Function *F,
- const LoopVectorizeHints *Hints, InterleavedAccessInfo &IAI,
- VFSelectionContext &Config)
+ LoopVectorizationCostModel(EpilogueLowering SEL, Loop *L,
+ PredicatedScalarEvolution &PSE, LoopInfo *LI,
+ LoopVectorizationLegality *Legal,
+ const TargetTransformInfo &TTI,
+ const TargetLibraryInfo *TLI, AssumptionCache *AC,
+ OptimizationRemarkEmitter *ORE,
+ std::function<BlockFrequencyInfo &()> GetBFI,
+ const Function *F, const LoopVectorizeHints *Hints,
+ InterleavedAccessInfo &IAI,
+ VFSelectionContext &Config)
: Config(Config), EpilogueLoweringStatus(SEL), TheLoop(L), PSE(PSE),
- LI(LI), Legal(Legal), TTI(TTI), TLI(TLI), DB(DB), AC(AC), ORE(ORE),
+ LI(LI), Legal(Legal), TTI(TTI), TLI(TLI), AC(AC), ORE(ORE),
GetBFI(GetBFI), TheFunction(F), Hints(Hints), InterleaveInfo(IAI) {}
/// \return An upper bound for the vectorization factors (both fixed and
@@ -855,13 +857,6 @@ class LoopVectorizationCostModel {
/// Collect values we want to ignore in the cost model.
void collectValuesToIgnore();
- /// \returns The smallest bitwidth each instruction can be represented with.
- /// The vector equivalents of these instructions should be truncated to this
- /// type.
- const MapVector<Instruction *, uint64_t> &getMinimalBitwidths() const {
- return MinBWs;
- }
-
/// \returns True if it is more profitable to scalarize instruction \p I for
/// vectorization factor \p VF.
bool isProfitableToScalarize(Instruction *I, ElementCount VF) const {
@@ -915,6 +910,7 @@ class LoopVectorizationCostModel {
/// \returns True if instruction \p I can be truncated to a smaller bitwidth
/// for vectorization factor \p VF.
bool canTruncateToMinimalBitwidth(Instruction *I, ElementCount VF) const {
+ const auto &MinBWs = Config.getMinimalBitwidths();
// Truncs must truncate at most to their destination type.
if (isa_and_nonnull<TruncInst>(I) && MinBWs.contains(I) &&
I->getType()->getScalarSizeInBits() < MinBWs.lookup(I))
@@ -1351,11 +1347,6 @@ class LoopVectorizationCostModel {
InstructionCost getScalarizationOverhead(Instruction *I,
ElementCount VF) const;
- /// Map of scalar integer values to the smallest bitwidth they can be legally
- /// represented as. The vector equivalents of these values should be truncated
- /// to this type.
- MapVector<Instruction *, uint64_t> MinBWs;
-
/// A type representing the costs for instructions if they were to be
/// scalarized rather than vectorized. The entries are Instruction-Cost
/// pairs.
@@ -1491,9 +1482,6 @@ class LoopVectorizationCostModel {
/// Target Library Info.
const TargetLibraryInfo *TLI;
- /// Demanded bits analysis.
- DemandedBits *DB;
-
/// Assumption cache.
AssumptionCache *AC;
@@ -2940,8 +2928,6 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
Uniforms.empty() && Scalars.empty() &&
"No cost-modeling decisions should have been taken at this point");
- MinBWs = computeMinimumValueSizes(TheLoop->getBlocks(), *DB, &TTI);
-
switch (EpilogueLoweringStatus) {
case CM_EpilogueAllowed:
return Config.computeFeasibleMaxVF(MaxTC, UserVF, UserIC, false,
@@ -5142,9 +5128,11 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
VF.getKnownMinValue();
}
+ const auto &MinBWs = Config.getMinimalBitwidths();
+ uint64_t InstrMinBWs = MinBWs.lookup(I);
Type *RetTy = I->getType();
if (canTruncateToMinimalBitwidth(I, VF))
- RetTy = IntegerType::get(RetTy->getContext(), MinBWs[I]);
+ RetTy = IntegerType::get(RetTy->getContext(), InstrMinBWs);
auto *SE = PSE.getSE();
Type *VectorTy;
@@ -5426,10 +5414,10 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
[[maybe_unused]] Instruction *Op0AsInstruction =
dyn_cast<Instruction>(I->getOperand(0));
assert((!canTruncateToMinimalBitwidth(Op0AsInstruction, VF) ||
- MinBWs[I] == MinBWs[Op0AsInstruction]) &&
+ InstrMinBWs == MinBWs.lookup(Op0AsInstruction)) &&
"if both the operand and the compare are marked for "
"truncation, they must have the same bitwidth");
- ValTy = IntegerType::get(ValTy->getContext(), MinBWs[I]);
+ ValTy = IntegerType::get(ValTy->getContext(), InstrMinBWs);
}
VectorTy = toVectorTy(ValTy, VF);
@@ -5529,8 +5517,8 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
Type *SrcScalarTy = I->getOperand(0)->getType();
Instruction *Op0AsInstruction = dyn_cast<Instruction>(I->getOperand(0));
if (canTruncateToMinimalBitwidth(Op0AsInstruction, VF))
- SrcScalarTy =
- IntegerType::get(SrcScalarTy->getContext(), MinBWs[Op0AsInstruction]);
+ SrcScalarTy = IntegerType::get(SrcScalarTy->getContext(),
+ MinBWs.lookup(Op0AsInstruction));
Type *SrcVecTy =
VectorTy->isVectorTy() ? toVectorTy(SrcScalarTy, VF) : SrcScalarTy;
@@ -5783,6 +5771,10 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
if (!MaxFactors) // Cases that should not to be vectorized nor interleaved.
return;
+ // Compute the minimal bitwidths required for integer operations in the loop
+ // for later use by the cost model.
+ Config.computeMinimalBitwidths();
+
// Invalidate interleave groups if all blocks of loop will be predicated.
if (CM.blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
!useMaskedInterleavedAccesses(TTI)) {
@@ -6908,7 +6900,7 @@ void LoopVectorizationPlanner::buildVPlansWithVPRecipes(ElementCount MinVF,
RUN_VPLAN_PASS(VPlanTransforms::hoistPredicatedLoads, *Plan, PSE, OrigLoop);
RUN_VPLAN_PASS(VPlanTransforms::sinkPredicatedStores, *Plan, PSE, OrigLoop);
RUN_VPLAN_PASS(VPlanTransforms::truncateToMinimalBitwidths, *Plan,
- CM.getMinimalBitwidths());
+ Config.getMinimalBitwidths());
RUN_VPLAN_PASS(VPlanTransforms::optimize, *Plan);
// TODO: try to put addExplicitVectorLength close to addActiveLaneMask
if (CM.foldTailWithEVL()) {
@@ -7505,8 +7497,8 @@ static bool processLoopInVPlanNativePath(
EpilogueLowering SEL =
getEpilogueLowering(F, L, Hints, OptForSize, TTI, TLI, *LVL, &IAI);
- VFSelectionContext Config(*TTI, LVL, L, *F, PSE, ORE, &Hints, OptForSize);
- LoopVectorizationCostModel CM(SEL, L, PSE, LI, LVL, *TTI, TLI, DB, AC, ORE,
+ VFSelectionContext Config(*TTI, LVL, L, *F, PSE, DB, ORE, &Hints, OptForSize);
+ LoopVectorizationCostModel CM(SEL, L, PSE, LI, LVL, *TTI, TLI, AC, ORE,
GetBFI, F, &Hints, IAI, Config);
// Use the planner for outer loop vectorization.
// TODO: CM is not used at this point inside the planner. Turn CM into an
@@ -8338,8 +8330,9 @@ bool LoopVectorizePass::processLoop(Loop *L) {
}
// Use the cost model.
- VFSelectionContext Config(*TTI, &LVL, L, *F, PSE, ORE, &Hints, OptForSize);
- LoopVectorizationCostModel CM(SEL, L, PSE, LI, &LVL, *TTI, TLI, DB, AC, ORE,
+ VFSelectionContext Config(*TTI, &LVL, L, *F, PSE, DB, ORE, &Hints,
+ OptForSize);
+ LoopVectorizationCostModel CM(SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE,
GetBFI, F, &Hints, IAI, Config);
// Use the planner for vectorization.
LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, CM, Config, IAI, PSE,
More information about the llvm-commits
mailing list