[llvm] [LV][NFC] Rename ScalarEpilogueLowering to remove 'Scalar' term. (PR #191871)
Hassnaa Hamdi via llvm-commits
llvm-commits at lists.llvm.org
Mon Apr 13 11:51:19 PDT 2026
https://github.com/hassnaaHamdi created https://github.com/llvm/llvm-project/pull/191871
Rename ScalarEpilogueLowering enum to EpilogueLowering and update related code.
The term 'scalar' here is misleading given that the epilogue could get vectorized.
>From 0dac0b181aaad40c332997dfda1badf9e85d708d Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Mon, 13 Apr 2026 17:39:14 +0000
Subject: [PATCH] [LV][NFC] Rename ScalarEpilogueLowering to remove 'Scalar'
term.
Rename ScalarEpilogueLowering enum to EpilogueLowering and update
related code.
The term 'scalar' here is misleading given that the epilogue could
get vectorized.
---
.../llvm/Analysis/TargetTransformInfo.h | 4 +-
.../llvm/Analysis/TargetTransformInfoImpl.h | 2 +-
llvm/include/llvm/CodeGen/BasicTTIImpl.h | 4 +-
llvm/lib/Analysis/TargetTransformInfo.cpp | 4 +-
.../AArch64/AArch64TargetTransformInfo.cpp | 6 +-
.../AArch64/AArch64TargetTransformInfo.h | 2 +-
.../lib/Target/ARM/ARMTargetTransformInfo.cpp | 19 +-
llvm/lib/Target/ARM/ARMTargetTransformInfo.h | 2 +-
.../Target/RISCV/RISCVTargetTransformInfo.h | 2 +-
.../Transforms/Vectorize/LoopVectorize.cpp | 169 +++++++++---------
10 files changed, 107 insertions(+), 107 deletions(-)
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index 3c4f00c0d87b5..5b427bbf1042f 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -754,9 +754,9 @@ class TargetTransformInfo {
// vectorization should be considered.
LLVM_ABI unsigned getEpilogueVectorizationMinVF() const;
- /// Query the target whether it would be prefered to create a predicated
+ /// Query the target whether it would be prefered to create a tail-folded
/// vector loop, which can avoid the need to emit a scalar epilogue loop.
- LLVM_ABI bool preferPredicateOverEpilogue(TailFoldingInfo *TFI) const;
+ LLVM_ABI bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const;
/// Query the target what the preferred style of tail folding is.
LLVM_ABI TailFoldingStyle getPreferredTailFoldingStyle() const;
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 7eb363c7b4404..3221f6b63e4f4 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -274,7 +274,7 @@ class TargetTransformInfoImplBase {
virtual unsigned getEpilogueVectorizationMinVF() const { return 16; }
- virtual bool preferPredicateOverEpilogue(TailFoldingInfo *TFI) const {
+ virtual bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const {
return false;
}
diff --git a/llvm/include/llvm/CodeGen/BasicTTIImpl.h b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
index 7dbd8bc658161..6dd3e264c4870 100644
--- a/llvm/include/llvm/CodeGen/BasicTTIImpl.h
+++ b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
@@ -801,8 +801,8 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
return BaseT::getEpilogueVectorizationMinVF();
}
- bool preferPredicateOverEpilogue(TailFoldingInfo *TFI) const override {
- return BaseT::preferPredicateOverEpilogue(TFI);
+ bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override {
+ return BaseT::preferTailFoldingOverEpilogue(TFI);
}
TailFoldingStyle getPreferredTailFoldingStyle() const override {
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 5111593d76a6d..28a418897a1d1 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -379,9 +379,9 @@ unsigned TargetTransformInfo::getEpilogueVectorizationMinVF() const {
return TTIImpl->getEpilogueVectorizationMinVF();
}
-bool TargetTransformInfo::preferPredicateOverEpilogue(
+bool TargetTransformInfo::preferTailFoldingOverEpilogue(
TailFoldingInfo *TFI) const {
- return TTIImpl->preferPredicateOverEpilogue(TFI);
+ return TTIImpl->preferTailFoldingOverEpilogue(TFI);
}
TailFoldingStyle TargetTransformInfo::getPreferredTailFoldingStyle() const {
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 734339e5c7a05..2d306b3eccbe6 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -6469,7 +6469,7 @@ unsigned AArch64TTIImpl::getEpilogueVectorizationMinVF() const {
return ST->getEpilogueVectorizationMinVF();
}
-bool AArch64TTIImpl::preferPredicateOverEpilogue(TailFoldingInfo *TFI) const {
+bool AArch64TTIImpl::preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const {
if (!ST->hasSVE())
return false;
@@ -6486,8 +6486,8 @@ bool AArch64TTIImpl::preferPredicateOverEpilogue(TailFoldingInfo *TFI) const {
Required |= TailFoldingOpts::Recurrences;
// We call this to discover whether any load/store pointers in the loop have
- // negative strides. This will require extra work to reverse the loop
- // predicate, which may be expensive.
+ // negative strides. This will require extra work to reverse the tail-folded
+ // loop, which may be expensive.
if (containsDecreasingPointers(TFI->LVL->getLoop(),
TFI->LVL->getPredicatedScalarEvolution(),
*TFI->LVL->getDominatorTree()))
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index cde391bdcaea8..db9d361b2d92a 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -473,7 +473,7 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
unsigned getEpilogueVectorizationMinVF() const override;
- bool preferPredicateOverEpilogue(TailFoldingInfo *TFI) const override;
+ bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override;
bool supportsScalableVectors() const override {
return ST->isSVEorStreamingSVEAvailable();
diff --git a/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp b/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
index 03392fc2d84bc..c1df7fbb9d702 100644
--- a/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
+++ b/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
@@ -2616,14 +2616,14 @@ static bool canTailPredicateLoop(Loop *L, LoopInfo *LI, ScalarEvolution &SE,
return true;
}
-bool ARMTTIImpl::preferPredicateOverEpilogue(TailFoldingInfo *TFI) const {
+bool ARMTTIImpl::preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const {
if (!EnableTailPredication) {
- LLVM_DEBUG(dbgs() << "Tail-predication not enabled.\n");
+ LLVM_DEBUG(dbgs() << "Tail-folding not enabled.\n");
return false;
}
- // Creating a predicated vector loop is the first step for generating a
- // tail-predicated hardware loop, for which we need the MVE masked
+ // Creating a tail-folded vector loop is the first step for generating a
+ // tail-folded hardware loop, for which we need the MVE masked
// load/stores instructions:
if (!ST->hasMVEIntegerOps())
return false;
@@ -2633,17 +2633,18 @@ bool ARMTTIImpl::preferPredicateOverEpilogue(TailFoldingInfo *TFI) const {
// For now, restrict this to single block loops.
if (L->getNumBlocks() > 1) {
- LLVM_DEBUG(dbgs() << "preferPredicateOverEpilogue: not a single block "
+ LLVM_DEBUG(dbgs() << "preferTailFoldingOverEpilogue: not a single block "
"loop.\n");
return false;
}
- assert(L->isInnermost() && "preferPredicateOverEpilogue: inner-loop expected");
+ assert(L->isInnermost() &&
+ "preferTailFoldingOverEpilogue: inner-loop expected");
LoopInfo *LI = LVL->getLoopInfo();
HardwareLoopInfo HWLoopInfo(L);
if (!HWLoopInfo.canAnalyze(*LI)) {
- LLVM_DEBUG(dbgs() << "preferPredicateOverEpilogue: hardware-loop is not "
+ LLVM_DEBUG(dbgs() << "preferTailFoldingOverEpilogue: hardware-loop is not "
"analyzable.\n");
return false;
}
@@ -2654,14 +2655,14 @@ bool ARMTTIImpl::preferPredicateOverEpilogue(TailFoldingInfo *TFI) const {
// This checks if we have the low-overhead branch architecture
// extension, and if we will create a hardware-loop:
if (!isHardwareLoopProfitable(L, *SE, *AC, TFI->TLI, HWLoopInfo)) {
- LLVM_DEBUG(dbgs() << "preferPredicateOverEpilogue: hardware-loop is not "
+ LLVM_DEBUG(dbgs() << "preferTailFoldingOverEpilogue: hardware-loop is not "
"profitable.\n");
return false;
}
DominatorTree *DT = LVL->getDominatorTree();
if (!HWLoopInfo.isHardwareLoopCandidate(*SE, *LI, *DT)) {
- LLVM_DEBUG(dbgs() << "preferPredicateOverEpilogue: hardware-loop is not "
+ LLVM_DEBUG(dbgs() << "preferTailFoldingOverEpilogue: hardware-loop is not "
"a candidate.\n");
return false;
}
diff --git a/llvm/lib/Target/ARM/ARMTargetTransformInfo.h b/llvm/lib/Target/ARM/ARMTargetTransformInfo.h
index 0d6d5d202bddf..e824839e39159 100644
--- a/llvm/lib/Target/ARM/ARMTargetTransformInfo.h
+++ b/llvm/lib/Target/ARM/ARMTargetTransformInfo.h
@@ -436,7 +436,7 @@ class ARMTTIImpl final : public BasicTTIImplBase<ARMTTIImpl> {
bool isHardwareLoopProfitable(Loop *L, ScalarEvolution &SE,
AssumptionCache &AC, TargetLibraryInfo *LibInfo,
HardwareLoopInfo &HWLoopInfo) const override;
- bool preferPredicateOverEpilogue(TailFoldingInfo *TFI) const override;
+ bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override;
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
TTI::UnrollingPreferences &UP,
OptimizationRemarkEmitter *ORE) const override;
diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
index 18e3ea0554881..9134bd20fccff 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
@@ -121,7 +121,7 @@ class RISCVTTIImpl final : public BasicTTIImplBase<RISCVTTIImpl> {
bool enableScalableVectorization() const override {
return ST->hasVInstructions();
}
- bool preferPredicateOverEpilogue(TailFoldingInfo *TFI) const override {
+ bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override {
return ST->hasVInstructions();
}
TailFoldingStyle getPreferredTailFoldingStyle() const override {
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index a7bbd94a0c776..cca84b5232df7 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -838,27 +838,27 @@ static void reportVectorization(OptimizationRemarkEmitter *ORE, Loop *TheLoop,
namespace llvm {
-// Loop vectorization cost-model hints how the scalar epilogue loop should be
+// Loop vectorization cost-model hints how the epilogue/tail loop should be
// lowered.
-enum ScalarEpilogueLowering {
+enum EpilogueLowering {
- // The default: allowing scalar epilogues.
- CM_ScalarEpilogueAllowed,
+ // The default: allowing epilogues.
+ CM_EpilogueAllowed,
// Vectorization with OptForSize: don't allow epilogues.
- CM_ScalarEpilogueNotAllowedOptSize,
+ CM_EpilogueNotAllowedOptSize,
// A special case of vectorisation with OptForSize: loops with a very small
// trip count are considered for vectorization under OptForSize, thereby
// making sure the cost of their loop body is dominant, free of runtime
// guards and scalar iteration overheads.
- CM_ScalarEpilogueNotAllowedLowTripLoop,
+ CM_EpilogueNotAllowedLowTripLoop,
- // Loop hint predicate indicating an epilogue is undesired.
- CM_ScalarEpilogueNotNeededUsePredicate,
+ // Loop hint indicating an epilogue is undesired, apply tail folding.
+ CM_EpilogueNotNeededFoldTail,
- // Directive indicating we must either tail fold or not vectorize
- CM_ScalarEpilogueNotAllowedUsePredicate
+ // Directive indicating we must either fold the epilogue/tail or not vectorize
+ CM_EpilogueNotAllowedFoldTail
};
/// LoopVectorizationCostModel - estimates the expected speedups due to
@@ -872,7 +872,7 @@ class LoopVectorizationCostModel {
friend class LoopVectorizationPlanner;
public:
- LoopVectorizationCostModel(ScalarEpilogueLowering SEL, Loop *L,
+ LoopVectorizationCostModel(EpilogueLowering SEL, Loop *L,
PredicatedScalarEvolution &PSE, LoopInfo *LI,
LoopVectorizationLegality *Legal,
const TargetTransformInfo &TTI,
@@ -882,7 +882,7 @@ class LoopVectorizationCostModel {
std::function<BlockFrequencyInfo &()> GetBFI,
const Function *F, const LoopVectorizeHints *Hints,
InterleavedAccessInfo &IAI, bool OptForSize)
- : ScalarEpilogueStatus(SEL), TheLoop(L), PSE(PSE), LI(LI), Legal(Legal),
+ : EpilogueLoweringStatus(SEL), TheLoop(L), PSE(PSE), LI(LI), Legal(Legal),
TTI(TTI), TLI(TLI), DB(DB), AC(AC), ORE(ORE), GetBFI(GetBFI),
TheFunction(F), Hints(Hints), InterleaveInfo(IAI),
OptForSize(OptForSize) {
@@ -1289,7 +1289,7 @@ class LoopVectorizationCostModel {
/// Returns true if we're required to use a scalar epilogue for at least
/// the final iteration of the original loop.
bool requiresScalarEpilogue(bool IsVectorizing) const {
- if (!isScalarEpilogueAllowed()) {
+ if (!isEpilogueAllowed()) {
LLVM_DEBUG(dbgs() << "LV: Loop does not require scalar epilogue\n");
return false;
}
@@ -1310,16 +1310,16 @@ class LoopVectorizationCostModel {
return false;
}
- /// Returns true if a scalar epilogue is allowed (e.g.., not prevented by
+ /// Returns true if an epilogue is allowed (e.g.., not prevented by
/// optsize or a loop hint annotation).
- bool isScalarEpilogueAllowed() const {
- return ScalarEpilogueStatus == CM_ScalarEpilogueAllowed;
+ bool isEpilogueAllowed() const {
+ return EpilogueLoweringStatus == CM_EpilogueAllowed;
}
- /// Returns true if tail-folding is preferred over a scalar epilogue.
- bool preferPredicatedLoop() const {
- return ScalarEpilogueStatus == CM_ScalarEpilogueNotNeededUsePredicate ||
- ScalarEpilogueStatus == CM_ScalarEpilogueNotAllowedUsePredicate;
+ /// Returns true if tail-folding is preferred over an epilogue.
+ bool preferTailFoldedLoop() const {
+ return EpilogueLoweringStatus == CM_EpilogueNotNeededFoldTail ||
+ EpilogueLoweringStatus == CM_EpilogueNotAllowedFoldTail;
}
/// Returns the TailFoldingStyle that is best for the current loop.
@@ -1351,10 +1351,10 @@ class LoopVectorizationCostModel {
TTI.hasActiveVectorLength() && !EnableVPlanNativePath;
if (EVLIsLegal)
return;
- // If for some reason EVL mode is unsupported, fallback to a scalar epilogue
+ // If for some reason EVL mode is unsupported, fallback to an epilogue
// if it's allowed, or DataWithoutLaneMask otherwise.
- if (ScalarEpilogueStatus == CM_ScalarEpilogueAllowed ||
- ScalarEpilogueStatus == CM_ScalarEpilogueNotNeededUsePredicate)
+ if (EpilogueLoweringStatus == CM_EpilogueAllowed ||
+ EpilogueLoweringStatus == CM_EpilogueNotNeededFoldTail)
ChosenTailFoldingStyle = TailFoldingStyle::None;
else
ChosenTailFoldingStyle = TailFoldingStyle::DataWithoutLaneMask;
@@ -1588,7 +1588,7 @@ class LoopVectorizationCostModel {
/// or as a peel-loop to handle gaps in interleave-groups.
/// Under optsize and when the trip count is very small we don't allow any
/// iterations to execute in the scalar loop.
- ScalarEpilogueLowering ScalarEpilogueStatus = CM_ScalarEpilogueAllowed;
+ EpilogueLowering EpilogueLoweringStatus = CM_EpilogueAllowed;
/// Control finally chosen tail folding style.
TailFoldingStyle ChosenTailFoldingStyle = TailFoldingStyle::None;
@@ -2764,7 +2764,7 @@ bool LoopVectorizationCostModel::isPredicatedInst(Instruction *I) const {
if (Legal->blockNeedsPredication(I->getParent()))
return true;
- // If we're not folding the tail by masking, predication is unnecessary.
+ // If we're not folding the tail by masking, tail-folding is unnecessary.
if (!foldTailByMasking())
return false;
@@ -2930,7 +2930,7 @@ bool LoopVectorizationCostModel::interleavedAccessCanBeWidened(
blockNeedsPredicationForAnyReason(I->getParent()) && isMaskRequired(I);
bool LoadAccessWithGapsRequiresEpilogMasking =
isa<LoadInst>(I) && Group->requiresScalarEpilogue() &&
- !isScalarEpilogueAllowed();
+ !isEpilogueAllowed();
bool StoreAccessWithGapsRequiresMasking =
isa<StoreInst>(I) && !Group->isFull();
if (!PredicatedAccessRequiresMasking &&
@@ -3467,7 +3467,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
ScalarEvolution *SE = PSE.getSE();
ElementCount TC = getSmallConstantTripCount(SE, TheLoop);
unsigned MaxTC = PSE.getSmallConstantMaxTripCount();
- if (!MaxTC && ScalarEpilogueStatus == CM_ScalarEpilogueAllowed)
+ if (!MaxTC && EpilogueLoweringStatus == CM_EpilogueAllowed)
MaxTC = getMaxTCFromNonZeroRange(PSE, TheLoop);
LLVM_DEBUG(dbgs() << "LV: Found trip count: " << TC << '\n');
if (TC != ElementCount::getFixed(MaxTC))
@@ -3495,25 +3495,23 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
return FixedScalableVFPair::getNone();
}
- switch (ScalarEpilogueStatus) {
- case CM_ScalarEpilogueAllowed:
+ switch (EpilogueLoweringStatus) {
+ case CM_EpilogueAllowed:
return computeFeasibleMaxVF(MaxTC, UserVF, UserIC, false);
- case CM_ScalarEpilogueNotAllowedUsePredicate:
+ case CM_EpilogueNotAllowedFoldTail:
[[fallthrough]];
- case CM_ScalarEpilogueNotNeededUsePredicate:
- LLVM_DEBUG(
- dbgs() << "LV: vector predicate hint/switch found.\n"
- << "LV: Not allowing scalar epilogue, creating predicated "
- << "vector loop.\n");
+ case CM_EpilogueNotNeededFoldTail:
+ LLVM_DEBUG(dbgs() << "LV: tail-folding hint/switch found.\n"
+ << "LV: Not allowing epilogue, creating tail-folded "
+ << "vector loop.\n");
break;
- case CM_ScalarEpilogueNotAllowedLowTripLoop:
+ case CM_EpilogueNotAllowedLowTripLoop:
// fallthrough as a special case of OptForSize
- case CM_ScalarEpilogueNotAllowedOptSize:
- if (ScalarEpilogueStatus == CM_ScalarEpilogueNotAllowedOptSize)
- LLVM_DEBUG(
- dbgs() << "LV: Not allowing scalar epilogue due to -Os/-Oz.\n");
+ case CM_EpilogueNotAllowedOptSize:
+ if (EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize)
+ LLVM_DEBUG(dbgs() << "LV: Not allowing epilogue due to -Os/-Oz.\n");
else
- LLVM_DEBUG(dbgs() << "LV: Not allowing scalar epilogue due to low trip "
+ LLVM_DEBUG(dbgs() << "LV: Not allowing epilogue due to low trip "
<< "count.\n");
// Bail if runtime checks are required, which are not good when optimising
@@ -3594,7 +3592,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// If we have a low-trip-count, and the fixed-width VF is known to divide
// the trip count but the scalable factor does not, use the fixed-width
// factor in preference to allow the generation of a non-predicated loop.
- if (ScalarEpilogueStatus == CM_ScalarEpilogueNotAllowedLowTripLoop &&
+ if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue())) {
LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width so that no tail will "
"remain for any chosen VF.\n");
@@ -3634,15 +3632,15 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
}
// If there was a tail-folding hint/switch, but we can't fold the tail by
- // masking, fallback to a vectorization with a scalar epilogue.
- if (ScalarEpilogueStatus == CM_ScalarEpilogueNotNeededUsePredicate) {
- LLVM_DEBUG(dbgs() << "LV: Cannot fold tail by masking: vectorize with a "
- "scalar epilogue instead.\n");
- ScalarEpilogueStatus = CM_ScalarEpilogueAllowed;
+ // masking, fallback to a vectorization with an epilogue.
+ if (EpilogueLoweringStatus == CM_EpilogueNotNeededFoldTail) {
+ LLVM_DEBUG(dbgs() << "LV: Cannot fold tail by masking: vectorize with an "
+ "epilogue instead.\n");
+ EpilogueLoweringStatus = CM_EpilogueAllowed;
return MaxFactors;
}
- if (ScalarEpilogueStatus == CM_ScalarEpilogueNotAllowedUsePredicate) {
+ if (EpilogueLoweringStatus == CM_EpilogueNotAllowedFoldTail) {
LLVM_DEBUG(dbgs() << "LV: Can't fold tail by masking: don't vectorize\n");
return FixedScalableVFPair::getNone();
}
@@ -4360,7 +4358,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
return nullptr;
}
- if (!CM.isScalarEpilogueAllowed()) {
+ if (!CM.isEpilogueAllowed()) {
LLVM_DEBUG(dbgs() << "LEV: Unable to vectorize epilogue because no "
"epilogue is allowed.\n");
return nullptr;
@@ -4621,10 +4619,10 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
// Only interleave tail-folded loops if wide lane masks are requested, as the
// overhead of multiple instructions to calculate the predicate is likely
- // not beneficial. If a scalar epilogue is not allowed for any other reason,
+ // not beneficial. If an epilogue is not allowed for any other reason,
// do not interleave.
- if (!CM.isScalarEpilogueAllowed() &&
- !(CM.preferPredicatedLoop() && CM.useWideActiveLaneMask()))
+ if (!CM.isEpilogueAllowed() &&
+ !(CM.preferTailFoldedLoop() && CM.useWideActiveLaneMask()))
return 1;
if (any_of(Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis(),
@@ -4736,7 +4734,7 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
auto BestKnownTC =
getSmallBestKnownTC(PSE, OrigLoop,
/*CanUseConstantMax=*/true,
- /*CanExcludeZeroTrips=*/CM.isScalarEpilogueAllowed());
+ /*CanExcludeZeroTrips=*/CM.isEpilogueAllowed());
// For fixed length VFs treat a scalable trip count as unknown.
if (BestKnownTC && (BestKnownTC->isFixed() || VF.isScalable())) {
@@ -4809,7 +4807,7 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
return IC;
}
- // For any scalar loop that either requires runtime checks or predication we
+ // For any scalar loop that either requires runtime checks or tail-folding we
// are better off leaving this to the unroller. Note that if we've already
// vectorized the loop we will have done the runtime check and so interleaving
// won't require further checks.
@@ -5386,7 +5384,7 @@ LoopVectorizationCostModel::getInterleaveGroupCost(Instruction *I,
// Calculate the cost of the whole interleaved group.
bool UseMaskForGaps =
- (Group->requiresScalarEpilogue() && !isScalarEpilogueAllowed()) ||
+ (Group->requiresScalarEpilogue() && !isEpilogueAllowed()) ||
(isa<StoreInst>(I) && !Group->isFull());
InstructionCost Cost = TTI.getInterleavedMemoryOpCost(
InsertPos->getOpcode(), WideVecTy, Group->getFactor(), Indices,
@@ -8256,7 +8254,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlanWithVPRecipes(
// for this VPlan, replace the Recipes widening its memory instructions with a
// single VPInterleaveRecipe at its insertion point.
RUN_VPLAN_PASS(VPlanTransforms::createInterleaveGroups, *Plan,
- InterleaveGroups, RecipeBuilder, CM.isScalarEpilogueAllowed());
+ InterleaveGroups, RecipeBuilder, CM.isEpilogueAllowed());
// Replace VPValues for known constant strides.
RUN_VPLAN_PASS(VPlanTransforms::replaceSymbolicStrides, *Plan, PSE,
@@ -8570,46 +8568,47 @@ void LoopVectorizationPlanner::addMinimumIterationCheck(
OrigLoop->getLoopPredecessor()->getTerminator()->getDebugLoc(), PSE);
}
-// Determine how to lower the scalar epilogue, which depends on 1) optimising
-// for minimum code-size, 2) predicate compiler options, 3) loop hints forcing
-// predication, and 4) a TTI hook that analyses whether the loop is suitable
-// for predication.
-static ScalarEpilogueLowering getScalarEpilogueLowering(
- Function *F, Loop *L, LoopVectorizeHints &Hints, bool OptForSize,
- TargetTransformInfo *TTI, TargetLibraryInfo *TLI,
- LoopVectorizationLegality &LVL, InterleavedAccessInfo *IAI) {
+// Determine how to lower the epilogue, which depends on 1) optimising
+// for minimum code-size, 2) tail-folding compiler options, 3) loop
+// hints forcing tail-folding, and 4) a TTI hook that analyses whether the loop
+// is suitable for tail-folding.
+static EpilogueLowering
+getEpilogueLowering(Function *F, Loop *L, LoopVectorizeHints &Hints,
+ bool OptForSize, TargetTransformInfo *TTI,
+ TargetLibraryInfo *TLI, LoopVectorizationLegality &LVL,
+ InterleavedAccessInfo *IAI) {
// 1) OptSize takes precedence over all other options, i.e. if this is set,
- // don't look at hints or options, and don't request a scalar epilogue.
+ // don't look at hints or options, and don't request an epilogue.
if (F->hasOptSize() ||
(OptForSize && Hints.getForce() != LoopVectorizeHints::FK_Enabled))
- return CM_ScalarEpilogueNotAllowedOptSize;
+ return CM_EpilogueNotAllowedOptSize;
// 2) If set, obey the directives
if (PreferPredicateOverEpilogue.getNumOccurrences()) {
switch (PreferPredicateOverEpilogue) {
case PreferPredicateTy::ScalarEpilogue:
- return CM_ScalarEpilogueAllowed;
+ return CM_EpilogueAllowed;
case PreferPredicateTy::PredicateElseScalarEpilogue:
- return CM_ScalarEpilogueNotNeededUsePredicate;
+ return CM_EpilogueNotNeededFoldTail;
case PreferPredicateTy::PredicateOrDontVectorize:
- return CM_ScalarEpilogueNotAllowedUsePredicate;
+ return CM_EpilogueNotAllowedFoldTail;
};
}
// 3) If set, obey the hints
switch (Hints.getPredicate()) {
case LoopVectorizeHints::FK_Enabled:
- return CM_ScalarEpilogueNotNeededUsePredicate;
+ return CM_EpilogueNotNeededFoldTail;
case LoopVectorizeHints::FK_Disabled:
- return CM_ScalarEpilogueAllowed;
+ return CM_EpilogueAllowed;
};
- // 4) if the TTI hook indicates this is profitable, request predication.
+ // 4) if the TTI hook indicates this is profitable, request tail-folding.
TailFoldingInfo TFI(TLI, &LVL, IAI);
- if (TTI->preferPredicateOverEpilogue(&TFI))
- return CM_ScalarEpilogueNotNeededUsePredicate;
+ if (TTI->preferTailFoldingOverEpilogue(&TFI))
+ return CM_EpilogueNotNeededFoldTail;
- return CM_ScalarEpilogueAllowed;
+ return CM_EpilogueAllowed;
}
// Process the loop in the VPlan-native vectorization path. This path builds
@@ -8632,8 +8631,8 @@ static bool processLoopInVPlanNativePath(
Function *F = L->getHeader()->getParent();
InterleavedAccessInfo IAI(PSE, L, DT, LI, LVL->getLAI());
- ScalarEpilogueLowering SEL =
- getScalarEpilogueLowering(F, L, Hints, OptForSize, TTI, TLI, *LVL, &IAI);
+ EpilogueLowering SEL =
+ getEpilogueLowering(F, L, Hints, OptForSize, TTI, TLI, *LVL, &IAI);
LoopVectorizationCostModel CM(SEL, L, PSE, LI, LVL, *TTI, TLI, DB, AC, ORE,
GetBFI, F, &Hints, IAI, OptForSize);
@@ -8761,7 +8760,7 @@ static bool isOutsideLoopWorkProfitable(GeneratedRTChecks &Checks,
VectorizationFactor &VF, Loop *L,
PredicatedScalarEvolution &PSE,
VPCostContext &CostCtx, VPlan &Plan,
- ScalarEpilogueLowering SEL,
+ EpilogueLowering SEL,
std::optional<unsigned> VScale) {
InstructionCost RtC = Checks.getCost();
if (!RtC.isValid())
@@ -8842,7 +8841,7 @@ static bool isOutsideLoopWorkProfitable(GeneratedRTChecks &Checks,
// epilogue is allowed, choose the next closest multiple of VF. This should
// partly compensate for ignoring the epilogue cost.
uint64_t MinTC = std::max(MinTC1, MinTC2);
- if (SEL == CM_ScalarEpilogueAllowed)
+ if (SEL == CM_EpilogueAllowed)
MinTC = alignTo(MinTC, IntVF);
VF.MinProfitableTripCount = ElementCount::getFixed(MinTC);
@@ -9415,8 +9414,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
// Check the function attributes and profiles to find out if this function
// should be optimized for size.
- ScalarEpilogueLowering SEL =
- getScalarEpilogueLowering(F, L, Hints, OptForSize, TTI, TLI, LVL, &IAI);
+ EpilogueLowering SEL =
+ getEpilogueLowering(F, L, Hints, OptForSize, TTI, TLI, LVL, &IAI);
// Check the loop for a trip count threshold: vectorize loops with a tiny trip
// count by optimizing for size, to minimize overheads.
@@ -9430,14 +9429,14 @@ bool LoopVectorizePass::processLoop(Loop *L) {
LLVM_DEBUG(dbgs() << " But vectorizing was explicitly forced.\n");
else {
LLVM_DEBUG(dbgs() << "\n");
- // Predicate tail-folded loops are efficient even when the loop
+ // Tail-folded loops are efficient even when the loop
// iteration count is low. However, setting the epilogue policy to
- // `CM_ScalarEpilogueNotAllowedLowTripLoop` prevents vectorizing loops
+ // `CM_EpilogueNotAllowedLowTripLoop` prevents vectorizing loops
// with runtime checks. It's more effective to let
// `isOutsideLoopWorkProfitable` determine if vectorization is
// beneficial for the loop.
- if (SEL != CM_ScalarEpilogueNotNeededUsePredicate)
- SEL = CM_ScalarEpilogueNotAllowedLowTripLoop;
+ if (SEL != CM_EpilogueNotNeededFoldTail)
+ SEL = CM_EpilogueNotAllowedLowTripLoop;
}
}
More information about the llvm-commits
mailing list