[llvm-branch-commits] [llvm] Patch 5: [LV] create predicated(tail-folded) vplans. (PR #207815)
via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Wed Jul 8 06:54:13 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: Hassnaa Hamdi (hassnaaHamdi)
<details>
<summary>Changes</summary>
---
Full diff: https://github.com/llvm/llvm-project/pull/207815.diff
2 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorize.cpp (+85-46)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-costs.ll (+4-4)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 3590353932540..8ed13784f6b02 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3147,6 +3147,8 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
using RecipeVFPair = std::pair<VPRecipeBase *, ElementCount>;
SmallVector<RecipeVFPair> InvalidCosts;
for (const auto &Plan : VPlans) {
+ if (!Plan->isCompatibleWithTF(CM.foldTailByMasking()))
+ continue;
for (ElementCount VF : Plan->vectorFactors()) {
// The VPlan-based cost model is designed for computing vector cost.
// Querying VPlan-based cost model with a scarlar VF will cause some
@@ -5521,9 +5523,9 @@ void LoopVectorizationPlanner::plan(
// for later use by the cost model.
Config.computeMinimalBitwidths();
- SmallVector<LoopVectorizationCostModel *, 2> EnabledCMs;
- EnabledCMs.push_back(&CM);
-
+ SmallVector<std::pair<LoopVectorizationCostModel *, VPlanPtr>, 2> EnabledCMs;
+ EnabledCMs.push_back(std::make_pair(&CM, std::move(VPlan1)));
+ FixedScalableVFPair MaxFactorsTF;
// Make sure firstly that the epilogue of main vector loop is allowed, then
// check if the tail-folded epilogue feature is enabled.
if (CM.EpilogueLoweringStatus == CM_EpilogueAllowed &&
@@ -5536,9 +5538,11 @@ void LoopVectorizationPlanner::plan(
// After making sure that we can get valid results of computeMaxVF, make
// sure that tail-folding for the epilogue loop still valid.
- if (EpilogueTailFoldingCM->computeMaxVF(UserVF, UserIC) &&
- EpilogueTailFoldingCM->foldTailByMasking()) {
- EnabledCMs.push_back(&*EpilogueTailFoldingCM);
+ MaxFactorsTF = EpilogueTailFoldingCM->computeMaxVF(UserVF, UserIC);
+ if (MaxFactorsTF && EpilogueTailFoldingCM->foldTailByMasking()) {
+ VPlanPtr PredicatedVPlan1 = tryToBuildVPlan1(*EpilogueTailFoldingCM);
+ EnabledCMs.push_back(
+ std::make_pair(&*EpilogueTailFoldingCM, std::move(PredicatedVPlan1)));
LLVM_DEBUG(dbgs() << "LV: CM instances: " << EnabledCMs.size() << "\n");
}
}
@@ -5546,22 +5550,23 @@ void LoopVectorizationPlanner::plan(
// Invalidate interleave groups if all blocks of loop will be predicated.
if (!useMaskedInterleavedAccesses(TTI)) {
for (auto &CurrentCM : EnabledCMs) {
- if (CurrentCM->blockNeedsPredicationForAnyReason(OrigLoop->getHeader())) {
+ if (CurrentCM.first->blockNeedsPredicationForAnyReason(
+ OrigLoop->getHeader())) {
LLVM_DEBUG(dbgs() << "LV: Invalidate all interleaved groups due to "
<< "fold-tail by masking which requires "
"masked-interleaved support.\n");
- if (CurrentCM->InterleaveInfo.invalidateGroups()) {
+ if (CurrentCM.first->InterleaveInfo.invalidateGroups()) {
// Invalidating interleave groups also requires invalidating all
// decisions based on them, which includes widening decisions and
// uniform and scalar values.
- CurrentCM->invalidateCostModelingDecisions();
+ CurrentCM.first->invalidateCostModelingDecisions();
}
}
}
}
for (auto &CurrentCM : EnabledCMs) {
- if (CurrentCM->foldTailByMasking())
+ if (CurrentCM.first->foldTailByMasking())
Legal->prepareToFoldTailByMasking();
}
@@ -5577,16 +5582,23 @@ void LoopVectorizationPlanner::plan(
"VF needs to be a power of two");
// Collect the instructions (and their associated costs) that will be more
// profitable to scalarize.
- EnabledCMs.front()->collectNonVectorizedAndSetWideningDecisions(UserVF);
+ auto PrepareCostAndPlan = [&](LoopVectorizationCostModel &CurrentCM,
+ VPlanPtr &InitialVPlan, ElementCount &VF) {
+ CurrentCM.collectNonVectorizedAndSetWideningDecisions(VF);
+ buildVPlans(*InitialVPlan, VF, VF, CurrentCM);
+ };
+
ElementCount EpilogueUserVF =
ElementCount::getFixed(EpilogueVectorizationForceVF);
if (EpilogueUserVF.isVector() &&
ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
- EnabledCMs.back()->collectNonVectorizedAndSetWideningDecisions(
- EpilogueUserVF);
- buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, CM);
+ // TODO: Relax isKnownLT to isKnownLE and let other epilogue-related
+ // functions to decide if the EpilogueUserVF is accepted or not.
+ PrepareCostAndPlan(*EnabledCMs.back().first, EnabledCMs.back().second,
+ EpilogueUserVF);
}
- buildVPlans(*VPlan1, UserVF, UserVF, CM);
+ PrepareCostAndPlan(*EnabledCMs.front().first, EnabledCMs.front().second,
+ UserVF);
if (!VPlans.empty() && VPlans.back()->getSingleVF() == UserVF) {
// For scalar VF, skip VPlan cost check as VPlan cost is designed for
// vector VFs only.
@@ -5612,15 +5624,29 @@ void LoopVectorizationPlanner::plan(
ElementCount::isKnownLE(VF, MaxFactors.ScalableVF); VF *= 2)
VFCandidates.push_back(VF);
- for (const auto &VF : VFCandidates) {
+ for (const auto &VF : VFCandidates)
// Collect Uniform and Scalar instructions after vectorization with VF.
- for (auto &CurrentCM : EnabledCMs) {
- CurrentCM->collectNonVectorizedAndSetWideningDecisions(VF);
- }
- }
+ EnabledCMs.front().first->collectNonVectorizedAndSetWideningDecisions(VF);
+
+ buildVPlans(*EnabledCMs.front().second, ElementCount::getFixed(1),
+ MaxFactors.FixedVF, *EnabledCMs.front().first);
+ buildVPlans(*EnabledCMs.front().second, ElementCount::getScalable(1),
+ MaxFactors.ScalableVF, *EnabledCMs.front().first);
- buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF, CM);
- buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF, CM);
+ if (EnabledCMs.size() > 1) {
+ for (auto VF = ElementCount::getFixed(1);
+ ElementCount::isKnownLE(VF, MaxFactorsTF.FixedVF); VF *= 2)
+ EnabledCMs.back().first->collectNonVectorizedAndSetWideningDecisions(VF);
+
+ for (auto VF = ElementCount::getScalable(1);
+ ElementCount::isKnownLE(VF, MaxFactorsTF.ScalableVF); VF *= 2)
+ EnabledCMs.back().first->collectNonVectorizedAndSetWideningDecisions(VF);
+
+ buildVPlans(*EnabledCMs.back().second, ElementCount::getFixed(1),
+ MaxFactorsTF.FixedVF, *EnabledCMs.back().first);
+ buildVPlans(*EnabledCMs.back().second, ElementCount::getScalable(1),
+ MaxFactorsTF.ScalableVF, *EnabledCMs.back().first);
+ }
LLVM_DEBUG(printPlans(dbgs()));
}
@@ -5911,33 +5937,32 @@ std::pair<VectorizationFactor, VPlan *> LoopVectorizationPlanner::computeBestVF(
continue;
}
- assert(P->isCompatibleWithTF(CM.foldTailByMasking()) &&
- "All vplans must be compatible with the CM");
- InstructionCost Cost =
- cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr, CM);
- VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
-
- if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
- BestFactor = CurrentFactor;
- PlanForBestVF = P.get();
- }
+ if (P->isCompatibleWithTF(CM.foldTailByMasking())) {
+ InstructionCost Cost =
+ cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr, CM);
+ VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
- // If profitable add it to ProfitableVF list.
- if (isMoreProfitable(CurrentFactor, ScalarFactor, P->hasScalarTail()))
- ProfitableVFs.push_back(CurrentFactor);
+ if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
+ BestFactor = CurrentFactor;
+ PlanForBestVF = P.get();
+ }
- if (EpilogueTailFoldingCM) {
- // Get the costs for the EpilogueTailFoldingCM:
+ // If profitable add it to ProfitableVF list.
+ if (isMoreProfitable(CurrentFactor, ScalarFactor, P->hasScalarTail()))
+ ProfitableVFs.push_back(CurrentFactor);
+ }
+ // Get the costs for the EpilogueTailFoldingCM:
+ if (EpilogueTailFoldingCM &&
+ P->isCompatibleWithTF(EpilogueTailFoldingCM->foldTailByMasking())) {
LLVM_DEBUG(dbgs() << "LV: Predicated CM, calculate costs for VF: " << VF
<< "\n");
- cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr,
- *EpilogueTailFoldingCM);
- // TODO: that cost is not accurate right now as it includes costs for
- // unpredicated vplans instead of predicated ones. That should be fixed
- // in future work.
-
- // TODO: consider the VF as a profitable one when we support predicated
- // vplans.
+ InstructionCost Cost =
+ cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr,
+ *EpilogueTailFoldingCM);
+ VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
+ // If profitable add it to ProfitableVF list.
+ if (isMoreProfitable(CurrentFactor, ScalarFactor, P->hasScalarTail()))
+ ProfitableVFs.push_back(CurrentFactor);
}
}
}
@@ -7238,6 +7263,7 @@ getEpilogueLowering(Function *F, Loop *L, LoopVectorizeHints &Hints,
/// otherwise CM_EpilogueAllowed.
static EpilogueLowering
getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
+ LoopVectorizationLegality &LVL,
OptimizationRemarkEmitter *ORE) {
// Epilogue TF is only enabled when explicitly requested via command line.
if (!EpilogueTailFoldingPolicy.getNumOccurrences() ||
@@ -7267,6 +7293,19 @@ getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
return CM_EpilogueAllowed;
}
+ bool HasReductions = !LVL.getReductionVars().empty();
+ bool HasSelectCmpReductions =
+ HasReductions &&
+ any_of(LVL.getReductionVars(), [&](auto &Reduction) -> bool {
+ const RecurrenceDescriptor &RdxDesc = Reduction.second;
+ RecurKind RK = RdxDesc.getRecurrenceKind();
+ return RecurrenceDescriptor::isAnyOfRecurrenceKind(RK) ||
+ RecurrenceDescriptor::isFindIVRecurrenceKind(RK) ||
+ RecurrenceDescriptor::isMinMaxRecurrenceKind(RK);
+ });
+ if (HasSelectCmpReductions)
+ return CM_EpilogueAllowed;
+
// We can apply tail-folding on the vectorized epilogue loop.
return CM_EpilogueNotNeededFoldTail;
}
@@ -8093,7 +8132,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
std::optional<InterleavedAccessInfo> TailFoldingCMIAI;
std::optional<LoopVectorizationCostModel> EpilogueTailFoldingCM;
EpilogueLowering EpilogueTailLoweringStatus =
- getEpilogueTailLowering(CM, L, ORE);
+ getEpilogueTailLowering(CM, L, LVL, ORE);
if (EpilogueTailLoweringStatus ==
EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-costs.ll
index c5d6e72f61c68..c578e230daa7a 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-costs.ll
@@ -21,13 +21,13 @@ target triple = "aarch64"
; EPILOGUE-TF-CM: LV: can fold tail by masking.
; EPILOGUE-TF-CM: LV: CM instances: 2
; EPILOGUE-TF-CM: LV: Predicated CM, calculate costs for VF: 2
-; EPILOGUE-TF-CM: Cost for VF 2: 6
+; EPILOGUE-TF-CM: Cost for VF 2: 7
; EPILOGUE-TF-CM: LV: Predicated CM, calculate costs for VF: 4
-; EPILOGUE-TF-CM: Cost for VF 4: 11
+; EPILOGUE-TF-CM: Cost for VF 4: 13
; EPILOGUE-TF-CM: LV: Predicated CM, calculate costs for VF: 8
-; EPILOGUE-TF-CM: Cost for VF 8: 21
+; EPILOGUE-TF-CM: Cost for VF 8: 25
; EPILOGUE-TF-CM: LV: Predicated CM, calculate costs for VF: 16
-; EPILOGUE-TF-CM: Cost for VF 16: 41
+; EPILOGUE-TF-CM: Cost for VF 16: 49
;
define void @test_epilogue_tf(ptr %A, i64 %n) {
; CHECK-LABEL: @test_epilogue_tf
``````````
</details>
https://github.com/llvm/llvm-project/pull/207815
More information about the llvm-branch-commits
mailing list