[llvm] d64dd5a - [LV] Factor out VF-independent code from cost model (NFC). (#192426)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Apr 23 04:09:03 PDT 2026
Author: Florian Hahn
Date: 2026-04-23T12:08:57+01:00
New Revision: d64dd5a2afea7c99a6c54827596e5f2172fe3342
URL: https://github.com/llvm/llvm-project/commit/d64dd5a2afea7c99a6c54827596e5f2172fe3342
DIFF: https://github.com/llvm/llvm-project/commit/d64dd5a2afea7c99a6c54827596e5f2172fe3342.diff
LOG: [LV] Factor out VF-independent code from cost model (NFC). (#192426)
LoopVectorizationCostModel currently contains state and helpers to deal
with both VF-dependent and independent aspects of cost modeling.
Some of the state it tracks is per-VF, other state is shared across all
VFs.
This patch tries to factor out most of the VF-independent state +
functions to a new class, to try to more clearly separate those aspects,
and make it easier to reason about what decisions are independent of
VF-specific costs.
PR: https://github.com/llvm/llvm-project/pull/192426
Added:
llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
Modified:
llvm/lib/Transforms/Vectorize/CMakeLists.txt
llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/CMakeLists.txt b/llvm/lib/Transforms/Vectorize/CMakeLists.txt
index db0f650c19509..984b0d97c60c8 100644
--- a/llvm/lib/Transforms/Vectorize/CMakeLists.txt
+++ b/llvm/lib/Transforms/Vectorize/CMakeLists.txt
@@ -2,6 +2,7 @@ add_llvm_component_library(LLVMVectorize
LoadStoreVectorizer.cpp
LoopIdiomVectorize.cpp
LoopVectorizationLegality.cpp
+ LoopVectorizationPlanner.cpp
LoopVectorize.cpp
SandboxVectorizer/DependencyGraph.cpp
SandboxVectorizer/InstrMaps.cpp
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
new file mode 100644
index 0000000000000..0e847f4767a8b
--- /dev/null
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -0,0 +1,616 @@
+//===- LoopVectorizationPlanner.cpp - VF selection and planning -----------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+///
+/// \file
+/// This file implements VFSelectionContext methods for loop vectorization
+/// VF selection, independent of cost-modeling decisions.
+///
+//===----------------------------------------------------------------------===//
+
+#include "LoopVectorizationPlanner.h"
+#include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/OptimizationRemarkEmitter.h"
+#include "llvm/Analysis/ScalarEvolution.h"
+#include "llvm/IR/DiagnosticInfo.h"
+#include "llvm/Support/CommandLine.h"
+#include "llvm/Support/Debug.h"
+#include "llvm/Support/MathExtras.h"
+#include "llvm/Transforms/Vectorize/LoopVectorizationLegality.h"
+#include "llvm/Transforms/Vectorize/LoopVectorize.h"
+
+using namespace llvm;
+
+#define DEBUG_TYPE "loop-vectorize"
+
+static cl::opt<bool> MaximizeBandwidth(
+ "vectorizer-maximize-bandwidth", cl::init(false), cl::Hidden,
+ cl::desc("Maximize bandwidth when selecting vectorization factor which "
+ "will be determined by the smallest type in loop."));
+
+static cl::opt<bool> UseWiderVFIfCallVariantsPresent(
+ "vectorizer-maximize-bandwidth-for-vector-calls", cl::init(true),
+ cl::Hidden,
+ cl::desc("Try wider VFs if they enable the use of vector variants"));
+
+static cl::opt<bool> ConsiderRegPressure(
+ "vectorizer-consider-reg-pressure", cl::init(false), cl::Hidden,
+ cl::desc("Discard VFs if their register pressure is too high."));
+
+static cl::opt<bool> ForceTargetSupportsScalableVectors(
+ "force-target-supports-scalable-vectors", cl::init(false), cl::Hidden,
+ cl::desc(
+ "Pretend that scalable vectors are supported, even if the target does "
+ "not support them. This flag should only be used for testing."));
+
+cl::opt<bool> llvm::PreferInLoopReductions(
+ "prefer-inloop-reductions", cl::init(false), cl::Hidden,
+ cl::desc("Prefer in-loop vector reductions, "
+ "overriding the targets preference."));
+
+/// Note: This currently only applies to `llvm.masked.load` and
+/// `llvm.masked.store`. TODO: Extend this to cover other operations as needed.
+static cl::opt<bool> ForceTargetSupportsMaskedMemoryOps(
+ "force-target-supports-masked-memory-ops", cl::init(false), cl::Hidden,
+ cl::desc("Assume the target supports masked memory operations (used for "
+ "testing)."));
+
+bool VFSelectionContext::isLegalMaskedStore(Type *DataType, Value *Ptr,
+ Align Alignment,
+ unsigned AddressSpace) const {
+ return Legal->isConsecutivePtr(DataType, Ptr) &&
+ (ForceTargetSupportsMaskedMemoryOps ||
+ TTI.isLegalMaskedStore(DataType, Alignment, AddressSpace));
+}
+
+bool VFSelectionContext::isLegalMaskedLoad(Type *DataType, Value *Ptr,
+ Align Alignment,
+ unsigned AddressSpace) const {
+ return Legal->isConsecutivePtr(DataType, Ptr) &&
+ (ForceTargetSupportsMaskedMemoryOps ||
+ TTI.isLegalMaskedLoad(DataType, Alignment, AddressSpace));
+}
+
+bool VFSelectionContext::isLegalGatherOrScatter(Value *V,
+ ElementCount VF) const {
+ bool LI = isa<LoadInst>(V);
+ bool SI = isa<StoreInst>(V);
+ if (!LI && !SI)
+ return false;
+ auto *Ty = getLoadStoreType(V);
+ Align Align = getLoadStoreAlignment(V);
+ if (VF.isVector())
+ Ty = VectorType::get(Ty, VF);
+ return (LI && TTI.isLegalMaskedGather(Ty, Align)) ||
+ (SI && TTI.isLegalMaskedScatter(Ty, Align));
+}
+
+bool VFSelectionContext::supportsScalableVectors() const {
+ return TTI.supportsScalableVectors() || ForceTargetSupportsScalableVectors;
+}
+
+bool VFSelectionContext::useMaxBandwidth(bool IsScalable) const {
+ TargetTransformInfo::RegisterKind RegKind =
+ IsScalable ? TargetTransformInfo::RGK_ScalableVector
+ : TargetTransformInfo::RGK_FixedWidthVector;
+ return MaximizeBandwidth || (MaximizeBandwidth.getNumOccurrences() == 0 &&
+ (TTI.shouldMaximizeVectorBandwidth(RegKind) ||
+ (UseWiderVFIfCallVariantsPresent &&
+ Legal->hasVectorCallVariants())));
+}
+
+bool VFSelectionContext::shouldConsiderRegPressureForVF(ElementCount VF) const {
+ if (ConsiderRegPressure.getNumOccurrences())
+ return ConsiderRegPressure;
+
+ // TODO: We should eventually consider register pressure for all targets. The
+ // TTI hook is temporary whilst target-specific issues are being fixed.
+ if (TTI.shouldConsiderVectorizationRegPressure())
+ return true;
+
+ if (!useMaxBandwidth(VF.isScalable()))
+ return false;
+ // Only calculate register pressure for VFs enabled by MaxBandwidth.
+ return ElementCount::isKnownGT(
+ VF, VF.isScalable() ? MaxPermissibleVFWithoutMaxBW.ScalableVF
+ : MaxPermissibleVFWithoutMaxBW.FixedVF);
+}
+
+ElementCount VFSelectionContext::clampVFByMaxTripCount(
+ ElementCount VF, unsigned MaxTripCount, unsigned UserIC,
+ bool FoldTailByMasking, bool RequiresScalarEpilogue) const {
+ unsigned EstimatedVF = VF.getKnownMinValue();
+ if (VF.isScalable() && F.hasFnAttribute(Attribute::VScaleRange)) {
+ auto Attr = F.getFnAttribute(Attribute::VScaleRange);
+ auto Min = Attr.getVScaleRangeMin();
+ EstimatedVF *= Min;
+ }
+
+ // When a scalar epilogue is required, at least one iteration of the scalar
+ // loop has to execute. Adjust MaxTripCount accordingly to avoid picking a
+ // max VF that results in a dead vector loop.
+ if (MaxTripCount > 0 && RequiresScalarEpilogue)
+ MaxTripCount -= 1;
+
+ // When the user specifies an interleave count, we need to ensure that
+ // VF * UserIC <= MaxTripCount to avoid a dead vector loop.
+ unsigned IC = UserIC > 0 ? UserIC : 1;
+ unsigned EstimatedVFTimesIC = EstimatedVF * IC;
+
+ if (MaxTripCount && MaxTripCount <= EstimatedVFTimesIC &&
+ (!FoldTailByMasking || isPowerOf2_32(MaxTripCount))) {
+ // If upper bound loop trip count (TC) is known at compile time there is no
+ // point in choosing VF greater than TC / IC (as done in the loop below).
+ // Select maximum power of two which doesn't exceed TC / IC. If VF is
+ // scalable, we only fall back on a fixed VF when the TC is less than or
+ // equal to the known number of lanes.
+ auto ClampedUpperTripCount = llvm::bit_floor(MaxTripCount / IC);
+ if (ClampedUpperTripCount == 0)
+ ClampedUpperTripCount = 1;
+ LLVM_DEBUG(dbgs() << "LV: Clamping the MaxVF to maximum power of two not "
+ "exceeding the constant trip count"
+ << (UserIC > 0 ? " divided by UserIC" : "") << ": "
+ << ClampedUpperTripCount << "\n");
+ return ElementCount::get(ClampedUpperTripCount,
+ FoldTailByMasking ? VF.isScalable() : false);
+ }
+ return VF;
+}
+
+ElementCount VFSelectionContext::getMaximizedVFForTarget(
+ unsigned MaxTripCount, unsigned SmallestType, unsigned WidestType,
+ ElementCount MaxSafeVF, unsigned UserIC, bool FoldTailByMasking,
+ bool RequiresScalarEpilogue) {
+ bool ComputeScalableMaxVF = MaxSafeVF.isScalable();
+ const TypeSize WidestRegister = TTI.getRegisterBitWidth(
+ ComputeScalableMaxVF ? TargetTransformInfo::RGK_ScalableVector
+ : TargetTransformInfo::RGK_FixedWidthVector);
+
+ // Convenience function to return the minimum of two ElementCounts.
+ auto MinVF = [](const ElementCount &LHS, const ElementCount &RHS) {
+ assert((LHS.isScalable() == RHS.isScalable()) &&
+ "Scalable flags must match");
+ return ElementCount::isKnownLT(LHS, RHS) ? LHS : RHS;
+ };
+
+ // Ensure MaxVF is a power of 2; the dependence distance bound may not be.
+ // Note that both WidestRegister and WidestType may not be a powers of 2.
+ auto MaxVectorElementCount = ElementCount::get(
+ llvm::bit_floor(WidestRegister.getKnownMinValue() / WidestType),
+ ComputeScalableMaxVF);
+ MaxVectorElementCount = MinVF(MaxVectorElementCount, MaxSafeVF);
+ LLVM_DEBUG(dbgs() << "LV: The Widest register safe to use is: "
+ << (MaxVectorElementCount * WidestType) << " bits.\n");
+
+ if (!MaxVectorElementCount) {
+ LLVM_DEBUG(dbgs() << "LV: The target has no "
+ << (ComputeScalableMaxVF ? "scalable" : "fixed")
+ << " vector registers.\n");
+ return ElementCount::getFixed(1);
+ }
+
+ ElementCount MaxVF =
+ clampVFByMaxTripCount(MaxVectorElementCount, MaxTripCount, UserIC,
+ FoldTailByMasking, RequiresScalarEpilogue);
+ // If the MaxVF was already clamped, there's no point in trying to pick a
+ // larger one.
+ if (MaxVF != MaxVectorElementCount)
+ return MaxVF;
+
+ if (MaxVF.isScalable())
+ MaxPermissibleVFWithoutMaxBW.ScalableVF = MaxVF;
+ else
+ MaxPermissibleVFWithoutMaxBW.FixedVF = MaxVF;
+
+ if (useMaxBandwidth(ComputeScalableMaxVF)) {
+ auto MaxVectorElementCountMaxBW = ElementCount::get(
+ llvm::bit_floor(WidestRegister.getKnownMinValue() / SmallestType),
+ ComputeScalableMaxVF);
+ MaxVF = MinVF(MaxVectorElementCountMaxBW, MaxSafeVF);
+
+ if (ElementCount MinVF =
+ TTI.getMinimumVF(SmallestType, ComputeScalableMaxVF)) {
+ if (ElementCount::isKnownLT(MaxVF, MinVF)) {
+ LLVM_DEBUG(dbgs() << "LV: Overriding calculated MaxVF(" << MaxVF
+ << ") with target's minimum: " << MinVF << '\n');
+ MaxVF = MinVF;
+ }
+ }
+
+ MaxVF = clampVFByMaxTripCount(MaxVF, MaxTripCount, UserIC,
+ FoldTailByMasking, RequiresScalarEpilogue);
+ }
+ return MaxVF;
+}
+
+std::optional<unsigned> llvm::getMaxVScale(const Function &F,
+ const TargetTransformInfo &TTI) {
+ if (std::optional<unsigned> MaxVScale = TTI.getMaxVScale())
+ return MaxVScale;
+
+ if (F.hasFnAttribute(Attribute::VScaleRange))
+ return F.getFnAttribute(Attribute::VScaleRange).getVScaleRangeMax();
+
+ return std::nullopt;
+}
+
+bool VFSelectionContext::isScalableVectorizationAllowed() {
+ if (IsScalableVectorizationAllowed)
+ return *IsScalableVectorizationAllowed;
+
+ IsScalableVectorizationAllowed = false;
+ if (!supportsScalableVectors())
+ return false;
+
+ if (Hints->isScalableVectorizationDisabled()) {
+ reportVectorizationInfo("Scalable vectorization is explicitly disabled",
+ "ScalableVectorizationDisabled", ORE, TheLoop);
+ return false;
+ }
+
+ LLVM_DEBUG(dbgs() << "LV: Scalable vectorization is available\n");
+
+ auto MaxScalableVF = ElementCount::getScalable(
+ std::numeric_limits<ElementCount::ScalarTy>::max());
+
+ // Test that the loop-vectorizer can legalize all operations for this MaxVF.
+ // FIXME: While for scalable vectors this is currently sufficient, this should
+ // be replaced by a more detailed mechanism that filters out specific VFs,
+ // instead of invalidating vectorization for a whole set of VFs based on the
+ // MaxVF.
+
+ // Disable scalable vectorization if the loop contains unsupported reductions.
+ if (!all_of(Legal->getReductionVars(), [&](const auto &Reduction) -> bool {
+ return TTI.isLegalToVectorizeReduction(Reduction.second, MaxScalableVF);
+ })) {
+ reportVectorizationInfo(
+ "Scalable vectorization not supported for the reduction "
+ "operations found in this loop.",
+ "ScalableVFUnfeasible", ORE, TheLoop);
+ return false;
+ }
+
+ // Disable scalable vectorization if the loop contains any instructions
+ // with element types not supported for scalable vectors.
+ if (any_of(ElementTypesInLoop, [&](Type *Ty) {
+ return !Ty->isVoidTy() && !TTI.isElementTypeLegalForScalableVector(Ty);
+ })) {
+ reportVectorizationInfo("Scalable vectorization is not supported "
+ "for all element types found in this loop.",
+ "ScalableVFUnfeasible", ORE, TheLoop);
+ return false;
+ }
+
+ if (!Legal->isSafeForAnyVectorWidth() && !getMaxVScale(F, TTI)) {
+ reportVectorizationInfo("The target does not provide maximum vscale value "
+ "for safe distance analysis.",
+ "ScalableVFUnfeasible", ORE, TheLoop);
+ return false;
+ }
+
+ IsScalableVectorizationAllowed = true;
+ return true;
+}
+
+ElementCount
+VFSelectionContext::getMaxLegalScalableVF(unsigned MaxSafeElements) {
+ if (!isScalableVectorizationAllowed())
+ return ElementCount::getScalable(0);
+
+ auto MaxScalableVF = ElementCount::getScalable(
+ std::numeric_limits<ElementCount::ScalarTy>::max());
+ if (Legal->isSafeForAnyVectorWidth())
+ return MaxScalableVF;
+
+ std::optional<unsigned> MaxVScale = getMaxVScale(F, TTI);
+ // Limit MaxScalableVF by the maximum safe dependence distance.
+ MaxScalableVF = ElementCount::getScalable(MaxSafeElements / *MaxVScale);
+
+ if (!MaxScalableVF)
+ reportVectorizationInfo(
+ "Max legal vector width too small, scalable vectorization "
+ "unfeasible.",
+ "ScalableVFUnfeasible", ORE, TheLoop);
+
+ return MaxScalableVF;
+}
+
+FixedScalableVFPair VFSelectionContext::computeFeasibleMaxVF(
+ unsigned MaxTripCount, ElementCount UserVF, unsigned UserIC,
+ bool FoldTailByMasking, bool RequiresScalarEpilogue) {
+ auto [SmallestType, WidestType] = getSmallestAndWidestTypes();
+
+ // Get the maximum safe dependence distance in bits computed by LAA.
+ // It is computed by MaxVF * sizeOf(type) * 8, where type is taken from
+ // the memory accesses that is most restrictive (involved in the smallest
+ // dependence distance).
+ unsigned MaxSafeElementsPowerOf2 =
+ llvm::bit_floor(Legal->getMaxSafeVectorWidthInBits() / WidestType);
+ if (!Legal->isSafeForAnyStoreLoadForwardDistances()) {
+ unsigned SLDist = Legal->getMaxStoreLoadForwardSafeDistanceInBits();
+ MaxSafeElementsPowerOf2 =
+ std::min(MaxSafeElementsPowerOf2, SLDist / WidestType);
+ }
+
+ auto MaxSafeFixedVF = ElementCount::getFixed(MaxSafeElementsPowerOf2);
+ auto MaxSafeScalableVF = getMaxLegalScalableVF(MaxSafeElementsPowerOf2);
+
+ if (!Legal->isSafeForAnyVectorWidth())
+ MaxSafeElements = MaxSafeElementsPowerOf2;
+
+ LLVM_DEBUG(dbgs() << "LV: The max safe fixed VF is: " << MaxSafeFixedVF
+ << ".\n");
+ LLVM_DEBUG(dbgs() << "LV: The max safe scalable VF is: " << MaxSafeScalableVF
+ << ".\n");
+
+ // First analyze the UserVF, fall back if the UserVF should be ignored.
+ if (UserVF) {
+ auto MaxSafeUserVF =
+ UserVF.isScalable() ? MaxSafeScalableVF : MaxSafeFixedVF;
+
+ if (ElementCount::isKnownLE(UserVF, MaxSafeUserVF)) {
+ // If `VF=vscale x N` is safe, then so is `VF=N`
+ if (UserVF.isScalable())
+ return FixedScalableVFPair(
+ ElementCount::getFixed(UserVF.getKnownMinValue()), UserVF);
+
+ return UserVF;
+ }
+
+ assert(ElementCount::isKnownGT(UserVF, MaxSafeUserVF));
+
+ // Only clamp if the UserVF is not scalable. If the UserVF is scalable, it
+ // is better to ignore the hint and let the compiler choose a suitable VF.
+ if (!UserVF.isScalable()) {
+ LLVM_DEBUG(dbgs() << "LV: User VF=" << UserVF
+ << " is unsafe, clamping to max safe VF="
+ << MaxSafeFixedVF << ".\n");
+ ORE->emit([&]() {
+ return OptimizationRemarkAnalysis(DEBUG_TYPE, "VectorizationFactor",
+ TheLoop->getStartLoc(),
+ TheLoop->getHeader())
+ << "User-specified vectorization factor "
+ << ore::NV("UserVectorizationFactor", UserVF)
+ << " is unsafe, clamping to maximum safe vectorization factor "
+ << ore::NV("VectorizationFactor", MaxSafeFixedVF);
+ });
+ return MaxSafeFixedVF;
+ }
+
+ if (!supportsScalableVectors()) {
+ LLVM_DEBUG(dbgs() << "LV: User VF=" << UserVF
+ << " is ignored because scalable vectors are not "
+ "available.\n");
+ ORE->emit([&]() {
+ return OptimizationRemarkAnalysis(DEBUG_TYPE, "VectorizationFactor",
+ TheLoop->getStartLoc(),
+ TheLoop->getHeader())
+ << "User-specified vectorization factor "
+ << ore::NV("UserVectorizationFactor", UserVF)
+ << " is ignored because the target does not support scalable "
+ "vectors. The compiler will pick a more suitable value.";
+ });
+ } else {
+ LLVM_DEBUG(dbgs() << "LV: User VF=" << UserVF
+ << " is unsafe. Ignoring scalable UserVF.\n");
+ ORE->emit([&]() {
+ return OptimizationRemarkAnalysis(DEBUG_TYPE, "VectorizationFactor",
+ TheLoop->getStartLoc(),
+ TheLoop->getHeader())
+ << "User-specified vectorization factor "
+ << ore::NV("UserVectorizationFactor", UserVF)
+ << " is unsafe. Ignoring the hint to let the compiler pick a "
+ "more suitable value.";
+ });
+ }
+ }
+
+ LLVM_DEBUG(dbgs() << "LV: The Smallest and Widest types: " << SmallestType
+ << " / " << WidestType << " bits.\n");
+
+ FixedScalableVFPair Result(ElementCount::getFixed(1),
+ ElementCount::getScalable(0));
+ if (auto MaxVF = getMaximizedVFForTarget(
+ MaxTripCount, SmallestType, WidestType, MaxSafeFixedVF, UserIC,
+ FoldTailByMasking, RequiresScalarEpilogue))
+ Result.FixedVF = MaxVF;
+
+ if (auto MaxVF = getMaximizedVFForTarget(
+ MaxTripCount, SmallestType, WidestType, MaxSafeScalableVF, UserIC,
+ FoldTailByMasking, RequiresScalarEpilogue))
+ if (MaxVF.isScalable()) {
+ Result.ScalableVF = MaxVF;
+ LLVM_DEBUG(dbgs() << "LV: Found feasible scalable VF = " << MaxVF
+ << "\n");
+ }
+
+ return Result;
+}
+
+std::pair<unsigned, unsigned>
+VFSelectionContext::getSmallestAndWidestTypes() const {
+ unsigned MinWidth = -1U;
+ unsigned MaxWidth = 8;
+ const DataLayout &DL = F.getDataLayout();
+ // For in-loop reductions, no element types are added to ElementTypesInLoop
+ // if there are no loads/stores in the loop. In this case, check through the
+ // reduction variables to determine the maximum width.
+ if (ElementTypesInLoop.empty() && !Legal->getReductionVars().empty()) {
+ for (const auto &[_, RdxDesc] : Legal->getReductionVars()) {
+ // When finding the min width used by the recurrence we need to account
+ // for casts on the input operands of the recurrence.
+ MinWidth = std::min(
+ MinWidth,
+ std::min(RdxDesc.getMinWidthCastToRecurrenceTypeInBits(),
+ RdxDesc.getRecurrenceType()->getScalarSizeInBits()));
+ MaxWidth = std::max(MaxWidth,
+ RdxDesc.getRecurrenceType()->getScalarSizeInBits());
+ }
+ } else {
+ for (Type *T : ElementTypesInLoop) {
+ MinWidth = std::min<unsigned>(
+ MinWidth, DL.getTypeSizeInBits(T->getScalarType()).getFixedValue());
+ MaxWidth = std::max<unsigned>(
+ MaxWidth, DL.getTypeSizeInBits(T->getScalarType()).getFixedValue());
+ }
+ }
+ return {MinWidth, MaxWidth};
+}
+
+void VFSelectionContext::collectElementTypesForWidening(
+ const SmallPtrSetImpl<const Value *> *ValuesToIgnore) {
+ ElementTypesInLoop.clear();
+ // For each block.
+ for (BasicBlock *BB : TheLoop->blocks()) {
+ // For each instruction in the loop.
+ for (Instruction &I : *BB) {
+ Type *T = I.getType();
+
+ // Skip ignored values.
+ if (ValuesToIgnore && ValuesToIgnore->contains(&I))
+ continue;
+
+ // Only examine Loads, Stores and PHINodes.
+ if (!isa<LoadInst, StoreInst, PHINode>(I))
+ continue;
+
+ // Examine PHI nodes that are reduction variables. Update the type to
+ // account for the recurrence type.
+ if (auto *PN = dyn_cast<PHINode>(&I)) {
+ if (!Legal->isReductionVariable(PN))
+ continue;
+ const RecurrenceDescriptor &RdxDesc =
+ Legal->getRecurrenceDescriptor(PN);
+ if (PreferInLoopReductions || useOrderedReductions(RdxDesc) ||
+ TTI.preferInLoopReduction(RdxDesc.getRecurrenceKind(),
+ RdxDesc.getRecurrenceType()))
+ continue;
+ T = RdxDesc.getRecurrenceType();
+ }
+
+ // Examine the stored values.
+ if (auto *ST = dyn_cast<StoreInst>(&I))
+ T = ST->getValueOperand()->getType();
+
+ assert(T->isSized() &&
+ "Expected the load/store/recurrence type to be sized");
+
+ ElementTypesInLoop.insert(T);
+ }
+ }
+}
+
+void VFSelectionContext::initializeVScaleForTuning() {
+ if (!supportsScalableVectors())
+ return;
+
+ if (F.hasFnAttribute(Attribute::VScaleRange)) {
+ auto Attr = F.getFnAttribute(Attribute::VScaleRange);
+ auto Min = Attr.getVScaleRangeMin();
+ auto Max = Attr.getVScaleRangeMax();
+ if (Max && Min == Max) {
+ VScaleForTuning = Max;
+ return;
+ }
+ }
+
+ VScaleForTuning = TTI.getVScaleForTuning();
+}
+
+bool VFSelectionContext::useOrderedReductions(
+ const RecurrenceDescriptor &RdxDesc) const {
+ return !Hints->allowReordering() && RdxDesc.isOrdered();
+}
+
+bool VFSelectionContext::runtimeChecksRequired() {
+ LLVM_DEBUG(dbgs() << "LV: Performing code size checks.\n");
+
+ Loop *L = const_cast<Loop *>(TheLoop);
+ if (Legal->getRuntimePointerChecking()->Need) {
+ reportVectorizationFailure(
+ "Runtime ptr check is required with -Os/-Oz",
+ "runtime pointer checks needed. Enable vectorization of this "
+ "loop with '#pragma clang loop vectorize(enable)' when "
+ "compiling with -Os/-Oz",
+ "CantVersionLoopWithOptForSize", ORE, L);
+ return true;
+ }
+
+ if (!PSE.getPredicate().isAlwaysTrue()) {
+ reportVectorizationFailure(
+ "Runtime SCEV check is required with -Os/-Oz",
+ "runtime SCEV checks needed. Enable vectorization of this "
+ "loop with '#pragma clang loop vectorize(enable)' when "
+ "compiling with -Os/-Oz",
+ "CantVersionLoopWithOptForSize", ORE, L);
+ return true;
+ }
+
+ // FIXME: Avoid specializing for stride==1 instead of bailing out.
+ if (!Legal->getLAI()->getSymbolicStrides().empty()) {
+ reportVectorizationFailure(
+ "Runtime stride check for small trip count",
+ "runtime stride == 1 checks needed. Enable vectorization of "
+ "this loop without such check by compiling with -Os/-Oz",
+ "CantVersionLoopWithOptForSize", ORE, L);
+ return true;
+ }
+
+ return false;
+}
+
+void VFSelectionContext::collectInLoopReductions() {
+ // Avoid duplicating work finding in-loop reductions.
+ if (!InLoopReductions.empty())
+ return;
+
+ for (const auto &Reduction : Legal->getReductionVars()) {
+ PHINode *Phi = Reduction.first;
+ const RecurrenceDescriptor &RdxDesc = Reduction.second;
+
+ // Multi-use reductions (e.g., used in FindLastIV patterns) are handled
+ // separately and should not be considered for in-loop reductions.
+ if (RdxDesc.hasUsesOutsideReductionChain())
+ continue;
+
+ // We don't collect reductions that are type promoted (yet).
+ if (RdxDesc.getRecurrenceType() != Phi->getType())
+ continue;
+
+ // In-loop AnyOf and FindIV reductions are not yet supported.
+ RecurKind Kind = RdxDesc.getRecurrenceKind();
+ if (RecurrenceDescriptor::isAnyOfRecurrenceKind(Kind) ||
+ RecurrenceDescriptor::isFindIVRecurrenceKind(Kind) ||
+ RecurrenceDescriptor::isFindLastRecurrenceKind(Kind))
+ continue;
+
+ // If the target would prefer this reduction to happen "in-loop", then we
+ // want to record it as such.
+ if (!PreferInLoopReductions && !useOrderedReductions(RdxDesc) &&
+ !TTI.preferInLoopReduction(Kind, Phi->getType()))
+ continue;
+
+ // Check that we can correctly put the reductions into the loop, by
+ // finding the chain of operations that leads from the phi to the loop
+ // exit value.
+ SmallVector<Instruction *, 4> ReductionOperations =
+ RdxDesc.getReductionOpChain(Phi, const_cast<Loop *>(TheLoop));
+ bool InLoop = !ReductionOperations.empty();
+
+ if (InLoop) {
+ InLoopReductions.insert(Phi);
+ // Add the elements to InLoopReductionImmediateChains for cost modelling.
+ Instruction *LastChain = Phi;
+ for (auto *I : ReductionOperations) {
+ InLoopReductionImmediateChains[I] = LastChain;
+ LastChain = I;
+ }
+ }
+ LLVM_DEBUG(dbgs() << "LV: Using " << (InLoop ? "inloop" : "out of loop")
+ << " reduction for phi: " << *Phi << "\n");
+ }
+}
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index f06b959700687..9aa4d006cc314 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -26,6 +26,7 @@
#include "VPlan.h"
#include "llvm/ADT/SmallSet.h"
+#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/Support/InstructionCost.h"
namespace {
@@ -40,9 +41,9 @@ class LoopVectorizationLegality;
class LoopVectorizationCostModel;
class PredicatedScalarEvolution;
class LoopVectorizeHints;
+class RecurrenceDescriptor;
class LoopVersioning;
class OptimizationRemarkEmitter;
-class TargetTransformInfo;
class TargetLibraryInfo;
class VPRecipeBuilder;
struct VPRegisterUsage;
@@ -50,6 +51,21 @@ struct VFRange;
extern cl::opt<bool> EnableVPlanNativePath;
extern cl::opt<unsigned> ForceTargetInstructionCost;
+extern cl::opt<bool> PreferInLoopReductions;
+
+/// \return An upper bound for vscale based on TTI or the vscale_range
+/// attribute.
+std::optional<unsigned> getMaxVScale(const Function &F,
+ const TargetTransformInfo &TTI);
+
+/// Reports an informative message: print \p Msg for debugging purposes as well
+/// as an optimization remark. Uses either \p I as location of the remark, or
+/// otherwise \p TheLoop. If \p DL is passed, use it as debug location for the
+/// remark.
+void reportVectorizationInfo(const StringRef Msg, const StringRef ORETag,
+ OptimizationRemarkEmitter *ORE,
+ const Loop *TheLoop, Instruction *I = nullptr,
+ DebugLoc DL = {});
/// VPlan-based builder utility analogous to IRBuilder.
class VPBuilder {
@@ -496,6 +512,185 @@ struct FixedScalableVFPair {
bool hasVector() const { return FixedVF.isVector() || ScalableVF.isVector(); }
};
+/// Holds state needed to make cost decisions before computing costs per-VF,
+/// including the maximum VFs.
+class VFSelectionContext {
+ /// \return True if maximizing vector bandwidth is enabled by the target or
+ /// user options, for the given register kind (scalable or fixed-width).
+ bool useMaxBandwidth(bool IsScalable) const;
+
+ /// \return the maximized element count based on the targets vector
+ /// registers and the loop trip-count, but limited to a maximum safe VF.
+ /// This is a helper function of computeFeasibleMaxVF.
+ ElementCount getMaximizedVFForTarget(unsigned MaxTripCount,
+ unsigned SmallestType,
+ unsigned WidestType,
+ ElementCount MaxSafeVF, unsigned UserIC,
+ bool FoldTailByMasking,
+ bool RequiresScalarEpilogue);
+
+ /// If \p VF * \p UserIC > MaxTripcount, clamps VF to the next lower VF
+ /// that results in VF * UserIC <= MaxTripCount.
+ ElementCount clampVFByMaxTripCount(ElementCount VF, unsigned MaxTripCount,
+ unsigned UserIC, bool FoldTailByMasking,
+ bool RequiresScalarEpilogue) const;
+
+ /// Checks if scalable vectorization is supported and enabled. Caches the
+ /// result to avoid repeated debug dumps for repeated queries.
+ bool isScalableVectorizationAllowed();
+
+ /// \return the maximum legal scalable VF, based on the safe max number
+ /// of elements.
+ ElementCount getMaxLegalScalableVF(unsigned MaxSafeElements);
+
+ /// Initializes the value of vscale used for tuning the cost model. If
+ /// vscale_range.min == vscale_range.max then return vscale_range.max, else
+ /// return the value returned by the corresponding TTI method.
+ void initializeVScaleForTuning();
+
+ const TargetTransformInfo &TTI;
+ const LoopVectorizationLegality *Legal;
+ const Loop *TheLoop;
+ const Function &F;
+ PredicatedScalarEvolution &PSE;
+ OptimizationRemarkEmitter *ORE;
+ const LoopVectorizeHints *Hints;
+
+ /// Cached result of isScalableVectorizationAllowed.
+ std::optional<bool> IsScalableVectorizationAllowed;
+
+ /// Used to store the value of vscale used for tuning the cost model. It is
+ /// initialized during object construction.
+ std::optional<unsigned> VScaleForTuning;
+
+ /// The highest VF possible for this loop, without using MaxBandwidth.
+ FixedScalableVFPair MaxPermissibleVFWithoutMaxBW;
+
+ /// All element types found in the loop.
+ SmallPtrSet<Type *, 16> ElementTypesInLoop;
+
+ /// PHINodes of the reductions that should be expanded in-loop. Set by
+ /// collectInLoopReductions.
+ SmallPtrSet<PHINode *, 4> InLoopReductions;
+
+ /// A Map of inloop reduction operations and their immediate chain operand.
+ /// FIXME: This can be removed once reductions can be costed correctly in
+ /// VPlan. This was added to allow quick lookup of the inloop operations.
+ /// Set by collectInLoopReductions.
+ DenseMap<Instruction *, Instruction *> InLoopReductionImmediateChains;
+
+ /// Maximum safe number of elements to be processed per vector iteration,
+ /// which do not prevent store-load forwarding and are safe with regard to the
+ /// memory dependencies. Required for EVL-based vectorization, where this
+ /// value is used as the upper bound of the safe AVL. Set by
+ /// computeFeasibleMaxVF.
+ std::optional<unsigned> MaxSafeElements;
+
+public:
+ /// The kind of cost that we are calculating.
+ const TTI::TargetCostKind CostKind;
+
+ /// Whether this loop should be optimized for size based on function attribute
+ /// or profile information.
+ const bool OptForSize;
+
+ VFSelectionContext(const TargetTransformInfo &TTI,
+ const LoopVectorizationLegality *Legal,
+ const Loop *TheLoop, const Function &F,
+ PredicatedScalarEvolution &PSE,
+ OptimizationRemarkEmitter *ORE,
+ const LoopVectorizeHints *Hints, bool OptForSize)
+ : TTI(TTI), Legal(Legal), TheLoop(TheLoop), F(F), PSE(PSE), ORE(ORE),
+ Hints(Hints),
+ CostKind(F.hasMinSize() ? TTI::TCK_CodeSize : TTI::TCK_RecipThroughput),
+ OptForSize(OptForSize) {
+ initializeVScaleForTuning();
+ }
+
+ /// \return The vscale value used for tuning the cost model.
+ std::optional<unsigned> getVScaleForTuning() const { return VScaleForTuning; }
+
+ /// \return True if register pressure should be considered for the given VF.
+ bool shouldConsiderRegPressureForVF(ElementCount VF) const;
+
+ /// \return True if scalable vectors are supported by the target or forced.
+ bool supportsScalableVectors() const;
+
+ /// Collect element types in the loop that need widening.
+ void collectElementTypesForWidening(
+ const SmallPtrSetImpl<const Value *> *ValuesToIgnore = nullptr);
+
+ /// \return The size (in bits) of the smallest and widest types in the code
+ /// that need to be vectorized. We ignore values that remain scalar such as
+ /// 64 bit loop indices.
+ std::pair<unsigned, unsigned> getSmallestAndWidestTypes() const;
+
+ /// \return An upper bound for the vectorization factors for both
+ /// fixed and scalable vectorization, where the minimum-known number of
+ /// elements is a power-of-2 larger than zero. If scalable vectorization is
+ /// disabled or unsupported, then the scalable part will be equal to
+ /// ElementCount::getScalable(0). Also sets MaxSafeElements.
+ FixedScalableVFPair computeFeasibleMaxVF(unsigned MaxTripCount,
+ ElementCount UserVF, unsigned UserIC,
+ bool FoldTailByMasking,
+ bool RequiresScalarEpilogue);
+
+ /// Return maximum safe number of elements to be processed per vector
+ /// iteration, which do not prevent store-load forwarding and are safe with
+ /// regard to the memory dependencies. Required for EVL-based VPlans to
+ /// correctly calculate AVL (application vector length) as min(remaining AVL,
+ /// MaxSafeElements). Set by computeFeasibleMaxVF.
+ /// TODO: need to consider adjusting cost model to use this value as a
+ /// vectorization factor for EVL-based vectorization.
+ std::optional<unsigned> getMaxSafeElements() const { return MaxSafeElements; }
+
+ /// Returns true if we should use strict in-order reductions for the given
+ /// RdxDesc. This is true if the -enable-strict-reductions flag is passed,
+ /// the IsOrdered flag of RdxDesc is set and we do not allow reordering
+ /// of FP operations.
+ bool useOrderedReductions(const RecurrenceDescriptor &RdxDesc) const;
+
+ /// Returns true if the target machine supports masked store operation
+ /// for the given \p DataType and kind of access to \p Ptr.
+ bool isLegalMaskedStore(Type *DataType, Value *Ptr, Align Alignment,
+ unsigned AddressSpace) const;
+
+ /// Returns true if the target machine supports masked load operation
+ /// for the given \p DataType and kind of access to \p Ptr.
+ bool isLegalMaskedLoad(Type *DataType, Value *Ptr, Align Alignment,
+ unsigned AddressSpace) const;
+
+ /// Returns true if the target machine can represent \p V as a masked gather
+ /// or scatter operation.
+ bool isLegalGatherOrScatter(Value *V, ElementCount VF) const;
+
+ /// Split reductions into those that happen in the loop, and those that
+ /// happen outside. In-loop reductions are collected into InLoopReductions.
+ /// InLoopReductionImmediateChains is filled with each in-loop reduction
+ /// operation and its immediate chain operand for use during cost modelling.
+ void collectInLoopReductions();
+
+ /// Returns true if the Phi is part of an inloop reduction.
+ bool isInLoopReduction(PHINode *Phi) const {
+ return InLoopReductions.contains(Phi);
+ }
+
+ /// Returns the set of in-loop reduction PHIs.
+ const SmallPtrSetImpl<PHINode *> &getInLoopReductions() const {
+ return InLoopReductions;
+ }
+
+ /// Returns the immediate chain operand of in-loop reduction operation \p I,
+ /// or nullptr if \p I is not an in-loop reduction operation.
+ Instruction *getInLoopReductionImmediateChain(Instruction *I) const {
+ return InLoopReductionImmediateChains.lookup(I);
+ }
+
+ /// Check whether vectorization would require runtime checks. When optimizing
+ /// for size, returning true here aborts vectorization.
+ bool runtimeChecksRequired();
+};
+
/// Planner drives the vectorization process after having passed
/// Legality checks.
class LoopVectorizationPlanner {
@@ -520,6 +715,9 @@ class LoopVectorizationPlanner {
/// The profitability analysis.
LoopVectorizationCostModel &CM;
+ /// VF selection state independent of cost-modeling decisions.
+ VFSelectionContext &Config;
+
/// The interleaved access analysis.
InterleavedAccessInfo &IAI;
@@ -557,11 +755,11 @@ class LoopVectorizationPlanner {
LoopVectorizationPlanner(
Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
- LoopVectorizationCostModel &CM, InterleavedAccessInfo &IAI,
- PredicatedScalarEvolution &PSE, const LoopVectorizeHints &Hints,
- OptimizationRemarkEmitter *ORE)
+ LoopVectorizationCostModel &CM, VFSelectionContext &Config,
+ InterleavedAccessInfo &IAI, PredicatedScalarEvolution &PSE,
+ const LoopVectorizeHints &Hints, OptimizationRemarkEmitter *ORE)
: OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal), CM(CM),
- IAI(IAI), PSE(PSE), Hints(Hints), ORE(ORE) {}
+ Config(Config), IAI(IAI), PSE(PSE), Hints(Hints), ORE(ORE) {}
/// Build VPlans for the specified \p UserVF and \p UserIC if they are
/// non-zero or all applicable candidate VFs otherwise. If vectorization and
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index c6ff52081b136..3ce41369fea9a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -198,13 +198,6 @@ static cl::opt<unsigned> VectorizeMemoryCheckThreshold(
"vectorize-memory-check-threshold", cl::init(128), cl::Hidden,
cl::desc("The maximum allowed number of runtime memory checks"));
-/// Note: This currently only applies to `llvm.masked.load` and
-/// `llvm.masked.store`. TODO: Extend this to cover other operations as needed.
-static cl::opt<bool> ForceTargetSupportsMaskedMemoryOps(
- "force-target-supports-masked-memory-ops", cl::init(false), cl::Hidden,
- cl::desc("Assume the target supports masked memory operations (used for "
- "testing)."));
-
/// Option tail-folding-policy indicates that an epilogue is undesired, that
/// tail folding is preferred, and this lists all options. I.e., the vectorizer
/// will try to fold the tail-loop (epilogue) into the vector body and predicate
@@ -248,11 +241,6 @@ cl::opt<bool> llvm::EnableWideActiveLaneMask(
cl::desc("Enable use of wide lane masks when used for control flow in "
"tail-folded loops"));
-static cl::opt<bool> MaximizeBandwidth(
- "vectorizer-maximize-bandwidth", cl::init(false), cl::Hidden,
- cl::desc("Maximize bandwidth when selecting vectorization factor which "
- "will be determined by the smallest type in loop."));
-
static cl::opt<bool> EnableInterleavedMemAccesses(
"enable-interleaved-mem-accesses", cl::init(false), cl::Hidden,
cl::desc("Enable vectorization on interleaved memory accesses in a loop"));
@@ -287,12 +275,6 @@ cl::opt<unsigned> llvm::ForceTargetInstructionCost(
"an instruction to a single constant value. Mostly "
"useful for getting consistent testing."));
-static cl::opt<bool> ForceTargetSupportsScalableVectors(
- "force-target-supports-scalable-vectors", cl::init(false), cl::Hidden,
- cl::desc(
- "Pretend that scalable vectors are supported, even if the target does "
- "not support them. This flag should only be used for testing."));
-
static cl::opt<unsigned> SmallLoopCost(
"small-loop-cost", cl::init(20), cl::Hidden,
cl::desc(
@@ -324,12 +306,6 @@ static cl::opt<unsigned> MaxNestedScalarReductionIC(
cl::desc("The maximum interleave count to use when interleaving a scalar "
"reduction in a nested loop."));
-static cl::opt<bool>
- PreferInLoopReductions("prefer-inloop-reductions", cl::init(false),
- cl::Hidden,
- cl::desc("Prefer in-loop vector reductions, "
- "overriding the targets preference."));
-
static cl::opt<bool> ForceOrderedReductions(
"force-ordered-reductions", cl::init(false), cl::Hidden,
cl::desc("Enable the vectorisation of loops with in-order (strict) "
@@ -393,20 +369,11 @@ static cl::opt<cl::boolOrDefault> ForceSafeDivisor(
cl::desc(
"Override cost based safe divisor widening for div/rem instructions"));
-static cl::opt<bool> UseWiderVFIfCallVariantsPresent(
- "vectorizer-maximize-bandwidth-for-vector-calls", cl::init(true),
- cl::Hidden,
- cl::desc("Try wider VFs if they enable the use of vector variants"));
-
static cl::opt<bool> EnableEarlyExitVectorization(
"enable-early-exit-vectorization", cl::init(true), cl::Hidden,
cl::desc(
"Enable vectorization of early exit loops with uncountable exits."));
-static cl::opt<bool> ConsiderRegPressure(
- "vectorizer-consider-reg-pressure", cl::init(false), cl::Hidden,
- cl::desc("Discard VFs if their register pressure is too high."));
-
// Likelyhood of bypassing the vectorized loop because there are zero trips left
// after prolog. See `emitIterationCountCheck`.
static constexpr uint32_t MinItersBypassWeights[] = {1, 127};
@@ -756,8 +723,8 @@ static void debugVectorizationMessage(const StringRef Prefix,
/// the location of the remark. If \p DL is passed, use it as debug location for
/// the remark. \return the remark object that can be streamed to.
static OptimizationRemarkAnalysis
-createLVAnalysis(const char *PassName, StringRef RemarkName, Loop *TheLoop,
- Instruction *I, DebugLoc DL = {}) {
+createLVAnalysis(const char *PassName, StringRef RemarkName,
+ const Loop *TheLoop, Instruction *I, DebugLoc DL = {}) {
BasicBlock *CodeRegion = I ? I->getParent() : TheLoop->getHeader();
// If debug location is attached to the instruction, use it. Otherwise if DL
// was not provided, use the loop's.
@@ -787,14 +754,9 @@ void reportVectorizationFailure(const StringRef DebugMsg,
<< "loop not vectorized: " << OREMsg);
}
-/// Reports an informative message: print \p Msg for debugging purposes as well
-/// as an optimization remark. Uses either \p I as location of the remark, or
-/// otherwise \p TheLoop. If \p DL is passed, use it as debug location for the
-/// remark. If \p DL is passed, use it as debug location for the remark.
-static void reportVectorizationInfo(const StringRef Msg, const StringRef ORETag,
- OptimizationRemarkEmitter *ORE,
- Loop *TheLoop, Instruction *I = nullptr,
- DebugLoc DL = {}) {
+void reportVectorizationInfo(const StringRef Msg, const StringRef ORETag,
+ OptimizationRemarkEmitter *ORE,
+ const Loop *TheLoop, Instruction *I, DebugLoc DL) {
LLVM_DEBUG(debugVectorizationMessage("", Msg, I));
LoopVectorizeHints Hints(TheLoop, true /* doesn't matter */, *ORE);
ORE->emit(createLVAnalysis(Hints.vectorizeAnalysisPassName(), ORETag, TheLoop,
@@ -857,46 +819,23 @@ class LoopVectorizationCostModel {
friend class LoopVectorizationPlanner;
public:
- LoopVectorizationCostModel(EpilogueLowering SEL, Loop *L,
- PredicatedScalarEvolution &PSE, LoopInfo *LI,
- LoopVectorizationLegality *Legal,
- const TargetTransformInfo &TTI,
- const TargetLibraryInfo *TLI, DemandedBits *DB,
- AssumptionCache *AC,
- OptimizationRemarkEmitter *ORE,
- std::function<BlockFrequencyInfo &()> GetBFI,
- const Function *F, const LoopVectorizeHints *Hints,
- InterleavedAccessInfo &IAI, bool OptForSize)
- : EpilogueLoweringStatus(SEL), TheLoop(L), PSE(PSE), LI(LI), Legal(Legal),
- TTI(TTI), TLI(TLI), DB(DB), AC(AC), ORE(ORE), GetBFI(GetBFI),
- TheFunction(F), Hints(Hints), InterleaveInfo(IAI),
- OptForSize(OptForSize) {
- if (TTI.supportsScalableVectors() || ForceTargetSupportsScalableVectors)
- initializeVScaleForTuning();
- CostKind = F->hasMinSize() ? TTI::TCK_CodeSize : TTI::TCK_RecipThroughput;
- }
+ LoopVectorizationCostModel(
+ EpilogueLowering SEL, Loop *L, PredicatedScalarEvolution &PSE,
+ LoopInfo *LI, LoopVectorizationLegality *Legal,
+ const TargetTransformInfo &TTI, const TargetLibraryInfo *TLI,
+ DemandedBits *DB, AssumptionCache *AC, OptimizationRemarkEmitter *ORE,
+ std::function<BlockFrequencyInfo &()> GetBFI, const Function *F,
+ const LoopVectorizeHints *Hints, InterleavedAccessInfo &IAI,
+ VFSelectionContext &Config)
+ : Config(Config), EpilogueLoweringStatus(SEL), TheLoop(L), PSE(PSE),
+ LI(LI), Legal(Legal), TTI(TTI), TLI(TLI), DB(DB), AC(AC), ORE(ORE),
+ GetBFI(GetBFI), TheFunction(F), Hints(Hints), InterleaveInfo(IAI) {}
/// \return An upper bound for the vectorization factors (both fixed and
/// scalable). If the factors are 0, vectorization and interleaving should be
/// avoided up front.
FixedScalableVFPair computeMaxVF(ElementCount UserVF, unsigned UserIC);
- /// \return True if runtime checks are required for vectorization, and false
- /// otherwise.
- bool runtimeChecksRequired();
-
- /// \return True if maximizing vector bandwidth is enabled by the target or
- /// user options, for the given register kind.
- bool useMaxBandwidth(TargetTransformInfo::RegisterKind RegKind);
-
- /// \return True if register pressure should be considered for the given VF.
- bool shouldConsiderRegPressureForVF(ElementCount VF);
-
- /// \return The size (in bits) of the smallest and widest types in the code
- /// that needs to be vectorized. We ignore values that remain scalar such as
- /// 64 bit loop indices.
- std::pair<unsigned, unsigned> getSmallestAndWidestTypes();
-
/// Memory access instruction may be vectorized in more than one way.
/// Form of instruction after vectorization depends on cost.
/// This function takes cost-based decisions for Load/Store instructions
@@ -916,21 +855,6 @@ class LoopVectorizationCostModel {
/// Collect values we want to ignore in the cost model.
void collectValuesToIgnore();
- /// Collect all element types in the loop for which widening is needed.
- void collectElementTypesForWidening();
-
- /// Split reductions into those that happen in the loop, and those that happen
- /// outside. In loop reductions are collected into InLoopReductions.
- void collectInLoopReductions();
-
- /// Returns true if we should use strict in-order reductions for the given
- /// RdxDesc. This is true if the -enable-strict-reductions flag is passed,
- /// the IsOrdered flag of RdxDesc is set and we do not allow reordering
- /// of FP operations.
- bool useOrderedReductions(const RecurrenceDescriptor &RdxDesc) const {
- return !Hints->allowReordering() && RdxDesc.isOrdered();
- }
-
/// \returns The smallest bitwidth each instruction can be represented with.
/// The vector equivalents of these instructions should be truncated to this
/// type.
@@ -1145,48 +1069,6 @@ class LoopVectorizationCostModel {
collectInstsToScalarize(VF);
}
- /// Returns true if the target machine supports masked store operation
- /// for the given \p DataType and kind of access to \p Ptr.
- bool isLegalMaskedStore(Type *DataType, Value *Ptr, Align Alignment,
- unsigned AddressSpace) const {
- return Legal->isConsecutivePtr(DataType, Ptr) &&
- (ForceTargetSupportsMaskedMemoryOps ||
- TTI.isLegalMaskedStore(DataType, Alignment, AddressSpace));
- }
-
- /// Returns true if the target machine supports masked load operation
- /// for the given \p DataType and kind of access to \p Ptr.
- bool isLegalMaskedLoad(Type *DataType, Value *Ptr, Align Alignment,
- unsigned AddressSpace) const {
- return Legal->isConsecutivePtr(DataType, Ptr) &&
- (ForceTargetSupportsMaskedMemoryOps ||
- TTI.isLegalMaskedLoad(DataType, Alignment, AddressSpace));
- }
-
- /// Returns true if the target machine can represent \p V as a masked gather
- /// or scatter operation.
- bool isLegalGatherOrScatter(Value *V, ElementCount VF) {
- bool LI = isa<LoadInst>(V);
- bool SI = isa<StoreInst>(V);
- if (!LI && !SI)
- return false;
- auto *Ty = getLoadStoreType(V);
- Align Align = getLoadStoreAlignment(V);
- if (VF.isVector())
- Ty = VectorType::get(Ty, VF);
- return (LI && TTI.isLegalMaskedGather(Ty, Align)) ||
- (SI && TTI.isLegalMaskedScatter(Ty, Align));
- }
-
- /// Returns true if the target machine supports all of the reduction
- /// variables found for the given VF.
- bool canVectorizeReductions(ElementCount VF) const {
- return (all_of(Legal->getReductionVars(), [&](auto &Reduction) -> bool {
- const RecurrenceDescriptor &RdxDesc = Reduction.second;
- return TTI.isLegalToVectorizeReduction(RdxDesc, VF);
- }));
- }
-
/// Given costs for both strategies, return true if the scalar predication
/// lowering should be used for div/rem. This incorporates an override
/// option so it is not simply a cost comparison.
@@ -1359,15 +1241,6 @@ class LoopVectorizationCostModel {
return getTailFoldingStyle() == TailFoldingStyle::DataAndControlFlow;
}
- /// Return maximum safe number of elements to be processed per vector
- /// iteration, which do not prevent store-load forwarding and are safe with
- /// regard to the memory dependencies. Required for EVL-based VPlans to
- /// correctly calculate AVL (application vector length) as min(remaining AVL,
- /// MaxSafeElements).
- /// TODO: need to consider adjusting cost model to use this value as a
- /// vectorization factor for EVL-based vectorization.
- std::optional<unsigned> getMaxSafeElements() const { return MaxSafeElements; }
-
/// Returns true if the instructions in this block requires predication
/// for any reason, e.g. because tail folding now requires a predicate
/// or because the block in the original loop was predicated.
@@ -1381,16 +1254,6 @@ class LoopVectorizationCostModel {
return getTailFoldingStyle() == TailFoldingStyle::DataWithEVL;
}
- /// Returns true if the Phi is part of an inloop reduction.
- bool isInLoopReduction(PHINode *Phi) const {
- return InLoopReductions.contains(Phi);
- }
-
- /// Returns the set of in-loop reduction PHIs.
- const SmallPtrSetImpl<PHINode *> &getInLoopReductions() const {
- return InLoopReductions;
- }
-
/// Returns true if the predicated reduction select should be used to set the
/// incoming value for the reduction phi.
bool usePredicatedReductionSelect(RecurKind RecurrenceKind) const {
@@ -1455,65 +1318,11 @@ class LoopVectorizationCostModel {
/// trivially hoistable.
bool shouldConsiderInvariant(Value *Op);
- /// Return the value of vscale used for tuning the cost model.
- std::optional<unsigned> getVScaleForTuning() const { return VScaleForTuning; }
-
private:
unsigned NumPredStores = 0;
- /// Used to store the value of vscale used for tuning the cost model. It is
- /// initialized during object construction.
- std::optional<unsigned> VScaleForTuning;
-
- /// Initializes the value of vscale used for tuning the cost model. If
- /// vscale_range.min == vscale_range.max then return vscale_range.max, else
- /// return the value returned by the corresponding TTI method.
- void initializeVScaleForTuning() {
- const Function *Fn = TheLoop->getHeader()->getParent();
- if (Fn->hasFnAttribute(Attribute::VScaleRange)) {
- auto Attr = Fn->getFnAttribute(Attribute::VScaleRange);
- auto Min = Attr.getVScaleRangeMin();
- auto Max = Attr.getVScaleRangeMax();
- if (Max && Min == Max) {
- VScaleForTuning = Max;
- return;
- }
- }
-
- VScaleForTuning = TTI.getVScaleForTuning();
- }
-
- /// \return An upper bound for the vectorization factors for both
- /// fixed and scalable vectorization, where the minimum-known number of
- /// elements is a power-of-2 larger than zero. If scalable vectorization is
- /// disabled or unsupported, then the scalable part will be equal to
- /// ElementCount::getScalable(0).
- FixedScalableVFPair computeFeasibleMaxVF(unsigned MaxTripCount,
- ElementCount UserVF, unsigned UserIC,
- bool FoldTailByMasking);
-
- /// If \p VF * \p UserIC > MaxTripcount, clamps VF to the next lower VF that
- /// results in VF * UserIC <= MaxTripCount.
- ElementCount clampVFByMaxTripCount(ElementCount VF, unsigned MaxTripCount,
- unsigned UserIC,
- bool FoldTailByMasking) const;
-
- /// \return the maximized element count based on the targets vector
- /// registers and the loop trip-count, but limited to a maximum safe VF.
- /// This is a helper function of computeFeasibleMaxVF.
- ElementCount getMaximizedVFForTarget(unsigned MaxTripCount,
- unsigned SmallestType,
- unsigned WidestType,
- ElementCount MaxSafeVF, unsigned UserIC,
- bool FoldTailByMasking);
-
- /// Checks if scalable vectorization is supported and enabled. Caches the
- /// result to avoid repeated debug dumps for repeated queries.
- bool isScalableVectorizationAllowed();
-
- /// \return the maximum legal scalable VF, based on the safe max number
- /// of elements.
- ElementCount getMaxLegalScalableVF(unsigned MaxSafeElements);
+ /// VF selection state independent of cost-modeling decisions.
+ VFSelectionContext &Config;
/// Calculate vectorization cost of memory instruction \p I.
InstructionCost getMemoryInstructionCost(Instruction *I, ElementCount VF);
@@ -1569,15 +1378,6 @@ class LoopVectorizationCostModel {
/// Control finally chosen tail folding style.
TailFoldingStyle ChosenTailFoldingStyle = TailFoldingStyle::None;
- /// true if scalable vectorization is supported and enabled.
- std::optional<bool> IsScalableVectorizationAllowed;
-
- /// Maximum safe number of elements to be processed per vector iteration,
- /// which do not prevent store-load forwarding and are safe with regard to the
- /// memory dependencies. Required for EVL-based veectorization, where this
- /// value is used as the upper bound of the safe AVL.
- std::optional<unsigned> MaxSafeElements;
-
/// A map holding scalar costs for
diff erent vectorization factors. The
/// presence of a cost for an instruction in the mapping indicates that the
/// instruction will be scalarized when vectorizing with the associated
@@ -1596,14 +1396,6 @@ class LoopVectorizationCostModel {
/// scalarized.
DenseMap<ElementCount, SmallPtrSet<Instruction *, 4>> ForcedScalars;
- /// PHINodes of the reductions that should be expanded in-loop.
- SmallPtrSet<PHINode *, 4> InLoopReductions;
-
- /// A Map of inloop reduction operations and their immediate chain operand.
- /// FIXME: This can be removed once reductions can be costed correctly in
- /// VPlan. This was added to allow quick lookup of the inloop operations.
- DenseMap<Instruction *, Instruction *> InLoopReductionImmediateChains;
-
/// Returns the expected
diff erence in cost from scalarizing the expression
/// feeding a predicated instruction \p PredInst. The instructions to
/// scalarize and their scalar costs are collected in \p ScalarCosts. A
@@ -1736,19 +1528,6 @@ class LoopVectorizationCostModel {
/// Values to ignore in the cost model when VF > 1.
SmallPtrSet<const Value *, 16> VecValuesToIgnore;
-
- /// All element types found in the loop.
- SmallPtrSet<Type *, 16> ElementTypesInLoop;
-
- /// The kind of cost that we are calculating
- TTI::TargetCostKind CostKind;
-
- /// Whether this loop should be optimized for size based on function attribute
- /// or profile information.
- bool OptForSize;
-
- /// The highest VF possible for this loop, without using MaxBandwidth.
- FixedScalableVFPair MaxPermissibleVFWithoutMaxBW;
};
} // end namespace llvm
@@ -2214,17 +1993,6 @@ llvm::emitTransformedIndex(IRBuilderBase &B, Value *Index, Value *StartValue,
llvm_unreachable("invalid enum");
}
-static std::optional<unsigned> getMaxVScale(const Function &F,
- const TargetTransformInfo &TTI) {
- if (std::optional<unsigned> MaxVScale = TTI.getMaxVScale())
- return MaxVScale;
-
- if (F.hasFnAttribute(Attribute::VScaleRange))
- return F.getFnAttribute(Attribute::VScaleRange).getVScaleRangeMax();
-
- return std::nullopt;
-}
-
/// For the given VF and UF and maximum trip count computed for the loop, return
/// whether the induction variable might overflow in the vectorized loop. If not,
/// then we know a runtime overflow check always evaluates to false and can be
@@ -2419,8 +2187,8 @@ LoopVectorizationCostModel::getVectorCallCost(CallInst *CI,
for (auto &ArgOp : CI->args())
Tys.push_back(ArgOp->getType());
- InstructionCost ScalarCallCost =
- TTI.getCallInstrCost(CI->getCalledFunction(), RetTy, Tys, CostKind);
+ InstructionCost ScalarCallCost = TTI.getCallInstrCost(
+ CI->getCalledFunction(), RetTy, Tys, Config.CostKind);
// If this is an intrinsic we may have a lower cost for it.
if (getVectorIntrinsicIDForCall(CI, TLI)) {
@@ -2456,7 +2224,7 @@ LoopVectorizationCostModel::getVectorIntrinsicCost(CallInst *CI,
IntrinsicCostAttributes CostAttrs(ID, RetTy, Arguments, ParamTys, FMF,
dyn_cast<IntrinsicInst>(CI),
InstructionCost::getInvalid());
- return TTI.getIntrinsicInstrCost(CostAttrs, CostKind);
+ return TTI.getIntrinsicInstrCost(CostAttrs, Config.CostKind);
}
void InnerLoopVectorizer::fixVectorizedLoop(VPTransformState &State) {
@@ -2704,10 +2472,11 @@ bool LoopVectorizationCostModel::isScalarWithPredication(Instruction *I,
if (VF.isVector())
VTy = VectorType::get(Ty, VF);
const Align Alignment = getLoadStoreAlignment(I);
- return isa<LoadInst>(I) ? !(isLegalMaskedLoad(Ty, Ptr, Alignment, AS) ||
- TTI.isLegalMaskedGather(VTy, Alignment))
- : !(isLegalMaskedStore(Ty, Ptr, Alignment, AS) ||
- TTI.isLegalMaskedScatter(VTy, Alignment));
+ return isa<LoadInst>(I)
+ ? !(Config.isLegalMaskedLoad(Ty, Ptr, Alignment, AS) ||
+ TTI.isLegalMaskedGather(VTy, Alignment))
+ : !(Config.isLegalMaskedStore(Ty, Ptr, Alignment, AS) ||
+ TTI.isLegalMaskedScatter(VTy, Alignment));
}
case Instruction::UDiv:
case Instruction::SDiv:
@@ -2820,13 +2589,13 @@ LoopVectorizationCostModel::getDivRemSpeculationCost(Instruction *I,
// that we will create. This cost is likely to be zero. The phi node
// cost, if any, should be scaled by the block probability because it
// models a copy at the end of each predicated block.
- ScalarizationCost +=
- VF.getFixedValue() * TTI.getCFInstrCost(Instruction::PHI, CostKind);
+ ScalarizationCost += VF.getFixedValue() *
+ TTI.getCFInstrCost(Instruction::PHI, Config.CostKind);
// The cost of the non-predicated instruction.
ScalarizationCost +=
- VF.getFixedValue() *
- TTI.getArithmeticInstrCost(I->getOpcode(), I->getType(), CostKind);
+ VF.getFixedValue() * TTI.getArithmeticInstrCost(
+ I->getOpcode(), I->getType(), Config.CostKind);
// The cost of insertelement and extractelement instructions needed for
// scalarization.
@@ -2836,7 +2605,8 @@ LoopVectorizationCostModel::getDivRemSpeculationCost(Instruction *I,
// This assumes the predicated block for each vector lane is equally
// likely.
ScalarizationCost =
- ScalarizationCost / getPredBlockCostDivisor(CostKind, I->getParent());
+ ScalarizationCost /
+ getPredBlockCostDivisor(Config.CostKind, I->getParent());
}
InstructionCost SafeDivisorCost = 0;
@@ -2846,11 +2616,11 @@ LoopVectorizationCostModel::getDivRemSpeculationCost(Instruction *I,
SafeDivisorCost +=
TTI.getCmpSelInstrCost(Instruction::Select, VecTy,
toVectorTy(Type::getInt1Ty(I->getContext()), VF),
- CmpInst::BAD_ICMP_PREDICATE, CostKind);
+ CmpInst::BAD_ICMP_PREDICATE, Config.CostKind);
SmallVector<const Value *, 4> Operands(I->operand_values());
SafeDivisorCost += TTI.getArithmeticInstrCost(
- I->getOpcode(), VecTy, CostKind,
+ I->getOpcode(), VecTy, Config.CostKind,
{TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
{TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
Operands, I);
@@ -3202,232 +2972,6 @@ void LoopVectorizationCostModel::collectLoopUniforms(ElementCount VF) {
Uniforms[VF].insert_range(Worklist);
}
-bool LoopVectorizationCostModel::runtimeChecksRequired() {
- LLVM_DEBUG(dbgs() << "LV: Performing code size checks.\n");
-
- if (Legal->getRuntimePointerChecking()->Need) {
- reportVectorizationFailure("Runtime ptr check is required with -Os/-Oz",
- "runtime pointer checks needed. Enable vectorization of this "
- "loop with '#pragma clang loop vectorize(enable)' when "
- "compiling with -Os/-Oz",
- "CantVersionLoopWithOptForSize", ORE, TheLoop);
- return true;
- }
-
- if (!PSE.getPredicate().isAlwaysTrue()) {
- reportVectorizationFailure("Runtime SCEV check is required with -Os/-Oz",
- "runtime SCEV checks needed. Enable vectorization of this "
- "loop with '#pragma clang loop vectorize(enable)' when "
- "compiling with -Os/-Oz",
- "CantVersionLoopWithOptForSize", ORE, TheLoop);
- return true;
- }
-
- // FIXME: Avoid specializing for stride==1 instead of bailing out.
- if (!Legal->getLAI()->getSymbolicStrides().empty()) {
- reportVectorizationFailure("Runtime stride check for small trip count",
- "runtime stride == 1 checks needed. Enable vectorization of "
- "this loop without such check by compiling with -Os/-Oz",
- "CantVersionLoopWithOptForSize", ORE, TheLoop);
- return true;
- }
-
- return false;
-}
-
-bool LoopVectorizationCostModel::isScalableVectorizationAllowed() {
- if (IsScalableVectorizationAllowed)
- return *IsScalableVectorizationAllowed;
-
- IsScalableVectorizationAllowed = false;
- if (!TTI.supportsScalableVectors() && !ForceTargetSupportsScalableVectors)
- return false;
-
- if (Hints->isScalableVectorizationDisabled()) {
- reportVectorizationInfo("Scalable vectorization is explicitly disabled",
- "ScalableVectorizationDisabled", ORE, TheLoop);
- return false;
- }
-
- LLVM_DEBUG(dbgs() << "LV: Scalable vectorization is available\n");
-
- auto MaxScalableVF = ElementCount::getScalable(
- std::numeric_limits<ElementCount::ScalarTy>::max());
-
- // Test that the loop-vectorizer can legalize all operations for this MaxVF.
- // FIXME: While for scalable vectors this is currently sufficient, this should
- // be replaced by a more detailed mechanism that filters out specific VFs,
- // instead of invalidating vectorization for a whole set of VFs based on the
- // MaxVF.
-
- // Disable scalable vectorization if the loop contains unsupported reductions.
- if (!canVectorizeReductions(MaxScalableVF)) {
- reportVectorizationInfo(
- "Scalable vectorization not supported for the reduction "
- "operations found in this loop.",
- "ScalableVFUnfeasible", ORE, TheLoop);
- return false;
- }
-
- // Disable scalable vectorization if the loop contains any instructions
- // with element types not supported for scalable vectors.
- if (any_of(ElementTypesInLoop, [&](Type *Ty) {
- return !Ty->isVoidTy() &&
- !this->TTI.isElementTypeLegalForScalableVector(Ty);
- })) {
- reportVectorizationInfo("Scalable vectorization is not supported "
- "for all element types found in this loop.",
- "ScalableVFUnfeasible", ORE, TheLoop);
- return false;
- }
-
- if (!Legal->isSafeForAnyVectorWidth() && !getMaxVScale(*TheFunction, TTI)) {
- reportVectorizationInfo("The target does not provide maximum vscale value "
- "for safe distance analysis.",
- "ScalableVFUnfeasible", ORE, TheLoop);
- return false;
- }
-
- IsScalableVectorizationAllowed = true;
- return true;
-}
-
-ElementCount
-LoopVectorizationCostModel::getMaxLegalScalableVF(unsigned MaxSafeElements) {
- if (!isScalableVectorizationAllowed())
- return ElementCount::getScalable(0);
-
- auto MaxScalableVF = ElementCount::getScalable(
- std::numeric_limits<ElementCount::ScalarTy>::max());
- if (Legal->isSafeForAnyVectorWidth())
- return MaxScalableVF;
-
- std::optional<unsigned> MaxVScale = getMaxVScale(*TheFunction, TTI);
- // Limit MaxScalableVF by the maximum safe dependence distance.
- MaxScalableVF = ElementCount::getScalable(MaxSafeElements / *MaxVScale);
-
- if (!MaxScalableVF)
- reportVectorizationInfo(
- "Max legal vector width too small, scalable vectorization "
- "unfeasible.",
- "ScalableVFUnfeasible", ORE, TheLoop);
-
- return MaxScalableVF;
-}
-
-FixedScalableVFPair LoopVectorizationCostModel::computeFeasibleMaxVF(
- unsigned MaxTripCount, ElementCount UserVF, unsigned UserIC,
- bool FoldTailByMasking) {
- MinBWs = computeMinimumValueSizes(TheLoop->getBlocks(), *DB, &TTI);
- unsigned SmallestType, WidestType;
- std::tie(SmallestType, WidestType) = getSmallestAndWidestTypes();
-
- // Get the maximum safe dependence distance in bits computed by LAA.
- // It is computed by MaxVF * sizeOf(type) * 8, where type is taken from
- // the memory accesses that is most restrictive (involved in the smallest
- // dependence distance).
- unsigned MaxSafeElementsPowerOf2 =
- bit_floor(Legal->getMaxSafeVectorWidthInBits() / WidestType);
- if (!Legal->isSafeForAnyStoreLoadForwardDistances()) {
- unsigned SLDist = Legal->getMaxStoreLoadForwardSafeDistanceInBits();
- MaxSafeElementsPowerOf2 =
- std::min(MaxSafeElementsPowerOf2, SLDist / WidestType);
- }
- auto MaxSafeFixedVF = ElementCount::getFixed(MaxSafeElementsPowerOf2);
- auto MaxSafeScalableVF = getMaxLegalScalableVF(MaxSafeElementsPowerOf2);
-
- if (!Legal->isSafeForAnyVectorWidth())
- this->MaxSafeElements = MaxSafeElementsPowerOf2;
-
- LLVM_DEBUG(dbgs() << "LV: The max safe fixed VF is: " << MaxSafeFixedVF
- << ".\n");
- LLVM_DEBUG(dbgs() << "LV: The max safe scalable VF is: " << MaxSafeScalableVF
- << ".\n");
-
- // First analyze the UserVF, fall back if the UserVF should be ignored.
- if (UserVF) {
- auto MaxSafeUserVF =
- UserVF.isScalable() ? MaxSafeScalableVF : MaxSafeFixedVF;
-
- if (ElementCount::isKnownLE(UserVF, MaxSafeUserVF)) {
- // If `VF=vscale x N` is safe, then so is `VF=N`
- if (UserVF.isScalable())
- return FixedScalableVFPair(
- ElementCount::getFixed(UserVF.getKnownMinValue()), UserVF);
-
- return UserVF;
- }
-
- assert(ElementCount::isKnownGT(UserVF, MaxSafeUserVF));
-
- // Only clamp if the UserVF is not scalable. If the UserVF is scalable, it
- // is better to ignore the hint and let the compiler choose a suitable VF.
- if (!UserVF.isScalable()) {
- LLVM_DEBUG(dbgs() << "LV: User VF=" << UserVF
- << " is unsafe, clamping to max safe VF="
- << MaxSafeFixedVF << ".\n");
- ORE->emit([&]() {
- return OptimizationRemarkAnalysis(DEBUG_TYPE, "VectorizationFactor",
- TheLoop->getStartLoc(),
- TheLoop->getHeader())
- << "User-specified vectorization factor "
- << ore::NV("UserVectorizationFactor", UserVF)
- << " is unsafe, clamping to maximum safe vectorization factor "
- << ore::NV("VectorizationFactor", MaxSafeFixedVF);
- });
- return MaxSafeFixedVF;
- }
-
- if (!TTI.supportsScalableVectors() && !ForceTargetSupportsScalableVectors) {
- LLVM_DEBUG(dbgs() << "LV: User VF=" << UserVF
- << " is ignored because scalable vectors are not "
- "available.\n");
- ORE->emit([&]() {
- return OptimizationRemarkAnalysis(DEBUG_TYPE, "VectorizationFactor",
- TheLoop->getStartLoc(),
- TheLoop->getHeader())
- << "User-specified vectorization factor "
- << ore::NV("UserVectorizationFactor", UserVF)
- << " is ignored because the target does not support scalable "
- "vectors. The compiler will pick a more suitable value.";
- });
- } else {
- LLVM_DEBUG(dbgs() << "LV: User VF=" << UserVF
- << " is unsafe. Ignoring scalable UserVF.\n");
- ORE->emit([&]() {
- return OptimizationRemarkAnalysis(DEBUG_TYPE, "VectorizationFactor",
- TheLoop->getStartLoc(),
- TheLoop->getHeader())
- << "User-specified vectorization factor "
- << ore::NV("UserVectorizationFactor", UserVF)
- << " is unsafe. Ignoring the hint to let the compiler pick a "
- "more suitable value.";
- });
- }
- }
-
- LLVM_DEBUG(dbgs() << "LV: The Smallest and Widest types: " << SmallestType
- << " / " << WidestType << " bits.\n");
-
- FixedScalableVFPair Result(ElementCount::getFixed(1),
- ElementCount::getScalable(0));
- if (auto MaxVF =
- getMaximizedVFForTarget(MaxTripCount, SmallestType, WidestType,
- MaxSafeFixedVF, UserIC, FoldTailByMasking))
- Result.FixedVF = MaxVF;
-
- if (auto MaxVF =
- getMaximizedVFForTarget(MaxTripCount, SmallestType, WidestType,
- MaxSafeScalableVF, UserIC, FoldTailByMasking))
- if (MaxVF.isScalable()) {
- Result.ScalableVF = MaxVF;
- LLVM_DEBUG(dbgs() << "LV: Found feasible scalable VF = " << MaxVF
- << "\n");
- }
-
- return Result;
-}
-
FixedScalableVFPair
LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
if (Legal->getRuntimePointerChecking()->Need && TTI.hasBranchDivergence()) {
@@ -3471,9 +3015,16 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
return FixedScalableVFPair::getNone();
}
+ assert(WideningDecisions.empty() && CallWideningDecisions.empty() &&
+ Uniforms.empty() && Scalars.empty() &&
+ "No cost-modeling decisions should have been taken at this point");
+
+ MinBWs = computeMinimumValueSizes(TheLoop->getBlocks(), *DB, &TTI);
+
switch (EpilogueLoweringStatus) {
case CM_EpilogueAllowed:
- return computeFeasibleMaxVF(MaxTC, UserVF, UserIC, false);
+ return Config.computeFeasibleMaxVF(MaxTC, UserVF, UserIC, false,
+ requiresScalarEpilogue(true));
case CM_EpilogueNotAllowedFoldTail:
[[fallthrough]];
case CM_EpilogueNotNeededFoldTail:
@@ -3492,7 +3043,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// Bail if runtime checks are required, which are not good when optimising
// for size.
- if (runtimeChecksRequired())
+ if (Config.runtimeChecksRequired())
return FixedScalableVFPair::getNone();
break;
@@ -3503,15 +3054,13 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// Invalidate interleave groups that require an epilogue if we can't mask
// the interleave-group.
if (!useMaskedInterleavedAccesses(TTI)) {
- assert(WideningDecisions.empty() && Uniforms.empty() && Scalars.empty() &&
- "No decisions should have been taken at this point");
// Note: There is no need to invalidate any cost modeling decisions here, as
- // none were taken so far.
+ // none were taken so far (see assertion above).
InterleaveInfo.invalidateGroupsRequiringScalarEpilogue();
}
- FixedScalableVFPair MaxFactors =
- computeFeasibleMaxVF(MaxTC, UserVF, UserIC, true);
+ FixedScalableVFPair MaxFactors = Config.computeFeasibleMaxVF(
+ MaxTC, UserVF, UserIC, true, requiresScalarEpilogue(true));
// Avoid tail folding if the trip count is known to be a multiple of any VF
// we choose.
@@ -3637,148 +3186,6 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
return FixedScalableVFPair::getNone();
}
-bool LoopVectorizationCostModel::shouldConsiderRegPressureForVF(
- ElementCount VF) {
- if (ConsiderRegPressure.getNumOccurrences())
- return ConsiderRegPressure;
-
- // TODO: We should eventually consider register pressure for all targets. The
- // TTI hook is temporary whilst target-specific issues are being fixed.
- if (TTI.shouldConsiderVectorizationRegPressure())
- return true;
-
- if (!useMaxBandwidth(VF.isScalable()
- ? TargetTransformInfo::RGK_ScalableVector
- : TargetTransformInfo::RGK_FixedWidthVector))
- return false;
- // Only calculate register pressure for VFs enabled by MaxBandwidth.
- return ElementCount::isKnownGT(
- VF, VF.isScalable() ? MaxPermissibleVFWithoutMaxBW.ScalableVF
- : MaxPermissibleVFWithoutMaxBW.FixedVF);
-}
-
-bool LoopVectorizationCostModel::useMaxBandwidth(
- TargetTransformInfo::RegisterKind RegKind) {
- return MaximizeBandwidth || (MaximizeBandwidth.getNumOccurrences() == 0 &&
- (TTI.shouldMaximizeVectorBandwidth(RegKind) ||
- (UseWiderVFIfCallVariantsPresent &&
- Legal->hasVectorCallVariants())));
-}
-
-ElementCount LoopVectorizationCostModel::clampVFByMaxTripCount(
- ElementCount VF, unsigned MaxTripCount, unsigned UserIC,
- bool FoldTailByMasking) const {
- unsigned EstimatedVF = VF.getKnownMinValue();
- if (VF.isScalable() && TheFunction->hasFnAttribute(Attribute::VScaleRange)) {
- auto Attr = TheFunction->getFnAttribute(Attribute::VScaleRange);
- auto Min = Attr.getVScaleRangeMin();
- EstimatedVF *= Min;
- }
-
- // When a scalar epilogue is required, at least one iteration of the scalar
- // loop has to execute. Adjust MaxTripCount accordingly to avoid picking a
- // max VF that results in a dead vector loop.
- if (MaxTripCount > 0 && requiresScalarEpilogue(true))
- MaxTripCount -= 1;
-
- // When the user specifies an interleave count, we need to ensure that
- // VF * UserIC <= MaxTripCount to avoid a dead vector loop.
- unsigned IC = UserIC > 0 ? UserIC : 1;
- unsigned EstimatedVFTimesIC = EstimatedVF * IC;
-
- if (MaxTripCount && MaxTripCount <= EstimatedVFTimesIC &&
- (!FoldTailByMasking || isPowerOf2_32(MaxTripCount))) {
- // If upper bound loop trip count (TC) is known at compile time there is no
- // point in choosing VF greater than TC / IC (as done in the loop below).
- // Select maximum power of two which doesn't exceed TC / IC. If VF is
- // scalable, we only fall back on a fixed VF when the TC is less than or
- // equal to the known number of lanes.
- auto ClampedUpperTripCount = llvm::bit_floor(MaxTripCount / IC);
- if (ClampedUpperTripCount == 0)
- ClampedUpperTripCount = 1;
- LLVM_DEBUG(dbgs() << "LV: Clamping the MaxVF to maximum power of two not "
- "exceeding the constant trip count"
- << (UserIC > 0 ? " divided by UserIC" : "") << ": "
- << ClampedUpperTripCount << "\n");
- return ElementCount::get(ClampedUpperTripCount,
- FoldTailByMasking ? VF.isScalable() : false);
- }
- return VF;
-}
-
-ElementCount LoopVectorizationCostModel::getMaximizedVFForTarget(
- unsigned MaxTripCount, unsigned SmallestType, unsigned WidestType,
- ElementCount MaxSafeVF, unsigned UserIC, bool FoldTailByMasking) {
- bool ComputeScalableMaxVF = MaxSafeVF.isScalable();
- const TypeSize WidestRegister = TTI.getRegisterBitWidth(
- ComputeScalableMaxVF ? TargetTransformInfo::RGK_ScalableVector
- : TargetTransformInfo::RGK_FixedWidthVector);
-
- // Convenience function to return the minimum of two ElementCounts.
- auto MinVF = [](const ElementCount &LHS, const ElementCount &RHS) {
- assert((LHS.isScalable() == RHS.isScalable()) &&
- "Scalable flags must match");
- return ElementCount::isKnownLT(LHS, RHS) ? LHS : RHS;
- };
-
- // Ensure MaxVF is a power of 2; the dependence distance bound may not be.
- // Note that both WidestRegister and WidestType may not be a powers of 2.
- auto MaxVectorElementCount = ElementCount::get(
- llvm::bit_floor(WidestRegister.getKnownMinValue() / WidestType),
- ComputeScalableMaxVF);
- MaxVectorElementCount = MinVF(MaxVectorElementCount, MaxSafeVF);
- LLVM_DEBUG(dbgs() << "LV: The Widest register safe to use is: "
- << (MaxVectorElementCount * WidestType) << " bits.\n");
-
- if (!MaxVectorElementCount) {
- LLVM_DEBUG(dbgs() << "LV: The target has no "
- << (ComputeScalableMaxVF ? "scalable" : "fixed")
- << " vector registers.\n");
- return ElementCount::getFixed(1);
- }
-
- ElementCount MaxVF = clampVFByMaxTripCount(
- MaxVectorElementCount, MaxTripCount, UserIC, FoldTailByMasking);
- // If the MaxVF was already clamped, there's no point in trying to pick a
- // larger one.
- if (MaxVF != MaxVectorElementCount)
- return MaxVF;
-
- TargetTransformInfo::RegisterKind RegKind =
- ComputeScalableMaxVF ? TargetTransformInfo::RGK_ScalableVector
- : TargetTransformInfo::RGK_FixedWidthVector;
-
- if (MaxVF.isScalable())
- MaxPermissibleVFWithoutMaxBW.ScalableVF = MaxVF;
- else
- MaxPermissibleVFWithoutMaxBW.FixedVF = MaxVF;
-
- if (useMaxBandwidth(RegKind)) {
- auto MaxVectorElementCountMaxBW = ElementCount::get(
- llvm::bit_floor(WidestRegister.getKnownMinValue() / SmallestType),
- ComputeScalableMaxVF);
- MaxVF = MinVF(MaxVectorElementCountMaxBW, MaxSafeVF);
-
- if (ElementCount MinVF =
- TTI.getMinimumVF(SmallestType, ComputeScalableMaxVF)) {
- if (ElementCount::isKnownLT(MaxVF, MinVF)) {
- LLVM_DEBUG(dbgs() << "LV: Overriding calculated MaxVF(" << MaxVF
- << ") with target's minimum: " << MinVF << '\n');
- MaxVF = MinVF;
- }
- }
-
- MaxVF =
- clampVFByMaxTripCount(MaxVF, MaxTripCount, UserIC, FoldTailByMasking);
-
- assert((MaxVectorElementCount == MaxVF ||
- (WideningDecisions.empty() && CallWideningDecisions.empty() &&
- Uniforms.empty() && Scalars.empty())) &&
- "No decisions should have been taken at this point");
- }
- return MaxVF;
-}
-
bool LoopVectorizationPlanner::isMoreProfitable(const VectorizationFactor &A,
const VectorizationFactor &B,
const unsigned MaxTripCount,
@@ -3796,7 +3203,7 @@ bool LoopVectorizationPlanner::isMoreProfitable(const VectorizationFactor &A,
// Improve estimate for the vector width if it is scalable.
unsigned EstimatedWidthA = A.Width.getKnownMinValue();
unsigned EstimatedWidthB = B.Width.getKnownMinValue();
- if (std::optional<unsigned> VScale = CM.getVScaleForTuning()) {
+ if (std::optional<unsigned> VScale = Config.getVScaleForTuning()) {
if (A.Width.isScalable())
EstimatedWidthA *= *VScale;
if (B.Width.isScalable())
@@ -3806,7 +3213,7 @@ bool LoopVectorizationPlanner::isMoreProfitable(const VectorizationFactor &A,
// When optimizing for size choose whichever is smallest, which will be the
// one with the smallest cost for the whole loop. On a tie pick the larger
// vector width, on the assumption that throughput will be greater.
- if (CM.CostKind == TTI::TCK_CodeSize)
+ if (Config.CostKind == TTI::TCK_CodeSize)
return CostA < CostB ||
(CostA == CostB && EstimatedWidthA > EstimatedWidthB);
@@ -3882,7 +3289,7 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
if (VF.isScalar())
continue;
- VPCostContext CostCtx(CM.TTI, *CM.TLI, *Plan, CM, CM.CostKind, CM.PSE,
+ VPCostContext CostCtx(CM.TTI, *CM.TLI, *Plan, CM, Config.CostKind, CM.PSE,
OrigLoop);
precomputeCosts(*Plan, VF, CostCtx);
auto Iter = vp_depth_first_deep(Plan->getVectorLoopRegion()->getEntry());
@@ -4173,7 +3580,8 @@ bool LoopVectorizationCostModel::isEpilogueVectorizationProfitable(
unsigned MinVFThreshold = EpilogueVectorizationMinVF.getNumOccurrences() > 0
? EpilogueVectorizationMinVF
: TTI.getEpilogueVectorizationMinVF();
- return estimateElementCount(VF * IC, VScaleForTuning) >= MinVFThreshold;
+ return estimateElementCount(VF * IC, Config.getVScaleForTuning()) >=
+ MinVFThreshold;
}
std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
@@ -4199,7 +3607,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
if (EpilogueVectorizationForceVF > 1) {
if (EpilogueVectorizationForceVF >=
- IC * estimateElementCount(MainLoopVF, CM.getVScaleForTuning())) {
+ IC * estimateElementCount(MainLoopVF, Config.getVScaleForTuning())) {
// Note that the main loop leaves IC * MainLoopVF iterations iff a scalar
// epilogue is required, but then the epilogue loop also requires a scalar
// epilogue.
@@ -4255,7 +3663,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
// the main loop handles 8 lanes per iteration. We could still benefit from
// vectorizing the epilogue loop with VF=4.
ElementCount EstimatedRuntimeVF = ElementCount::getFixed(
- estimateElementCount(MainLoopVF, CM.getVScaleForTuning()));
+ estimateElementCount(MainLoopVF, Config.getVScaleForTuning()));
Type *TCType = Legal->getWidestInductionType();
const SCEV *RemainingIterations = nullptr;
@@ -4274,7 +3682,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
} else if (ScalableTC) {
const SCEV *EstimatedTC = SE.getMulExpr(
KnownMinTC,
- SE.getConstant(TCType, CM.getVScaleForTuning().value_or(1)));
+ SE.getConstant(TCType, Config.getVScaleForTuning().value_or(1)));
RemainingIterations = SE.getURemExpr(
EstimatedTC, SE.getElementCount(TCType, MainLoopVF * IC));
} else
@@ -4329,7 +3737,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
//
diff erent numerical spaces.
if (EffectiveVF.isScalable())
EffectiveVF = ElementCount::getFixed(
- estimateElementCount(EffectiveVF, CM.getVScaleForTuning()));
+ estimateElementCount(EffectiveVF, Config.getVScaleForTuning()));
if (SkipVF(SE.getElementCount(TCType, EffectiveVF), RemainingIterations))
continue;
}
@@ -4352,79 +3760,6 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
return Clone;
}
-std::pair<unsigned, unsigned>
-LoopVectorizationCostModel::getSmallestAndWidestTypes() {
- unsigned MinWidth = -1U;
- unsigned MaxWidth = 8;
- const DataLayout &DL = TheFunction->getDataLayout();
- // For in-loop reductions, no element types are added to ElementTypesInLoop
- // if there are no loads/stores in the loop. In this case, check through the
- // reduction variables to determine the maximum width.
- if (ElementTypesInLoop.empty() && !Legal->getReductionVars().empty()) {
- for (const auto &PhiDescriptorPair : Legal->getReductionVars()) {
- const RecurrenceDescriptor &RdxDesc = PhiDescriptorPair.second;
- // When finding the min width used by the recurrence we need to account
- // for casts on the input operands of the recurrence.
- MinWidth = std::min(
- MinWidth,
- std::min(RdxDesc.getMinWidthCastToRecurrenceTypeInBits(),
- RdxDesc.getRecurrenceType()->getScalarSizeInBits()));
- MaxWidth = std::max(MaxWidth,
- RdxDesc.getRecurrenceType()->getScalarSizeInBits());
- }
- } else {
- for (Type *T : ElementTypesInLoop) {
- MinWidth = std::min<unsigned>(
- MinWidth, DL.getTypeSizeInBits(T->getScalarType()).getFixedValue());
- MaxWidth = std::max<unsigned>(
- MaxWidth, DL.getTypeSizeInBits(T->getScalarType()).getFixedValue());
- }
- }
- return {MinWidth, MaxWidth};
-}
-
-void LoopVectorizationCostModel::collectElementTypesForWidening() {
- ElementTypesInLoop.clear();
- // For each block.
- for (BasicBlock *BB : TheLoop->blocks()) {
- // For each instruction in the loop.
- for (Instruction &I : *BB) {
- Type *T = I.getType();
-
- // Skip ignored values.
- if (ValuesToIgnore.count(&I))
- continue;
-
- // Only examine Loads, Stores and PHINodes.
- if (!isa<LoadInst>(I) && !isa<StoreInst>(I) && !isa<PHINode>(I))
- continue;
-
- // Examine PHI nodes that are reduction variables. Update the type to
- // account for the recurrence type.
- if (auto *PN = dyn_cast<PHINode>(&I)) {
- if (!Legal->isReductionVariable(PN))
- continue;
- const RecurrenceDescriptor &RdxDesc =
- Legal->getRecurrenceDescriptor(PN);
- if (PreferInLoopReductions || useOrderedReductions(RdxDesc) ||
- TTI.preferInLoopReduction(RdxDesc.getRecurrenceKind(),
- RdxDesc.getRecurrenceType()))
- continue;
- T = RdxDesc.getRecurrenceType();
- }
-
- // Examine the stored values.
- if (auto *ST = dyn_cast<StoreInst>(&I))
- T = ST->getValueOperand()->getType();
-
- assert(T->isSized() &&
- "Expected the load/store/recurrence type to be sized");
-
- ElementTypesInLoop.insert(T);
- }
- }
-}
-
unsigned
LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
InstructionCost LoopCost) {
@@ -4565,8 +3900,9 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
if (BestKnownTC && (BestKnownTC->isFixed() || VF.isScalable())) {
// Re-evaluate trip counts and VFs to be in the same numerical space.
unsigned AvailableTC =
- estimateElementCount(*BestKnownTC, CM.getVScaleForTuning());
- unsigned EstimatedVF = estimateElementCount(VF, CM.getVScaleForTuning());
+ estimateElementCount(*BestKnownTC, Config.getVScaleForTuning());
+ unsigned EstimatedVF =
+ estimateElementCount(VF, Config.getVScaleForTuning());
// At least one iteration must be scalar when this constraint holds. So the
// maximum available iterations for interleaving is one less.
@@ -4926,10 +4262,10 @@ InstructionCost LoopVectorizationCostModel::computePredInstDiscount(
ScalarCost += TTI.getScalarizationOverhead(
cast<VectorType>(VectorTy), APInt::getAllOnes(VF.getFixedValue()),
/*Insert=*/true,
- /*Extract=*/false, CostKind);
+ /*Extract=*/false, Config.CostKind);
}
- ScalarCost +=
- VF.getFixedValue() * TTI.getCFInstrCost(Instruction::PHI, CostKind);
+ ScalarCost += VF.getFixedValue() *
+ TTI.getCFInstrCost(Instruction::PHI, Config.CostKind);
}
// Compute the scalarization overhead of needed extractelement
@@ -4948,13 +4284,13 @@ InstructionCost LoopVectorizationCostModel::computePredInstDiscount(
ScalarCost += TTI.getScalarizationOverhead(
cast<VectorType>(VectorTy),
APInt::getAllOnes(VF.getFixedValue()), /*Insert*/ false,
- /*Extract*/ true, CostKind);
+ /*Extract*/ true, Config.CostKind);
}
}
}
// Scale the total scalar cost by block probability.
- ScalarCost /= getPredBlockCostDivisor(CostKind, I->getParent());
+ ScalarCost /= getPredBlockCostDivisor(Config.CostKind, I->getParent());
// Compute the discount. A non-negative discount means the vector version
// of the instruction costs more, and scalarizing would be beneficial.
@@ -4995,7 +4331,7 @@ InstructionCost LoopVectorizationCostModel::expectedCost(ElementCount VF) {
// is an if-else block. Thus, scale the block's cost by the probability of
// executing it. getPredBlockCostDivisor will return 1 for blocks that are
// only predicated by the header mask when folding the tail.
- Cost += BlockCost / getPredBlockCostDivisor(CostKind, BB);
+ Cost += BlockCost / getPredBlockCostDivisor(Config.CostKind, BB);
}
return Cost;
@@ -5037,8 +4373,9 @@ LoopVectorizationCostModel::getMemInstScalarizationCost(Instruction *I,
const SCEV *PtrSCEV = getAddressAccessSCEV(Ptr, PSE, TheLoop);
// Get the cost of the scalar memory instruction and address computation.
- InstructionCost Cost = VF.getFixedValue() * TTI.getAddressComputationCost(
- PtrTy, SE, PtrSCEV, CostKind);
+ InstructionCost Cost =
+ VF.getFixedValue() *
+ TTI.getAddressComputationCost(PtrTy, SE, PtrSCEV, Config.CostKind);
// Don't pass *I here, since it is scalar but will actually be part of a
// vectorized loop where the user of it is a vectorized instruction.
@@ -5046,7 +4383,7 @@ LoopVectorizationCostModel::getMemInstScalarizationCost(Instruction *I,
TTI::OperandValueInfo OpInfo = TTI::getOperandInfo(I->getOperand(0));
Cost += VF.getFixedValue() *
TTI.getMemoryOpCost(I->getOpcode(), ValTy->getScalarType(), Alignment,
- AS, CostKind, OpInfo);
+ AS, Config.CostKind, OpInfo);
// Get the overhead of the extractelement and insertelement instructions
// we might create due to scalarization.
@@ -5056,15 +4393,15 @@ LoopVectorizationCostModel::getMemInstScalarizationCost(Instruction *I,
// conditional branches, but may not be executed for each vector lane. Scale
// the cost by the probability of executing the predicated block.
if (isPredicatedInst(I)) {
- Cost /= getPredBlockCostDivisor(CostKind, I->getParent());
+ Cost /= getPredBlockCostDivisor(Config.CostKind, I->getParent());
// Add the cost of an i1 extract and a branch
auto *VecI1Ty =
VectorType::get(IntegerType::getInt1Ty(ValTy->getContext()), VF);
Cost += TTI.getScalarizationOverhead(
VecI1Ty, APInt::getAllOnes(VF.getFixedValue()),
- /*Insert=*/false, /*Extract=*/true, CostKind);
- Cost += TTI.getCFInstrCost(Instruction::CondBr, CostKind);
+ /*Insert=*/false, /*Extract=*/true, Config.CostKind);
+ Cost += TTI.getCFInstrCost(Instruction::CondBr, Config.CostKind);
if (useEmulatedMaskMemRefHack(I, VF))
// Artificially setting to a high enough value to practically disable
@@ -5093,17 +4430,18 @@ LoopVectorizationCostModel::getConsecutiveMemOpCost(Instruction *I,
? Intrinsic::masked_load
: Intrinsic::masked_store;
Cost += TTI.getMemIntrinsicInstrCost(
- MemIntrinsicCostAttributes(IID, VectorTy, Alignment, AS), CostKind);
+ MemIntrinsicCostAttributes(IID, VectorTy, Alignment, AS),
+ Config.CostKind);
} else {
TTI::OperandValueInfo OpInfo = TTI::getOperandInfo(I->getOperand(0));
Cost += TTI.getMemoryOpCost(I->getOpcode(), VectorTy, Alignment, AS,
- CostKind, OpInfo, I);
+ Config.CostKind, OpInfo, I);
}
bool Reverse = ConsecutiveStride < 0;
if (Reverse)
Cost += TTI.getShuffleCost(TargetTransformInfo::SK_Reverse, VectorTy,
- VectorTy, {}, CostKind, 0);
+ VectorTy, {}, Config.CostKind, 0);
return Cost;
}
@@ -5118,11 +4456,12 @@ LoopVectorizationCostModel::getUniformMemOpCost(Instruction *I,
const Align Alignment = getLoadStoreAlignment(I);
unsigned AS = getLoadStoreAddressSpace(I);
if (isa<LoadInst>(I)) {
- return TTI.getAddressComputationCost(PtrTy, nullptr, nullptr, CostKind) +
+ return TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
+ Config.CostKind) +
TTI.getMemoryOpCost(Instruction::Load, ValTy, Alignment, AS,
- CostKind) +
+ Config.CostKind) +
TTI.getShuffleCost(TargetTransformInfo::SK_Broadcast, VectorTy,
- VectorTy, {}, CostKind);
+ VectorTy, {}, Config.CostKind);
}
StoreInst *SI = cast<StoreInst>(I);
@@ -5132,11 +4471,12 @@ LoopVectorizationCostModel::getUniformMemOpCost(Instruction *I,
// the actual generated code, which involves extracting the last element of
// a scalable vector where the lane to extract is unknown at compile time.
InstructionCost Cost =
- TTI.getAddressComputationCost(PtrTy, nullptr, nullptr, CostKind) +
- TTI.getMemoryOpCost(Instruction::Store, ValTy, Alignment, AS, CostKind);
+ TTI.getAddressComputationCost(PtrTy, nullptr, nullptr, Config.CostKind) +
+ TTI.getMemoryOpCost(Instruction::Store, ValTy, Alignment, AS,
+ Config.CostKind);
if (!IsLoopInvariantStoreValue)
Cost += TTI.getIndexedVectorInstrCostFromEnd(Instruction::ExtractElement,
- VectorTy, CostKind, 0);
+ VectorTy, Config.CostKind, 0);
return Cost;
}
@@ -5155,11 +4495,12 @@ LoopVectorizationCostModel::getGatherScatterCost(Instruction *I,
unsigned IID = I->getOpcode() == Instruction::Load
? Intrinsic::masked_gather
: Intrinsic::masked_scatter;
- return TTI.getAddressComputationCost(PtrTy, nullptr, nullptr, CostKind) +
+ return TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
+ Config.CostKind) +
TTI.getMemIntrinsicInstrCost(
MemIntrinsicCostAttributes(IID, VectorTy, Ptr, isMaskRequired(I),
Alignment, I),
- CostKind);
+ Config.CostKind);
}
InstructionCost
@@ -5188,7 +4529,8 @@ LoopVectorizationCostModel::getInterleaveGroupCost(Instruction *I,
(isa<StoreInst>(I) && !Group->isFull());
InstructionCost Cost = TTI.getInterleavedMemoryOpCost(
InsertPos->getOpcode(), WideVecTy, Group->getFactor(), Indices,
- Group->getAlign(), AS, CostKind, isMaskRequired(I), UseMaskForGaps);
+ Group->getAlign(), AS, Config.CostKind, isMaskRequired(I),
+ UseMaskForGaps);
if (Group->isReverse()) {
// TODO: Add support for reversed masked interleaved access.
@@ -5196,7 +4538,7 @@ LoopVectorizationCostModel::getInterleaveGroupCost(Instruction *I,
"Reverse masked interleaved access not supported.");
Cost += Group->getNumMembers() *
TTI.getShuffleCost(TargetTransformInfo::SK_Reverse, VectorTy,
- VectorTy, {}, CostKind, 0);
+ VectorTy, {}, Config.CostKind, 0);
}
return Cost;
}
@@ -5207,7 +4549,8 @@ LoopVectorizationCostModel::getReductionPatternCost(Instruction *I,
Type *Ty) const {
using namespace llvm::PatternMatch;
// Early exit for no inloop reductions
- if (InLoopReductions.empty() || VF.isScalar() || !isa<VectorType>(Ty))
+ if (Config.getInLoopReductions().empty() || VF.isScalar() ||
+ !isa<VectorType>(Ty))
return std::nullopt;
auto *VectorTy = cast<VectorType>(Ty);
@@ -5237,7 +4580,7 @@ LoopVectorizationCostModel::getReductionPatternCost(Instruction *I,
// Test if the found instruction is a reduction, and if not return an invalid
// cost specifying the parent to use the original cost modelling.
- Instruction *LastChain = InLoopReductionImmediateChains.lookup(RetI);
+ Instruction *LastChain = Config.getInLoopReductionImmediateChain(RetI);
if (!LastChain)
return std::nullopt;
@@ -5245,7 +4588,7 @@ LoopVectorizationCostModel::getReductionPatternCost(Instruction *I,
// the reduction on its own.
Instruction *ReductionPhi = LastChain;
while (!isa<PHINode>(ReductionPhi))
- ReductionPhi = InLoopReductionImmediateChains.at(ReductionPhi);
+ ReductionPhi = Config.getInLoopReductionImmediateChain(ReductionPhi);
const RecurrenceDescriptor &RdxDesc =
Legal->getRecurrenceDescriptor(cast<PHINode>(ReductionPhi));
@@ -5254,23 +4597,24 @@ LoopVectorizationCostModel::getReductionPatternCost(Instruction *I,
RecurKind RK = RdxDesc.getRecurrenceKind();
if (RecurrenceDescriptor::isMinMaxRecurrenceKind(RK)) {
Intrinsic::ID MinMaxID = getMinMaxReductionIntrinsicOp(RK);
- BaseCost = TTI.getMinMaxReductionCost(MinMaxID, VectorTy,
- RdxDesc.getFastMathFlags(), CostKind);
+ BaseCost = TTI.getMinMaxReductionCost(
+ MinMaxID, VectorTy, RdxDesc.getFastMathFlags(), Config.CostKind);
} else {
- BaseCost = TTI.getArithmeticReductionCost(
- RdxDesc.getOpcode(), VectorTy, RdxDesc.getFastMathFlags(), CostKind);
+ BaseCost = TTI.getArithmeticReductionCost(RdxDesc.getOpcode(), VectorTy,
+ RdxDesc.getFastMathFlags(),
+ Config.CostKind);
}
// For a call to the llvm.fmuladd intrinsic we need to add the cost of a
// normal fmul instruction to the cost of the fadd reduction.
if (RK == RecurKind::FMulAdd)
- BaseCost +=
- TTI.getArithmeticInstrCost(Instruction::FMul, VectorTy, CostKind);
+ BaseCost += TTI.getArithmeticInstrCost(Instruction::FMul, VectorTy,
+ Config.CostKind);
// If we're using ordered reductions then we can just return the base cost
// here, since getArithmeticReductionCost calculates the full ordered
// reduction cost when FP reassociation is not allowed.
- if (useOrderedReductions(RdxDesc))
+ if (Config.useOrderedReductions(RdxDesc))
return BaseCost;
// Get the operand that was not the reduction chain and match it to one of the
@@ -5301,16 +4645,16 @@ LoopVectorizationCostModel::getReductionPatternCost(Instruction *I,
InstructionCost ExtCost =
TTI.getCastInstrCost(Op0->getOpcode(), MulType, ExtType,
- TTI::CastContextHint::None, CostKind, Op0);
+ TTI::CastContextHint::None, Config.CostKind, Op0);
InstructionCost MulCost =
- TTI.getArithmeticInstrCost(Instruction::Mul, MulType, CostKind);
- InstructionCost Ext2Cost =
- TTI.getCastInstrCost(RedOp->getOpcode(), VectorTy, MulType,
- TTI::CastContextHint::None, CostKind, RedOp);
+ TTI.getArithmeticInstrCost(Instruction::Mul, MulType, Config.CostKind);
+ InstructionCost Ext2Cost = TTI.getCastInstrCost(
+ RedOp->getOpcode(), VectorTy, MulType, TTI::CastContextHint::None,
+ Config.CostKind, RedOp);
InstructionCost RedCost = TTI.getMulAccReductionCost(
IsUnsigned, RdxDesc.getOpcode(), RdxDesc.getRecurrenceType(), ExtType,
- CostKind);
+ Config.CostKind);
if (RedCost.isValid() &&
RedCost < ExtCost * 2 + MulCost + Ext2Cost + BaseCost)
@@ -5322,11 +4666,11 @@ LoopVectorizationCostModel::getReductionPatternCost(Instruction *I,
auto *ExtType = VectorType::get(RedOp->getOperand(0)->getType(), VectorTy);
InstructionCost RedCost = TTI.getExtendedReductionCost(
RdxDesc.getOpcode(), IsUnsigned, RdxDesc.getRecurrenceType(), ExtType,
- RdxDesc.getFastMathFlags(), CostKind);
+ RdxDesc.getFastMathFlags(), Config.CostKind);
- InstructionCost ExtCost =
- TTI.getCastInstrCost(RedOp->getOpcode(), VectorTy, ExtType,
- TTI::CastContextHint::None, CostKind, RedOp);
+ InstructionCost ExtCost = TTI.getCastInstrCost(
+ RedOp->getOpcode(), VectorTy, ExtType, TTI::CastContextHint::None,
+ Config.CostKind, RedOp);
if (RedCost.isValid() && RedCost < BaseCost + ExtCost)
return I == RetI ? RedCost : 0;
} else if (RedOp && RdxDesc.getOpcode() == Instruction::Add &&
@@ -5347,23 +4691,23 @@ LoopVectorizationCostModel::getReductionPatternCost(Instruction *I,
// the remaining cost as, for example reduce(mul(ext(ext(A)), ext(B))).
InstructionCost ExtCost0 = TTI.getCastInstrCost(
Op0->getOpcode(), VectorTy, VectorType::get(Op0Ty, VectorTy),
- TTI::CastContextHint::None, CostKind, Op0);
+ TTI::CastContextHint::None, Config.CostKind, Op0);
InstructionCost ExtCost1 = TTI.getCastInstrCost(
Op1->getOpcode(), VectorTy, VectorType::get(Op1Ty, VectorTy),
- TTI::CastContextHint::None, CostKind, Op1);
- InstructionCost MulCost =
- TTI.getArithmeticInstrCost(Instruction::Mul, VectorTy, CostKind);
+ TTI::CastContextHint::None, Config.CostKind, Op1);
+ InstructionCost MulCost = TTI.getArithmeticInstrCost(
+ Instruction::Mul, VectorTy, Config.CostKind);
InstructionCost RedCost = TTI.getMulAccReductionCost(
IsUnsigned, RdxDesc.getOpcode(), RdxDesc.getRecurrenceType(), ExtType,
- CostKind);
+ Config.CostKind);
InstructionCost ExtraExtCost = 0;
if (Op0Ty != LargestOpTy || Op1Ty != LargestOpTy) {
Instruction *ExtraExtOp = (Op0Ty != LargestOpTy) ? Op0 : Op1;
ExtraExtCost = TTI.getCastInstrCost(
ExtraExtOp->getOpcode(), ExtType,
VectorType::get(ExtraExtOp->getOperand(0)->getType(), VectorTy),
- TTI::CastContextHint::None, CostKind, ExtraExtOp);
+ TTI::CastContextHint::None, Config.CostKind, ExtraExtOp);
}
if (RedCost.isValid() &&
@@ -5371,12 +4715,12 @@ LoopVectorizationCostModel::getReductionPatternCost(Instruction *I,
return I == RetI ? RedCost : 0;
} else if (!match(I, m_ZExtOrSExt(m_Value()))) {
// Matched reduce.add(mul())
- InstructionCost MulCost =
- TTI.getArithmeticInstrCost(Instruction::Mul, VectorTy, CostKind);
+ InstructionCost MulCost = TTI.getArithmeticInstrCost(
+ Instruction::Mul, VectorTy, Config.CostKind);
InstructionCost RedCost = TTI.getMulAccReductionCost(
true, RdxDesc.getOpcode(), RdxDesc.getRecurrenceType(), VectorTy,
- CostKind);
+ Config.CostKind);
if (RedCost.isValid() && RedCost < MulCost + BaseCost)
return I == RetI ? RedCost : 0;
@@ -5398,9 +4742,10 @@ LoopVectorizationCostModel::getMemoryInstructionCost(Instruction *I,
unsigned AS = getLoadStoreAddressSpace(I);
TTI::OperandValueInfo OpInfo = TTI::getOperandInfo(I->getOperand(0));
- return TTI.getAddressComputationCost(PtrTy, nullptr, nullptr, CostKind) +
- TTI.getMemoryOpCost(I->getOpcode(), ValTy, Alignment, AS, CostKind,
- OpInfo, I);
+ return TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
+ Config.CostKind) +
+ TTI.getMemoryOpCost(I->getOpcode(), ValTy, Alignment, AS,
+ Config.CostKind, OpInfo, I);
}
return getWideningCost(I, VF);
}
@@ -5431,7 +4776,7 @@ LoopVectorizationCostModel::getScalarizationOverhead(Instruction *I,
for (Type *VectorTy : getContainedTypes(RetTy)) {
Cost += TTI.getScalarizationOverhead(
cast<VectorType>(VectorTy), APInt::getAllOnes(VF.getFixedValue()),
- /*Insert=*/true, /*Extract=*/false, CostKind,
+ /*Insert=*/true, /*Extract=*/false, Config.CostKind,
/*ForPoisonSrc=*/true, {}, VIC);
}
}
@@ -5457,7 +4802,8 @@ LoopVectorizationCostModel::getScalarizationOverhead(Instruction *I,
TTI::VectorInstrContext OperandVIC = isa<StoreInst>(I)
? TTI::VectorInstrContext::Store
: TTI::VectorInstrContext::None;
- return Cost + TTI.getOperandsScalarizationOverhead(Tys, CostKind, OperandVIC);
+ return Cost +
+ TTI.getOperandsScalarizationOverhead(Tys, Config.CostKind, OperandVIC);
}
void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
@@ -5503,8 +4849,9 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
};
const InstructionCost GatherScatterCost =
- isLegalGatherOrScatter(&I, VF) ?
- getGatherScatterCost(&I, VF) : InstructionCost::getInvalid();
+ Config.isLegalGatherOrScatter(&I, VF)
+ ? getGatherScatterCost(&I, VF)
+ : InstructionCost::getInvalid();
// Load: Scalar load + broadcast
// Store: Scalar store + isLoopInvariantStoreValue ? 0 : extract
@@ -5554,7 +4901,7 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
}
InstructionCost GatherScatterCost =
- isLegalGatherOrScatter(&I, VF)
+ Config.isLegalGatherOrScatter(&I, VF)
? getGatherScatterCost(&I, VF) * NumAccesses
: InstructionCost::getInvalid();
@@ -5713,8 +5060,8 @@ void LoopVectorizationCostModel::setVectorizedCallDecision(ElementCount VF) {
// there, execute VF scalar calls, and then gather the result into the
// vector return value.
if (VF.isFixed()) {
- InstructionCost ScalarCallCost =
- TTI.getCallInstrCost(ScalarFunc, ScalarRetTy, ScalarTys, CostKind);
+ InstructionCost ScalarCallCost = TTI.getCallInstrCost(
+ ScalarFunc, ScalarRetTy, ScalarTys, Config.CostKind);
// Compute costs of unpacking argument values for the scalar calls and
// packing the return values to a vector.
@@ -5816,7 +5163,7 @@ void LoopVectorizationCostModel::setVectorizedCallDecision(ElementCount VF) {
}
if (TLI && VecFunc && !CI->isNoBuiltin())
- VectorCost = TTI.getCallInstrCost(nullptr, RetTy, Tys, CostKind);
+ VectorCost = TTI.getCallInstrCost(nullptr, RetTy, Tys, Config.CostKind);
// Find the cost of an intrinsic; some targets may have instructions that
// perform the operation without needing an actual call.
@@ -5949,14 +5296,14 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
VectorType::get(IntegerType::getInt1Ty(RetTy->getContext()), VF);
return (TTI.getScalarizationOverhead(
VecI1Ty, APInt::getAllOnes(VF.getFixedValue()),
- /*Insert*/ false, /*Extract*/ true, CostKind) +
- (TTI.getCFInstrCost(Instruction::CondBr, CostKind) *
+ /*Insert*/ false, /*Extract*/ true, Config.CostKind) +
+ (TTI.getCFInstrCost(Instruction::CondBr, Config.CostKind) *
VF.getFixedValue()));
}
if (I->getParent() == TheLoop->getLoopLatch() || VF.isScalar())
// The back-edge branch will remain, as will all scalar branches.
- return TTI.getCFInstrCost(Instruction::UncondBr, CostKind);
+ return TTI.getCFInstrCost(Instruction::UncondBr, Config.CostKind);
// This branch will be eliminated by if-conversion.
return 0;
@@ -5966,23 +5313,23 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
}
case Instruction::Switch: {
if (VF.isScalar())
- return TTI.getCFInstrCost(Instruction::Switch, CostKind);
+ return TTI.getCFInstrCost(Instruction::Switch, Config.CostKind);
auto *Switch = cast<SwitchInst>(I);
return Switch->getNumCases() *
TTI.getCmpSelInstrCost(
Instruction::ICmp,
toVectorTy(Switch->getCondition()->getType(), VF),
toVectorTy(Type::getInt1Ty(I->getContext()), VF),
- CmpInst::ICMP_EQ, CostKind);
+ CmpInst::ICMP_EQ, Config.CostKind);
}
case Instruction::PHI: {
auto *Phi = cast<PHINode>(I);
// First-order recurrences are replaced by vector shuffles inside the loop.
if (VF.isVector() && Legal->isFixedOrderRecurrence(Phi)) {
- return TTI.getShuffleCost(TargetTransformInfo::SK_Splice,
- cast<VectorType>(VectorTy),
- cast<VectorType>(VectorTy), {}, CostKind, -1);
+ return TTI.getShuffleCost(
+ TargetTransformInfo::SK_Splice, cast<VectorType>(VectorTy),
+ cast<VectorType>(VectorTy), {}, Config.CostKind, -1);
}
// Phi nodes in non-header blocks (not inductions, reductions, etc.) are
@@ -6012,20 +5359,21 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
TTI.getCmpSelInstrCost(
Instruction::Select, toVectorTy(ResultTy, VF),
toVectorTy(Type::getInt1Ty(Phi->getContext()), VF),
- CmpInst::BAD_ICMP_PREDICATE, CostKind);
+ CmpInst::BAD_ICMP_PREDICATE, Config.CostKind);
}
// When tail folding with EVL, if the phi is part of an out of loop
// reduction then it will be transformed into a wide vp_merge.
if (VF.isVector() && foldTailWithEVL() &&
- Legal->getReductionVars().contains(Phi) && !isInLoopReduction(Phi)) {
+ Legal->getReductionVars().contains(Phi) &&
+ !Config.isInLoopReduction(Phi)) {
IntrinsicCostAttributes ICA(
Intrinsic::vp_merge, toVectorTy(Phi->getType(), VF),
{toVectorTy(Type::getInt1Ty(Phi->getContext()), VF)});
- return TTI.getIntrinsicInstrCost(ICA, CostKind);
+ return TTI.getIntrinsicInstrCost(ICA, Config.CostKind);
}
- return TTI.getCFInstrCost(Instruction::PHI, CostKind);
+ return TTI.getCFInstrCost(Instruction::PHI, Config.CostKind);
}
case Instruction::UDiv:
case Instruction::SDiv:
@@ -6048,8 +5396,8 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
InstructionCost MulCost = TTI::TCC_Free;
ConstantInt *RHS = dyn_cast<ConstantInt>(I->getOperand(1));
if (!RHS || RHS->getZExtValue() != 1)
- MulCost =
- TTI.getArithmeticInstrCost(Instruction::Mul, VectorTy, CostKind);
+ MulCost = TTI.getArithmeticInstrCost(Instruction::Mul, VectorTy,
+ Config.CostKind);
// Find the cost of the histogram operation itself.
Type *PtrTy = VectorType::get(HGram->Load->getPointerOperandType(), VF);
@@ -6060,8 +5408,9 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
{PtrTy, ScalarTy, MaskTy});
// Add the costs together with the add/sub operation.
- return TTI.getIntrinsicInstrCost(ICA, CostKind) + MulCost +
- TTI.getArithmeticInstrCost(I->getOpcode(), VectorTy, CostKind);
+ return TTI.getIntrinsicInstrCost(ICA, Config.CostKind) + MulCost +
+ TTI.getArithmeticInstrCost(I->getOpcode(), VectorTy,
+ Config.CostKind);
}
[[fallthrough]];
}
@@ -6106,13 +5455,13 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
SmallVector<const Value *, 4> Operands(I->operand_values());
return TTI.getArithmeticInstrCost(
- I->getOpcode(), VectorTy, CostKind,
+ I->getOpcode(), VectorTy, Config.CostKind,
{TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
Op2Info, Operands, I, TLI);
}
case Instruction::FNeg: {
return TTI.getArithmeticInstrCost(
- I->getOpcode(), VectorTy, CostKind,
+ I->getOpcode(), VectorTy, Config.CostKind,
{TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
{TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
I->getOperand(0), I);
@@ -6135,7 +5484,8 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
return TTI.getArithmeticInstrCost(
match(I, m_LogicalOr()) ? Instruction::Or : Instruction::And,
- VectorTy, CostKind, {Op1VK, Op1VP}, {Op2VK, Op2VP}, {Op0, Op1}, I);
+ VectorTy, Config.CostKind, {Op1VK, Op1VP}, {Op2VK, Op2VP}, {Op0, Op1},
+ I);
}
Type *CondTy = SI->getCondition()->getType();
@@ -6145,9 +5495,9 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
CmpInst::Predicate Pred = CmpInst::BAD_ICMP_PREDICATE;
if (auto *Cmp = dyn_cast<CmpInst>(SI->getCondition()))
Pred = Cmp->getPredicate();
- return TTI.getCmpSelInstrCost(I->getOpcode(), VectorTy, CondTy, Pred,
- CostKind, {TTI::OK_AnyValue, TTI::OP_None},
- {TTI::OK_AnyValue, TTI::OP_None}, I);
+ return TTI.getCmpSelInstrCost(
+ I->getOpcode(), VectorTy, CondTy, Pred, Config.CostKind,
+ {TTI::OK_AnyValue, TTI::OP_None}, {TTI::OK_AnyValue, TTI::OP_None}, I);
}
case Instruction::ICmp:
case Instruction::FCmp: {
@@ -6166,7 +5516,7 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
VectorTy = toVectorTy(ValTy, VF);
return TTI.getCmpSelInstrCost(
I->getOpcode(), VectorTy, CmpInst::makeCmpResultType(VectorTy),
- cast<CmpInst>(I)->getPredicate(), CostKind,
+ cast<CmpInst>(I)->getPredicate(), Config.CostKind,
{TTI::OK_AnyValue, TTI::OP_None}, {TTI::OK_AnyValue, TTI::OP_None}, I);
}
case Instruction::Store:
@@ -6249,7 +5599,8 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
if (isOptimizableIVTruncate(I, VF)) {
auto *Trunc = cast<TruncInst>(I);
return TTI.getCastInstrCost(Instruction::Trunc, Trunc->getDestTy(),
- Trunc->getSrcTy(), CCH, CostKind, Trunc);
+ Trunc->getSrcTy(), CCH, Config.CostKind,
+ Trunc);
}
// Detect reduction patterns
@@ -6273,21 +5624,23 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
return 0;
}
- return TTI.getCastInstrCost(Opcode, VectorTy, SrcVecTy, CCH, CostKind, I);
+ return TTI.getCastInstrCost(Opcode, VectorTy, SrcVecTy, CCH,
+ Config.CostKind, I);
}
case Instruction::Call:
return getVectorCallCost(cast<CallInst>(I), VF);
case Instruction::ExtractValue:
- return TTI.getInstructionCost(I, CostKind);
+ return TTI.getInstructionCost(I, Config.CostKind);
case Instruction::Alloca:
// We cannot easily widen alloca to a scalable alloca, as
// the result would need to be a vector of pointers.
if (VF.isScalable())
return InstructionCost::getInvalid();
- return TTI.getArithmeticInstrCost(Instruction::Mul, RetTy, CostKind);
+ return TTI.getArithmeticInstrCost(Instruction::Mul, RetTy, Config.CostKind);
default:
// This opcode is unknown. Assume that it is the same as 'mul'.
- return TTI.getArithmeticInstrCost(Instruction::Mul, VectorTy, CostKind);
+ return TTI.getArithmeticInstrCost(Instruction::Mul, VectorTy,
+ Config.CostKind);
} // end of switch.
}
@@ -6427,58 +5780,6 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
}
}
-void LoopVectorizationCostModel::collectInLoopReductions() {
- // Avoid duplicating work finding in-loop reductions.
- if (!InLoopReductions.empty())
- return;
-
- for (const auto &Reduction : Legal->getReductionVars()) {
- PHINode *Phi = Reduction.first;
- const RecurrenceDescriptor &RdxDesc = Reduction.second;
-
- // Multi-use reductions (e.g., used in FindLastIV patterns) are handled
- // separately and should not be considered for in-loop reductions.
- if (RdxDesc.hasUsesOutsideReductionChain())
- continue;
-
- // We don't collect reductions that are type promoted (yet).
- if (RdxDesc.getRecurrenceType() != Phi->getType())
- continue;
-
- // In-loop AnyOf and FindIV reductions are not yet supported.
- RecurKind Kind = RdxDesc.getRecurrenceKind();
- if (RecurrenceDescriptor::isAnyOfRecurrenceKind(Kind) ||
- RecurrenceDescriptor::isFindIVRecurrenceKind(Kind) ||
- RecurrenceDescriptor::isFindLastRecurrenceKind(Kind))
- continue;
-
- // If the target would prefer this reduction to happen "in-loop", then we
- // want to record it as such.
- if (!PreferInLoopReductions && !useOrderedReductions(RdxDesc) &&
- !TTI.preferInLoopReduction(Kind, Phi->getType()))
- continue;
-
- // Check that we can correctly put the reductions into the loop, by
- // finding the chain of operations that leads from the phi to the loop
- // exit value.
- SmallVector<Instruction *, 4> ReductionOperations =
- RdxDesc.getReductionOpChain(Phi, TheLoop);
- bool InLoop = !ReductionOperations.empty();
-
- if (InLoop) {
- InLoopReductions.insert(Phi);
- // Add the elements to InLoopReductionImmediateChains for cost modelling.
- Instruction *LastChain = Phi;
- for (auto *I : ReductionOperations) {
- InLoopReductionImmediateChains[I] = LastChain;
- LastChain = I;
- }
- }
- LLVM_DEBUG(dbgs() << "LV: Using " << (InLoop ? "inloop" : "out of loop")
- << " reduction for phi: " << *Phi << "\n");
- }
-}
-
// This function will select a scalable VF if the target supports scalable
// vectors and a fixed one otherwise.
// TODO: we could return a pair of values that specify the max VF and
@@ -6487,9 +5788,8 @@ void LoopVectorizationCostModel::collectInLoopReductions() {
// doesn't have a cost model that can choose which plan to execute if
// more than one is generated.
static ElementCount determineVPlanVF(const TargetTransformInfo &TTI,
- LoopVectorizationCostModel &CM) {
- unsigned WidestType;
- std::tie(std::ignore, WidestType) = CM.getSmallestAndWidestTypes();
+ VFSelectionContext &Config) {
+ unsigned WidestType = Config.getSmallestAndWidestTypes().second;
TargetTransformInfo::RegisterKind RegKind =
TTI.enableScalableVectorization()
@@ -6512,7 +5812,7 @@ LoopVectorizationPlanner::planInVPlanNativePath(ElementCount UserVF) {
// If the user doesn't provide a vectorization factor, determine a
// reasonable one.
if (UserVF.isZero()) {
- VF = determineVPlanVF(TTI, CM);
+ VF = determineVPlanVF(TTI, Config);
LLVM_DEBUG(dbgs() << "LV: VPlan computed VF " << VF << ".\n");
// Make sure we have a VF > 1 for stress testing.
@@ -6521,8 +5821,7 @@ LoopVectorizationPlanner::planInVPlanNativePath(ElementCount UserVF) {
<< "overriding computed VF.\n");
VF = ElementCount::getFixed(4);
}
- } else if (UserVF.isScalable() && !TTI.supportsScalableVectors() &&
- !ForceTargetSupportsScalableVectors) {
+ } else if (UserVF.isScalable() && !Config.supportsScalableVectors()) {
LLVM_DEBUG(dbgs() << "LV: Not vectorizing. Scalable VF requested, but "
<< "not supported by the target.\n");
reportVectorizationFailure(
@@ -6559,7 +5858,7 @@ LoopVectorizationPlanner::planInVPlanNativePath(ElementCount UserVF) {
void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
assert(OrigLoop->isInnermost() && "Inner loop expected.");
CM.collectValuesToIgnore();
- CM.collectElementTypesForWidening();
+ Config.collectElementTypesForWidening(&CM.ValuesToIgnore);
FixedScalableVFPair MaxFactors = CM.computeMaxVF(UserVF, UserIC);
if (!MaxFactors) // Cases that should not to be vectorized nor interleaved.
@@ -6594,7 +5893,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
"VF needs to be a power of two");
// Collect the instructions (and their associated costs) that will be more
// profitable to scalarize.
- CM.collectInLoopReductions();
+ Config.collectInLoopReductions();
CM.collectNonVectorizedAndSetWideningDecisions(UserVF);
ElementCount EpilogueUserVF =
ElementCount::getFixed(EpilogueVectorizationForceVF);
@@ -6629,7 +5928,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
ElementCount::isKnownLE(VF, MaxFactors.ScalableVF); VF *= 2)
VFCandidates.push_back(VF);
- CM.collectInLoopReductions();
+ Config.collectInLoopReductions();
for (const auto &VF : VFCandidates) {
// Collect Uniform and Scalar instructions after vectorization with VF.
CM.collectNonVectorizedAndSetWideningDecisions(VF);
@@ -6822,18 +6121,20 @@ LoopVectorizationPlanner::precomputeCosts(VPlan &Plan, ElementCount VF,
InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
VPRegisterUsage *RU) const {
- VPCostContext CostCtx(CM.TTI, *CM.TLI, Plan, CM, CM.CostKind, PSE, OrigLoop);
+ VPCostContext CostCtx(CM.TTI, *CM.TLI, Plan, CM, Config.CostKind, PSE,
+ OrigLoop);
InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
// Now compute and add the VPlan-based cost.
Cost += Plan.cost(VF, CostCtx);
// Add the cost of spills due to excess register usage
- if (RU && CM.shouldConsiderRegPressureForVF(VF))
+ if (RU && Config.shouldConsiderRegPressureForVF(VF))
Cost += RU->spillCost(CostCtx, ForceTargetNumVectorRegs);
#ifndef NDEBUG
- unsigned EstimatedWidth = estimateElementCount(VF, CM.getVScaleForTuning());
+ unsigned EstimatedWidth =
+ estimateElementCount(VF, Config.getVScaleForTuning());
LLVM_DEBUG(dbgs() << "Cost for VF " << VF << ": " << Cost
<< " (Estimated cost per lane: ");
if (Cost.isValid()) {
@@ -6871,12 +6172,12 @@ LoopVectorizationPlanner::computeBestVF() {
}
LLVM_DEBUG(dbgs() << "LV: Computing best VF using cost kind: "
- << (CM.CostKind == TTI::TCK_RecipThroughput
+ << (Config.CostKind == TTI::TCK_RecipThroughput
? "Reciprocal Throughput\n"
- : CM.CostKind == TTI::TCK_Latency
+ : Config.CostKind == TTI::TCK_Latency
? "Instruction Latency\n"
- : CM.CostKind == TTI::TCK_CodeSize ? "Code Size\n"
- : CM.CostKind == TTI::TCK_SizeAndLatency
+ : Config.CostKind == TTI::TCK_CodeSize ? "Code Size\n"
+ : Config.CostKind == TTI::TCK_SizeAndLatency
? "Code Size and Latency\n"
: "Unknown\n"));
@@ -6906,7 +6207,7 @@ LoopVectorizationPlanner::computeBestVF() {
SmallVector<VPRegisterUsage, 8> RUs;
bool ConsiderRegPressure = any_of(VFs, [this](ElementCount VF) {
- return CM.shouldConsiderRegPressureForVF(VF);
+ return Config.shouldConsiderRegPressureForVF(VF);
});
if (ConsiderRegPressure)
RUs = calculateRegisterUsageForPlan(*P, VFs, TTI, CM.ValuesToIgnore);
@@ -6922,7 +6223,7 @@ LoopVectorizationPlanner::computeBestVF() {
<< " because it will not generate any vector instructions.\n");
continue;
}
- if (CM.OptForSize && !ForceVectorization && hasReplicatorRegion(*P)) {
+ if (Config.OptForSize && !ForceVectorization && hasReplicatorRegion(*P)) {
LLVM_DEBUG(
dbgs()
<< "LV: Not considering vector loop of width " << VF
@@ -6974,7 +6275,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
bool HasBranchWeights =
hasBranchWeightMD(*OrigLoop->getLoopLatch()->getTerminator());
if (HasBranchWeights) {
- std::optional<unsigned> VScale = CM.getVScaleForTuning();
+ std::optional<unsigned> VScale = Config.getVScaleForTuning();
RUN_VPLAN_PASS(VPlanTransforms::addBranchWeightToMiddleTerminator,
BestVPlan, BestVF, VScale);
}
@@ -7103,7 +6404,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
HeaderVPBB, BestVPlan,
EpilogueVecKind == EpilogueVectorizationKind::Epilogue, LID,
OrigAverageTripCount, OrigLoopInvocationWeight,
- estimateElementCount(BestVF * BestUF, CM.getVScaleForTuning()),
+ estimateElementCount(BestVF * BestUF, Config.getVScaleForTuning()),
DisableRuntimeUnroll);
// 3. Fix the vectorized code: take care of header phi's, live-outs,
@@ -7641,7 +6942,7 @@ void LoopVectorizationPlanner::buildVPlansWithVPRecipes(ElementCount MinVF,
RUN_VPLAN_PASS(VPlanTransforms::createHeaderPhiRecipes, *VPlan0, PSE,
*OrigLoop, Legal->getInductionVars(),
Legal->getReductionVars(), Legal->getFixedOrderRecurrences(),
- CM.getInLoopReductions(), Hints.allowReordering());
+ Config.getInLoopReductions(), Hints.allowReordering());
RUN_VPLAN_PASS(VPlanTransforms::simplifyRecipes, *VPlan0);
// If we're vectorizing a loop with an uncountable exit, make sure that the
@@ -7680,7 +6981,7 @@ void LoopVectorizationPlanner::buildVPlansWithVPRecipes(ElementCount MinVF,
// TODO: try to put addExplicitVectorLength close to addActiveLaneMask
if (CM.foldTailWithEVL()) {
RUN_VPLAN_PASS(VPlanTransforms::addExplicitVectorLength, *Plan,
- CM.getMaxSafeElements());
+ Config.getMaxSafeElements());
RUN_VPLAN_PASS(VPlanTransforms::optimizeEVLMasks, *Plan);
}
@@ -7792,7 +7093,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlanWithVPRecipes(
RUN_VPLAN_PASS(VPlanTransforms::createInLoopReductionRecipes, *Plan,
BlocksNeedingPredication, Range.Start);
- VPCostContext CostCtx(CM.TTI, *CM.TLI, *Plan, CM, CM.CostKind, CM.PSE,
+ VPCostContext CostCtx(CM.TTI, *CM.TLI, *Plan, CM, Config.CostKind, CM.PSE,
OrigLoop);
RUN_VPLAN_PASS_NO_VERIFY(VPlanTransforms::makeMemOpWideningDecisions, *Plan,
@@ -8162,7 +7463,7 @@ void LoopVectorizationPlanner::attachRuntimeChecks(
VPlan &Plan, GeneratedRTChecks &RTChecks, bool HasBranchWeights) const {
const auto &[SCEVCheckCond, SCEVCheckBlock] = RTChecks.getSCEVChecks();
if (SCEVCheckBlock && SCEVCheckBlock->hasNPredecessors(0)) {
- assert((!CM.OptForSize ||
+ assert((!Config.OptForSize ||
CM.Hints->getForce() == LoopVectorizeHints::FK_Enabled) &&
"Cannot SCEV check stride or overflow when optimizing for size");
VPlanTransforms::attachCheckBlock(Plan, SCEVCheckCond, SCEVCheckBlock,
@@ -8175,7 +7476,7 @@ void LoopVectorizationPlanner::attachRuntimeChecks(
assert((!EnableVPlanNativePath || OrigLoop->isInnermost()) &&
"Runtime checks are not supported for outer loops yet");
- if (CM.OptForSize) {
+ if (Config.OptForSize) {
assert(
CM.Hints->getForce() == LoopVectorizeHints::FK_Enabled &&
"Cannot emit memory checks when optimizing for size, unless forced "
@@ -8275,18 +7576,19 @@ static bool processLoopInVPlanNativePath(
EpilogueLowering SEL =
getEpilogueLowering(F, L, Hints, OptForSize, TTI, TLI, *LVL, &IAI);
+ VFSelectionContext Config(*TTI, LVL, L, *F, PSE, ORE, &Hints, OptForSize);
LoopVectorizationCostModel CM(SEL, L, PSE, LI, LVL, *TTI, TLI, DB, AC, ORE,
- GetBFI, F, &Hints, IAI, OptForSize);
+ GetBFI, F, &Hints, IAI, Config);
// Use the planner for outer loop vectorization.
// TODO: CM is not used at this point inside the planner. Turn CM into an
// optional argument if we don't need it in the future.
- LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, LVL, CM, IAI, PSE, Hints,
- ORE);
+ LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, LVL, CM, Config, IAI, PSE,
+ Hints, ORE);
// Get user vectorization factor.
ElementCount UserVF = Hints.getWidth();
- CM.collectElementTypesForWidening();
+ Config.collectElementTypesForWidening();
// Plan how to best vectorize, return the best VF and its cost.
const VectorizationFactor VF = LVP.planInVPlanNativePath(UserVF);
@@ -8300,7 +7602,7 @@ static bool processLoopInVPlanNativePath(
VPlan &BestPlan = LVP.getPlanFor(VF.Width);
{
- GeneratedRTChecks Checks(PSE, DT, LI, TTI, CM.CostKind);
+ GeneratedRTChecks Checks(PSE, DT, LI, TTI, Config.CostKind);
InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, /*UF=*/1, &CM,
Checks, BestPlan);
LLVM_DEBUG(dbgs() << "Vectorizing outer loop in \"" << F->getName()
@@ -8601,7 +7903,7 @@ preparePlanForMainVectorLoop(VPlan &MainPlan, VPlan &EpiPlan) {
static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
VPlan &Plan, Loop *L, const SCEV2ValueTy &ExpandedSCEVs,
EpilogueLoopVectorizationInfo &EPI, LoopVectorizationCostModel &CM,
- ScalarEvolution &SE) {
+ VFSelectionContext &Config, ScalarEvolution &SE) {
VPRegionBlock *VectorLoop = Plan.getVectorLoopRegion();
VPBasicBlock *Header = VectorLoop->getEntryBasicBlock();
Header->setName("vec.epilog.vector.body");
@@ -8785,7 +8087,7 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
ExpandR->eraseFromParent();
}
- auto VScale = CM.getVScaleForTuning();
+ auto VScale = Config.getVScaleForTuning();
unsigned MainLoopStep =
estimateElementCount(EPI.MainLoopVF * EPI.MainLoopUF, VScale);
unsigned EpilogueLoopStep =
@@ -9107,11 +8409,12 @@ bool LoopVectorizePass::processLoop(Loop *L) {
}
// Use the cost model.
+ VFSelectionContext Config(*TTI, &LVL, L, *F, PSE, ORE, &Hints, OptForSize);
LoopVectorizationCostModel CM(SEL, L, PSE, LI, &LVL, *TTI, TLI, DB, AC, ORE,
- GetBFI, F, &Hints, IAI, OptForSize);
+ GetBFI, F, &Hints, IAI, Config);
// Use the planner for vectorization.
- LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, CM, IAI, PSE, Hints,
- ORE);
+ LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, CM, Config, IAI, PSE,
+ Hints, ORE);
// Get user vectorization factor and interleave count.
ElementCount UserVF = Hints.getWidth();
@@ -9127,7 +8430,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
if (ORE->allowExtraAnalysis(LV_NAME))
LVP.emitInvalidCostRemarks(ORE);
- GeneratedRTChecks Checks(PSE, DT, LI, TTI, CM.CostKind);
+ GeneratedRTChecks Checks(PSE, DT, LI, TTI, Config.CostKind);
if (LVP.hasPlanWithVF(VF.Width)) {
// Select the interleave count.
IC = LVP.selectInterleaveCount(*BestPlanPtr, VF.Width, VF.Cost);
@@ -9153,11 +8456,11 @@ bool LoopVectorizePass::processLoop(Loop *L) {
// Check if it is profitable to vectorize with runtime checks.
bool ForceVectorization =
Hints.getForce() == LoopVectorizeHints::FK_Enabled;
- VPCostContext CostCtx(CM.TTI, *CM.TLI, *BestPlanPtr, CM, CM.CostKind,
+ VPCostContext CostCtx(CM.TTI, *CM.TLI, *BestPlanPtr, CM, Config.CostKind,
CM.PSE, L);
if (!ForceVectorization &&
!isOutsideLoopWorkProfitable(Checks, VF, L, PSE, CostCtx, *BestPlanPtr,
- SEL, CM.getVScaleForTuning())) {
+ SEL, Config.getVScaleForTuning())) {
ORE->emit([&]() {
return OptimizationRemarkAnalysisAliasing(
DEBUG_TYPE, "CantReorderMemOps", L->getStartLoc(),
@@ -9355,7 +8658,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
EpilogueVectorizerEpilogueLoop EpilogILV(L, PSE, LI, DT, TTI, AC, EPI, &CM,
Checks, BestEpiPlan);
SmallVector<Instruction *> InstsToMove = preparePlanForEpilogueVectorLoop(
- BestEpiPlan, L, ExpandedSCEVs, EPI, CM, *PSE.getSE());
+ BestEpiPlan, L, ExpandedSCEVs, EPI, CM, Config, *PSE.getSE());
LVP.attachRuntimeChecks(BestEpiPlan, Checks, HasBranchWeights);
LVP.executePlan(
EPI.EpilogueVF, EPI.EpilogueUF, BestEpiPlan, EpilogILV, DT,
More information about the llvm-commits
mailing list