[llvm] [LV] Remove legacy setVectorizedCallDecision & co (NFC). (PR #195519)
via llvm-commits
llvm-commits at lists.llvm.org
Sun May 3 04:38:44 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-vectorizers
@llvm/pr-subscribers-llvm-transforms
Author: Florian Hahn (fhahn)
<details>
<summary>Changes</summary>
Remove setVectorizedCallDecision & co after being superseded by
https://github.com/llvm/llvm-project/pull/195518.
Note that we still need to retain some of the call cost logic in the
legacy cost model, to compute if scalarization is profitable.
Depends on https://github.com/llvm/llvm-project/pull/195518 (included in
PR)
---
Patch is 48.15 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/195519.diff
9 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorize.cpp (+95-329)
- (modified) llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h (+2-12)
- (modified) llvm/lib/Transforms/Vectorize/VPlan.h (+17-4)
- (modified) llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp (+3)
- (modified) llvm/lib/Transforms/Vectorize/VPlanHelpers.h (+15)
- (modified) llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp (+59-51)
- (modified) llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp (+175)
- (modified) llvm/lib/Transforms/Vectorize/VPlanTransforms.h (+6)
- (modified) llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll (+1)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 78163b5fe35d5..eec04001ade14 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -847,13 +847,6 @@ class LoopVectorizationCostModel {
/// avoid redundant calculations.
void setCostBasedWideningDecision(ElementCount VF);
- /// A call may be vectorized in different ways depending on whether we have
- /// vectorized variants available and whether the target supports masking.
- /// This function analyzes all calls in the function at the supplied VF,
- /// makes a decision based on the costs of available options, and stores that
- /// decision in a map for use in planning and plan execution.
- void setVectorizedCallDecision(ElementCount VF);
-
/// Collect values we want to ignore in the cost model.
void collectValuesToIgnore();
@@ -928,8 +921,6 @@ class LoopVectorizationCostModel {
CM_Interleave,
CM_GatherScatter,
CM_Scalarize,
- CM_VectorCall,
- CM_IntrinsicCall
};
/// Save vectorization decision \p W and \p Cost taken by the cost model for
@@ -990,31 +981,6 @@ class LoopVectorizationCostModel {
return WideningDecisions[InstOnVF].second;
}
- struct CallWideningDecision {
- InstWidening Kind;
- Function *Variant;
- Intrinsic::ID IID;
- std::optional<unsigned> MaskPos;
- InstructionCost Cost;
- };
-
- void setCallWideningDecision(CallInst *CI, ElementCount VF, InstWidening Kind,
- Function *Variant, Intrinsic::ID IID,
- std::optional<unsigned> MaskPos,
- InstructionCost Cost) {
- assert(!VF.isScalar() && "Expected vector VF");
- CallWideningDecisions[{CI, VF}] = {Kind, Variant, IID, MaskPos, Cost};
- }
-
- CallWideningDecision getCallWideningDecision(CallInst *CI,
- ElementCount VF) const {
- assert(!VF.isScalar() && "Expected vector VF");
- auto I = CallWideningDecisions.find({CI, VF});
- if (I == CallWideningDecisions.end())
- return {CM_Unknown, nullptr, Intrinsic::not_intrinsic, std::nullopt, 0};
- return I->second;
- }
-
/// Return True if instruction \p I is an optimizable truncate whose operand
/// is an induction variable. Such a truncate will be removed by adding a new
/// induction variable with the destination type.
@@ -1058,7 +1024,6 @@ class LoopVectorizationCostModel {
return;
setCostBasedWideningDecision(VF);
collectLoopUniforms(VF);
- setVectorizedCallDecision(VF);
collectLoopScalars(VF);
collectInstsToScalarize(VF);
}
@@ -1279,7 +1244,6 @@ class LoopVectorizationCostModel {
/// Invalidates decisions already taken by the cost model.
void invalidateCostModelingDecisions() {
WideningDecisions.clear();
- CallWideningDecisions.clear();
Uniforms.clear();
Scalars.clear();
}
@@ -1312,6 +1276,12 @@ class LoopVectorizationCostModel {
/// trivially hoistable.
bool shouldConsiderInvariant(Value *Op);
+ /// Returns true if \p I has been forced to be scalarized at \p VF.
+ bool isForcedScalar(Instruction *I, ElementCount VF) const {
+ auto FS = ForcedScalars.find(VF);
+ return FS != ForcedScalars.end() && FS->second.contains(I);
+ }
+
private:
unsigned NumPredStores = 0;
@@ -1421,20 +1391,13 @@ class LoopVectorizationCostModel {
DecisionList WideningDecisions;
- using CallDecisionList =
- DenseMap<std::pair<CallInst *, ElementCount>, CallWideningDecision>;
-
- CallDecisionList CallWideningDecisions;
-
/// Returns true if \p V is expected to be vectorized and it needs to be
/// extracted.
bool needsExtract(Value *V, ElementCount VF) const {
Instruction *I = dyn_cast<Instruction>(V);
if (VF.isScalar() || !I || !TheLoop->contains(I) ||
TheLoop->isLoopInvariant(I) ||
- getWideningDecision(I, VF) == CM_Scalarize ||
- (isa<CallInst>(I) &&
- getCallWideningDecision(cast<CallInst>(I), VF).Kind == CM_Scalarize))
+ getWideningDecision(I, VF) == CM_Scalarize)
return false;
// Assume we can vectorize V (and hence we need extraction) if the
@@ -2077,32 +2040,74 @@ static unsigned estimateElementCount(ElementCount VF,
return EstimatedVF;
}
+/// Returns true iff \p CI has a library vector variant usable at \p VF: a
+/// mapping with matching VF, masked if required, whose vector function is
+/// declared in the module. Such variants are priced by
+/// VPWidenCallRecipe::computeCost rather than by scalarization.
+static bool hasVectorLibraryVariantFor(const CallInst &CI, ElementCount VF,
+ bool MaskRequired,
+ const TargetLibraryInfo *TLI) {
+ if (!TLI || CI.isNoBuiltin())
+ return false;
+ for (const VFInfo &Info : VFDatabase::getMappings(CI)) {
+ if (Info.Shape.VF != VF)
+ continue;
+ if (MaskRequired && !Info.isMasked())
+ continue;
+ if (CI.getModule()->getFunction(Info.VectorName))
+ return true;
+ }
+ return false;
+}
+
InstructionCost
LoopVectorizationCostModel::getVectorCallCost(CallInst *CI,
ElementCount VF) const {
- // We only need to calculate a cost if the VF is scalar; for actual vectors
- // we should already have a pre-calculated cost at each VF.
- if (!VF.isScalar())
- return getCallWideningDecision(CI, VF).Cost;
-
Type *RetTy = CI->getType();
- if (RecurrenceDescriptor::isFMulAddIntrinsic(CI))
- if (auto RedCost = getReductionPatternCost(CI, VF, RetTy))
- return *RedCost;
-
- SmallVector<Type *, 4> Tys;
- for (auto &ArgOp : CI->args())
- Tys.push_back(ArgOp->getType());
- InstructionCost ScalarCallCost = TTI.getCallInstrCost(
- CI->getCalledFunction(), RetTy, Tys, Config.CostKind);
+ // Scalar VF: pick the cheaper of the scalar call and any matching vector
+ // intrinsic lowering. In-loop fmuladd reductions are priced specially.
+ if (VF.isScalar()) {
+ if (RecurrenceDescriptor::isFMulAddIntrinsic(CI))
+ if (auto RedCost = getReductionPatternCost(CI, VF, RetTy))
+ return *RedCost;
+
+ SmallVector<Type *, 4> Tys;
+ for (Value *Arg : CI->args())
+ Tys.push_back(Arg->getType());
+ InstructionCost ScalarCallCost = TTI.getCallInstrCost(
+ CI->getCalledFunction(), RetTy, Tys, Config.CostKind);
+
+ if (getVectorIntrinsicIDForCall(CI, TLI))
+ return std::min(ScalarCallCost, getVectorIntrinsicCost(CI, VF));
+ return ScalarCallCost;
+ }
+
+ // Vector VF: compare scalarization against any matching vector intrinsic
+ // lowering. Vector library variants are priced by
+ // VPWidenCallRecipe::computeCost and should not reach this function.
+ assert(!hasVectorLibraryVariantFor(*CI, VF, isMaskRequired(CI), TLI) &&
+ "getVectorCallCost does not price vector library variants");
+
+ // Scalarization is only meaningful for fixed VFs.
+ InstructionCost Cost = InstructionCost::getInvalid();
+ if (VF.isFixed()) {
+ SmallVector<Type *, 4> Tys;
+ for (Value *Arg : CI->args())
+ Tys.push_back(Arg->getType());
+ InstructionCost ScalarCallCost = TTI.getCallInstrCost(
+ CI->getCalledFunction(), RetTy, Tys, Config.CostKind);
+ Cost = ScalarCallCost * VF.getKnownMinValue() +
+ getScalarizationOverhead(CI, VF);
+ }
- // If this is an intrinsic we may have a lower cost for it.
if (getVectorIntrinsicIDForCall(CI, TLI)) {
InstructionCost IntrinsicCost = getVectorIntrinsicCost(CI, VF);
- return std::min(ScalarCallCost, IntrinsicCost);
+ if (IntrinsicCost.isValid() && (!Cost.isValid() || IntrinsicCost <= Cost))
+ Cost = IntrinsicCost;
}
- return ScalarCallCost;
+
+ return Cost;
}
static Type *maybeVectorizeType(Type *Ty, ElementCount VF) {
@@ -2366,10 +2371,16 @@ bool LoopVectorizationCostModel::isScalarWithPredication(Instruction *I,
switch(I->getOpcode()) {
default:
return true;
- case Instruction::Call:
+ case Instruction::Call: {
if (VF.isScalar())
return true;
- return getCallWideningDecision(cast<CallInst>(I), VF).Kind == CM_Scalarize;
+ CallInst *CI = cast<CallInst>(I);
+ // A vector intrinsic lowering is always preferred over scalarization.
+ if (getVectorIntrinsicIDForCall(CI, TLI))
+ return false;
+ // A matching vector library variant also avoids scalarization.
+ return !hasVectorLibraryVariantFor(*CI, VF, isMaskRequired(CI), TLI);
+ }
case Instruction::Load:
case Instruction::Store: {
auto *Ptr = getLoadStorePointerOperand(I);
@@ -2919,8 +2930,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
return FixedScalableVFPair::getNone();
}
- assert(WideningDecisions.empty() && CallWideningDecisions.empty() &&
- Uniforms.empty() && Scalars.empty() &&
+ assert(WideningDecisions.empty() && Uniforms.empty() && Scalars.empty() &&
"No cost-modeling decisions should have been taken at this point");
switch (EpilogueLoweringStatus) {
@@ -4059,15 +4069,6 @@ void LoopVectorizationCostModel::collectInstsToScalarize(ElementCount VF) {
computePredInstDiscount(&I, ScalarCosts, VF) >= 0) {
for (const auto &[I, IC] : ScalarCosts)
ScalarCostsVF.insert({I, IC});
- // Check if we decided to scalarize a call. If so, update the widening
- // decision of the call to CM_Scalarize with the computed scalar cost.
- for (const auto &[I, Cost] : ScalarCosts) {
- auto *CI = dyn_cast<CallInst>(I);
- if (!CI || !CallWideningDecisions.contains({CI, VF}))
- continue;
- CallWideningDecisions[{CI, VF}].Kind = CM_Scalarize;
- CallWideningDecisions[{CI, VF}].Cost = Cost;
- }
}
// Remember that BB will remain after vectorization.
PredicatedBBsAfterVectorization[VF].insert(BB);
@@ -4927,163 +4928,6 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
}
}
-void LoopVectorizationCostModel::setVectorizedCallDecision(ElementCount VF) {
- assert(!VF.isScalar() &&
- "Trying to set a vectorization decision for a scalar VF");
-
- auto ForcedScalar = ForcedScalars.find(VF);
- for (BasicBlock *BB : TheLoop->blocks()) {
- // For each instruction in the old loop.
- for (Instruction &I : *BB) {
- CallInst *CI = dyn_cast<CallInst>(&I);
-
- if (!CI)
- continue;
-
- InstructionCost ScalarCost = InstructionCost::getInvalid();
- InstructionCost VectorCost = InstructionCost::getInvalid();
- InstructionCost IntrinsicCost = InstructionCost::getInvalid();
- Function *ScalarFunc = CI->getCalledFunction();
- Type *ScalarRetTy = CI->getType();
- SmallVector<Type *, 4> Tys, ScalarTys;
- for (auto &ArgOp : CI->args())
- ScalarTys.push_back(ArgOp->getType());
-
- // Estimate cost of scalarized vector call. The source operands are
- // assumed to be vectors, so we need to extract individual elements from
- // there, execute VF scalar calls, and then gather the result into the
- // vector return value.
- if (VF.isFixed()) {
- InstructionCost ScalarCallCost = TTI.getCallInstrCost(
- ScalarFunc, ScalarRetTy, ScalarTys, Config.CostKind);
-
- // Compute costs of unpacking argument values for the scalar calls and
- // packing the return values to a vector.
- InstructionCost ScalarizationCost = getScalarizationOverhead(CI, VF);
- ScalarCost = ScalarCallCost * VF.getKnownMinValue() + ScalarizationCost;
- } else {
- // There is no point attempting to calculate the scalar cost for a
- // scalable VF as we know it will be Invalid.
- assert(!getScalarizationOverhead(CI, VF).isValid() &&
- "Unexpected valid cost for scalarizing scalable vectors");
- ScalarCost = InstructionCost::getInvalid();
- }
-
- // Honor ForcedScalars and UniformAfterVectorization decisions.
- // TODO: For calls, it might still be more profitable to widen. Use
- // VPlan-based cost model to compare different options.
- if (VF.isVector() && ((ForcedScalar != ForcedScalars.end() &&
- ForcedScalar->second.contains(CI)) ||
- isUniformAfterVectorization(CI, VF))) {
- setCallWideningDecision(CI, VF, CM_Scalarize, nullptr,
- Intrinsic::not_intrinsic, std::nullopt,
- ScalarCost);
- continue;
- }
-
- bool MaskRequired = isMaskRequired(CI);
- // Compute corresponding vector type for return value and arguments.
- Type *RetTy = toVectorizedTy(ScalarRetTy, VF);
- for (Type *ScalarTy : ScalarTys)
- Tys.push_back(toVectorizedTy(ScalarTy, VF));
-
- // An in-loop reduction using an fmuladd intrinsic is a special case;
- // we don't want the normal cost for that intrinsic.
- if (RecurrenceDescriptor::isFMulAddIntrinsic(CI))
- if (auto RedCost = getReductionPatternCost(CI, VF, RetTy)) {
- setCallWideningDecision(CI, VF, CM_IntrinsicCall, nullptr,
- getVectorIntrinsicIDForCall(CI, TLI),
- std::nullopt, *RedCost);
- continue;
- }
-
- // Find the cost of vectorizing the call, if we can find a suitable
- // vector variant of the function.
- VFInfo FuncInfo;
- Function *VecFunc = nullptr;
- // Search through any available variants for one we can use at this VF.
- for (VFInfo &Info : VFDatabase::getMappings(*CI)) {
- // Must match requested VF.
- if (Info.Shape.VF != VF)
- continue;
-
- // Must take a mask argument if one is required
- if (MaskRequired && !Info.isMasked())
- continue;
-
- // Check that all parameter kinds are supported
- bool ParamsOk = true;
- for (VFParameter Param : Info.Shape.Parameters) {
- switch (Param.ParamKind) {
- case VFParamKind::Vector:
- break;
- case VFParamKind::OMP_Uniform: {
- Value *ScalarParam = CI->getArgOperand(Param.ParamPos);
- // Make sure the scalar parameter in the loop is invariant.
- if (!PSE.getSE()->isLoopInvariant(PSE.getSCEV(ScalarParam),
- TheLoop))
- ParamsOk = false;
- break;
- }
- case VFParamKind::OMP_Linear: {
- Value *ScalarParam = CI->getArgOperand(Param.ParamPos);
- // Find the stride for the scalar parameter in this loop and see if
- // it matches the stride for the variant.
- // TODO: do we need to figure out the cost of an extract to get the
- // first lane? Or do we hope that it will be folded away?
- ScalarEvolution *SE = PSE.getSE();
- if (!match(SE->getSCEV(ScalarParam),
- m_scev_AffineAddRec(
- m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
- m_SpecificLoop(TheLoop))))
- ParamsOk = false;
- break;
- }
- case VFParamKind::GlobalPredicate:
- break;
- default:
- ParamsOk = false;
- break;
- }
- }
-
- if (!ParamsOk)
- continue;
-
- // Found a suitable candidate, stop here.
- VecFunc = CI->getModule()->getFunction(Info.VectorName);
- FuncInfo = Info;
- break;
- }
-
- if (TLI && VecFunc && !CI->isNoBuiltin())
- VectorCost = TTI.getCallInstrCost(nullptr, RetTy, Tys, Config.CostKind);
-
- // Find the cost of an intrinsic; some targets may have instructions that
- // perform the operation without needing an actual call.
- Intrinsic::ID IID = getVectorIntrinsicIDForCall(CI, TLI);
- if (IID != Intrinsic::not_intrinsic)
- IntrinsicCost = getVectorIntrinsicCost(CI, VF);
-
- InstructionCost Cost = ScalarCost;
- InstWidening Decision = CM_Scalarize;
-
- if (VectorCost.isValid() && VectorCost <= Cost) {
- Cost = VectorCost;
- Decision = CM_VectorCall;
- }
-
- if (IntrinsicCost.isValid() && IntrinsicCost <= Cost) {
- Cost = IntrinsicCost;
- Decision = CM_IntrinsicCall;
- }
-
- setCallWideningDecision(CI, VF, Decision, VecFunc, IID,
- FuncInfo.getParamIndexForOptionalMask(), Cost);
- }
- }
-}
-
bool LoopVectorizationCostModel::shouldConsiderInvariant(Value *Op) {
if (!Legal->isInvariant(Op))
return false;
@@ -5466,9 +5310,6 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
return TTI::CastContextHint::Reversed;
case LoopVectorizationCostModel::CM_Unknown:
llvm_unreachable("Instr did not go through cost modelling?");
- case LoopVectorizationCostModel::CM_VectorCall:
- case LoopVectorizationCostModel::CM_IntrinsicCall:
- llvm_unreachable_internal("Instr has invalid widening decision");
}
llvm_unreachable("Unhandled case!");
@@ -5858,6 +5699,16 @@ uint64_t VPCostContext::getPredBlockCostDivisor(BasicBlock *BB) const {
return CM.getPredBlockCostDivisor(CostKind, BB);
}
+bool VPCostContext::willBeScalarized(Instruction *I, ElementCount VF) const {
+ return CM.isScalarWithPredication(I, VF) ||
+ CM.isUniformAfterVectorization(I, VF) || CM.isForcedScalar(I, VF) ||
+ (VF.isVector() && CM.isProfitableToScalarize(I, VF));
+}
+
+bool VPCostContext::isMaskRequired(Instruction *I) const {
+ return CM.isMaskRequired(I);
+}
+
InstructionCost
LoopVectorizationPlanner::precomputeCosts(VPlan &Plan, ElementCount VF,
VPCostContext &CostCtx) const {
@@ -6505,93 +6356,6 @@ VPRecipeBuilder::tryToOptimizeInductionTruncate(VPInstruction *VPI,
Phi, Start, Step, &Plan.getVF(), IndDesc, I, Flags, VPI->getDebugLoc());
}
-VPSingleDefRecipe *VPRecipeBuilder::tryToWidenCall(VPInstruction *VPI,
- VFRange &Range) {
- CallInst *CI = cast<CallInst>(VPI->getUnderlyingInstr());
- bool IsPredicated = LoopVectorizationPlanner::getDecisionAndClampRange(
- [this, CI](ElementCount VF) {
- return CM.isScalarWithPredication(CI, VF);
- },
- Range);
-
- if (IsPredicated)
- return nullptr;
-
- Intrinsic::ID ID = getVectorIntrinsicIDForCall(CI, TLI);
- if (ID && (ID == Intrinsic::assume || ID == Intrinsic::lifetime_end ||
- ID == Intrinsic::lifetime_start || ID == Intrinsic::sideeffect ||
- ID == Intrinsic::pseudoprobe ||
- ID == Intrinsic::experimental_noalias_scope_decl))
- return nullptr;
-
- SmallVector<VPValue *, 4> Ops(VPI->op_begin(),
- VPI->op_begin() + CI->arg_size());
-
- // Is it beneficial to perform intrinsic call compared to lib call?
- bool ShouldUseVectorIntrinsic =
- ID && LoopVectorizationPlanner::getDecisionAndClampRange(
- [&](ElementCount VF) -> bool {
- return CM.getCallWideningDecision(CI, VF).Kind ==
- LoopVectorizationCostModel::CM_IntrinsicCall;
- },
- Range);
- if (ShouldUseVectorIntrinsic)
- return new VPWidenIntrinsicRecipe(*CI, ID, Ops, CI->getType(), *VPI, *VPI,
- VPI->getDebugLoc());
-
- Function *Variant = nullptr;
- std::optional<unsigned> MaskPos;
- // Is better to call a vectorized version of the function than to to scalarize
- // the call?
- auto ShouldUseVectorCall = LoopVectorizationPlanner::getDecisionAndClampRange(
- [&](ElementCount VF) -> bool {
- // The following case may be scalarized depending on the VF.
- // The flag shows whether we can use a usual Call for vectorized
- // version of the instruction.
-
- // If we've found a variant at a previous VF, then stop looking. A
- // vectorized variant of a function expects input in a certain shape
- // -- basically the number of input registers, the number of lanes
- //...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/195519
More information about the llvm-commits
mailing list