[llvm] [LV] Remove legacy setVectorizedCallDecision & co (NFC). (PR #195519)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Sun May 3 04:38:09 PDT 2026
https://github.com/fhahn created https://github.com/llvm/llvm-project/pull/195519
Remove setVectorizedCallDecision & co after being superseded by
https://github.com/llvm/llvm-project/pull/195518.
Note that we still need to retain some of the call cost logic in the
legacy cost model, to compute if scalarization is profitable.
Depends on https://github.com/llvm/llvm-project/pull/195518 (included in
PR)
>From 0370284a9eab6cc8e4fd886827f54617022e9c5e Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sun, 3 May 2026 12:26:03 +0100
Subject: [PATCH 1/2] [VPlan] Move call widening decision to VPlan. (NFCI)
This patch adds a new makeCallWideningDecisions transform which converts
Call VPInstructions to VPWidenCallRecipe/VPWidenIntrinsicRecipe/VPReplicateRecipe
depending on their costs.
To compute the costs, static helpers are introduced to re-use the
existing VPlan cost model logic:
* VPWidenIntrinsicRecipe::getIntrinsicCost
* VPReplicateRecipe::computeScalarCallCost
The cost-model logic is still retained; we assert that the decisions
match to make sure we do not miss any edge cases. The legacy logic will
be removed in a follow-up.
---
.../Transforms/Vectorize/LoopVectorize.cpp | 137 ++++---------
.../Transforms/Vectorize/VPRecipeBuilder.h | 14 +-
llvm/lib/Transforms/Vectorize/VPlan.h | 21 +-
.../Vectorize/VPlanConstruction.cpp | 3 +
llvm/lib/Transforms/Vectorize/VPlanHelpers.h | 23 +++
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 110 ++++++-----
.../Transforms/Vectorize/VPlanTransforms.cpp | 186 ++++++++++++++++++
.../Transforms/Vectorize/VPlanTransforms.h | 6 +
.../VPlan/vplan-print-after-all.ll | 1 +
9 files changed, 339 insertions(+), 162 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 78163b5fe35d5..e46593eb22172 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1312,6 +1312,12 @@ class LoopVectorizationCostModel {
/// trivially hoistable.
bool shouldConsiderInvariant(Value *Op);
+ /// Returns true if \p I has been forced to be scalarized at \p VF.
+ bool isForcedScalar(Instruction *I, ElementCount VF) const {
+ auto FS = ForcedScalars.find(VF);
+ return FS != ForcedScalars.end() && FS->second.contains(I);
+ }
+
private:
unsigned NumPredStores = 0;
@@ -5858,6 +5864,32 @@ uint64_t VPCostContext::getPredBlockCostDivisor(BasicBlock *BB) const {
return CM.getPredBlockCostDivisor(CostKind, BB);
}
+bool VPCostContext::willBeScalarized(Instruction *I, ElementCount VF) const {
+ return CM.isScalarWithPredication(I, VF) ||
+ CM.isUniformAfterVectorization(I, VF) || CM.isForcedScalar(I, VF) ||
+ (VF.isVector() && CM.isProfitableToScalarize(I, VF));
+}
+
+bool VPCostContext::isMaskRequired(Instruction *I) const {
+ return CM.isMaskRequired(I);
+}
+
+std::optional<VPCostContext::CallWideningKind>
+VPCostContext::getLegacyCallKind(CallInst *CI, ElementCount VF) const {
+ if (VF.isScalar())
+ return std::nullopt;
+ switch (CM.getCallWideningDecision(CI, VF).Kind) {
+ case LoopVectorizationCostModel::CM_Scalarize:
+ return CallWideningKind::Scalarize;
+ case LoopVectorizationCostModel::CM_IntrinsicCall:
+ return CallWideningKind::Intrinsic;
+ case LoopVectorizationCostModel::CM_VectorCall:
+ return CallWideningKind::VectorVariant;
+ default:
+ return std::nullopt;
+ }
+}
+
InstructionCost
LoopVectorizationPlanner::precomputeCosts(VPlan &Plan, ElementCount VF,
VPCostContext &CostCtx) const {
@@ -6505,93 +6537,6 @@ VPRecipeBuilder::tryToOptimizeInductionTruncate(VPInstruction *VPI,
Phi, Start, Step, &Plan.getVF(), IndDesc, I, Flags, VPI->getDebugLoc());
}
-VPSingleDefRecipe *VPRecipeBuilder::tryToWidenCall(VPInstruction *VPI,
- VFRange &Range) {
- CallInst *CI = cast<CallInst>(VPI->getUnderlyingInstr());
- bool IsPredicated = LoopVectorizationPlanner::getDecisionAndClampRange(
- [this, CI](ElementCount VF) {
- return CM.isScalarWithPredication(CI, VF);
- },
- Range);
-
- if (IsPredicated)
- return nullptr;
-
- Intrinsic::ID ID = getVectorIntrinsicIDForCall(CI, TLI);
- if (ID && (ID == Intrinsic::assume || ID == Intrinsic::lifetime_end ||
- ID == Intrinsic::lifetime_start || ID == Intrinsic::sideeffect ||
- ID == Intrinsic::pseudoprobe ||
- ID == Intrinsic::experimental_noalias_scope_decl))
- return nullptr;
-
- SmallVector<VPValue *, 4> Ops(VPI->op_begin(),
- VPI->op_begin() + CI->arg_size());
-
- // Is it beneficial to perform intrinsic call compared to lib call?
- bool ShouldUseVectorIntrinsic =
- ID && LoopVectorizationPlanner::getDecisionAndClampRange(
- [&](ElementCount VF) -> bool {
- return CM.getCallWideningDecision(CI, VF).Kind ==
- LoopVectorizationCostModel::CM_IntrinsicCall;
- },
- Range);
- if (ShouldUseVectorIntrinsic)
- return new VPWidenIntrinsicRecipe(*CI, ID, Ops, CI->getType(), *VPI, *VPI,
- VPI->getDebugLoc());
-
- Function *Variant = nullptr;
- std::optional<unsigned> MaskPos;
- // Is better to call a vectorized version of the function than to to scalarize
- // the call?
- auto ShouldUseVectorCall = LoopVectorizationPlanner::getDecisionAndClampRange(
- [&](ElementCount VF) -> bool {
- // The following case may be scalarized depending on the VF.
- // The flag shows whether we can use a usual Call for vectorized
- // version of the instruction.
-
- // If we've found a variant at a previous VF, then stop looking. A
- // vectorized variant of a function expects input in a certain shape
- // -- basically the number of input registers, the number of lanes
- // per register, and whether there's a mask required.
- // We store a pointer to the variant in the VPWidenCallRecipe, so
- // once we have an appropriate variant it's only valid for that VF.
- // This will force a different vplan to be generated for each VF that
- // finds a valid variant.
- if (Variant)
- return false;
- LoopVectorizationCostModel::CallWideningDecision Decision =
- CM.getCallWideningDecision(CI, VF);
- if (Decision.Kind == LoopVectorizationCostModel::CM_VectorCall) {
- Variant = Decision.Variant;
- MaskPos = Decision.MaskPos;
- return true;
- }
-
- return false;
- },
- Range);
- if (ShouldUseVectorCall) {
- if (MaskPos.has_value()) {
- // We have 2 cases that would require a mask:
- // 1) The call needs to be predicated, either due to a conditional
- // in the scalar loop or use of an active lane mask with
- // tail-folding, and we use the appropriate mask for the block.
- // 2) No mask is required for the call instruction, but the only
- // available vector variant at this VF requires a mask, so we
- // synthesize an all-true mask.
- VPValue *Mask = VPI->isMasked() ? VPI->getMask() : Plan.getTrue();
-
- Ops.insert(Ops.begin() + *MaskPos, Mask);
- }
-
- Ops.push_back(VPI->getOperand(VPI->getNumOperandsWithoutMask() - 1));
- return new VPWidenCallRecipe(CI, Variant, Ops, *VPI, *VPI,
- VPI->getDebugLoc());
- }
-
- return nullptr;
-}
-
bool VPRecipeBuilder::shouldWiden(Instruction *I, VFRange &Range) const {
assert((!isa<UncondBrInst, CondBrInst, PHINode, LoadInst, StoreInst>(I)) &&
"Instruction should have been handled earlier");
@@ -6777,10 +6722,12 @@ VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
VFRange &Range) {
assert(!R->isPhi() && "phis must be handled earlier");
// First, check for specific widening recipes that deal with optimizing
- // truncates, calls and memory operations.
+ // truncates and memory operations
+ auto *VPI = cast<VPInstruction>(R);
+ assert(VPI->getOpcode() != Instruction::Call &&
+ "Call should have been handled by makeCallWideningDecisions");
VPRecipeBase *Recipe;
- auto *VPI = cast<VPInstruction>(R);
if (VPI->getOpcode() == Instruction::Trunc &&
(Recipe = tryToOptimizeInductionTruncate(VPI, Range)))
return Recipe;
@@ -6790,9 +6737,6 @@ VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
[&](ElementCount VF) { return VF.isScalar(); }, Range))
return nullptr;
- if (VPI->getOpcode() == Instruction::Call)
- return tryToWidenCall(VPI, Range);
-
Instruction *Instr = R->getUnderlyingInstr();
assert(!is_contained({Instruction::Load, Instruction::Store},
VPI->getOpcode()) &&
@@ -6989,7 +6933,7 @@ LoopVectorizationPlanner::tryToBuildVPlanWithVPRecipes(VPlanPtr Plan,
// Construct wide recipes and apply predication for original scalar
// VPInstructions in the loop.
// ---------------------------------------------------------------------------
- VPRecipeBuilder RecipeBuilder(*Plan, TLI, Legal, CM, Builder);
+ VPRecipeBuilder RecipeBuilder(*Plan, Legal, CM, Builder);
// Scan the body of the loop in a topological order to visit each basic block
// after having visited its predecessor basic blocks.
@@ -7006,6 +6950,9 @@ LoopVectorizationPlanner::tryToBuildVPlanWithVPRecipes(VPlanPtr Plan,
RUN_VPLAN_PASS_NO_VERIFY(VPlanTransforms::makeMemOpWideningDecisions, *Plan,
Range, RecipeBuilder);
+ RUN_VPLAN_PASS_NO_VERIFY(VPlanTransforms::makeCallWideningDecisions, *Plan,
+ Range, RecipeBuilder, CostCtx);
+
// Now process all other blocks and instructions.
for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(RPOT)) {
// Convert input VPInstructions to widened recipes.
@@ -7015,8 +6962,8 @@ LoopVectorizationPlanner::tryToBuildVPlanWithVPRecipes(VPlanPtr Plan,
// transformed.
if (isa<VPWidenCanonicalIVRecipe, VPBlendRecipe, VPReductionRecipe,
VPReplicateRecipe, VPWidenLoadRecipe, VPWidenStoreRecipe,
- VPVectorPointerRecipe, VPVectorEndPointerRecipe,
- VPHistogramRecipe>(&R))
+ VPWidenCallRecipe, VPWidenIntrinsicRecipe, VPVectorPointerRecipe,
+ VPVectorEndPointerRecipe, VPHistogramRecipe>(&R))
continue;
auto *VPI = cast<VPInstruction>(&R);
if (!VPI->getUnderlyingValue())
diff --git a/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h b/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h
index a84c77d614673..aff84cdbd0cf7 100644
--- a/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h
+++ b/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h
@@ -17,7 +17,6 @@ namespace llvm {
class LoopVectorizationLegality;
class LoopVectorizationCostModel;
-class TargetLibraryInfo;
struct HistogramInfo;
struct VFRange;
@@ -26,9 +25,6 @@ class VPRecipeBuilder {
/// The VPlan new recipes are added to.
VPlan &Plan;
- /// Target Library Info.
- const TargetLibraryInfo *TLI;
-
/// The legality analysis.
LoopVectorizationLegality *Legal;
@@ -47,21 +43,15 @@ class VPRecipeBuilder {
VPWidenIntOrFpInductionRecipe *
tryToOptimizeInductionTruncate(VPInstruction *VPI, VFRange &Range);
- /// Handle call instructions. If \p VPI can be widened for \p Range.Start,
- /// return a new VPWidenCallRecipe or VPWidenIntrinsicRecipe. Range.End may be
- /// decreased to ensure same decision from \p Range.Start to \p Range.End.
- VPSingleDefRecipe *tryToWidenCall(VPInstruction *VPI, VFRange &Range);
-
/// Check if \p VPI has an opcode that can be widened and return a
/// VPWidenRecipe if it can. The function should only be called if the
/// cost-model indicates that widening should be performed.
VPWidenRecipe *tryToWiden(VPInstruction *VPI);
public:
- VPRecipeBuilder(VPlan &Plan, const TargetLibraryInfo *TLI,
- LoopVectorizationLegality *Legal,
+ VPRecipeBuilder(VPlan &Plan, LoopVectorizationLegality *Legal,
LoopVectorizationCostModel &CM, VPBuilder &Builder)
- : Plan(Plan), TLI(TLI), Legal(Legal), CM(CM), Builder(Builder) {}
+ : Plan(Plan), Legal(Legal), CM(CM), Builder(Builder) {}
/// Create and return a widened recipe for a non-phi recipe \p R if one can be
/// created within the given VF \p Range.
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 4a5420185224b..98c0ede701a3a 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -70,10 +70,6 @@ class LoopVectorizationCostModel;
struct VPCostContext;
-namespace Intrinsic {
-typedef unsigned ID;
-}
-
using VPlanPtr = std::unique_ptr<VPlan>;
/// \enum UncountableExitStyle
@@ -1947,6 +1943,12 @@ class VPWidenIntrinsicRecipe : public VPRecipeWithIRFlags, public VPIRMetadata {
/// Produce a widened version of the vector intrinsic.
LLVM_ABI_FOR_TEST void execute(VPTransformState &State) override;
+ /// Compute the cost of a vector intrinsic with \p ID and \p Operands.
+ static InstructionCost
+ computeIntrinsicCost(Intrinsic::ID ID, ArrayRef<const VPValue *> Operands,
+ const VPRecipeWithIRFlags &R, ElementCount VF,
+ VPCostContext &Ctx);
+
/// Return the cost of this vector intrinsic.
LLVM_ABI_FOR_TEST InstructionCost
computeCost(ElementCount VF, VPCostContext &Ctx) const override;
@@ -2017,6 +2019,10 @@ class LLVM_ABI_FOR_TEST VPWidenCallRecipe : public VPRecipeWithIRFlags,
InstructionCost computeCost(ElementCount VF,
VPCostContext &Ctx) const override;
+ /// Return the cost of widening a call using the vector function \p Variant.
+ static InstructionCost computeVectorCallCost(Function *Variant,
+ VPCostContext &Ctx);
+
Function *getCalledScalarFunction() const {
return cast<Function>(getOperand(getNumOperands() - 1)->getLiveInIRValue());
}
@@ -3228,6 +3234,13 @@ class LLVM_ABI_FOR_TEST VPReplicateRecipe : public VPRecipeWithIRFlags,
InstructionCost computeCost(ElementCount VF,
VPCostContext &Ctx) const override;
+ /// Return the cost of scalarizing a call to \p CalledFn with argument
+ /// operands \p ArgOps for a given \p VF.
+ static InstructionCost
+ computeScalarCallCost(Function *CalledFn, Type *ResultTy,
+ ArrayRef<const VPValue *> ArgOps, bool IsSingleScalar,
+ ElementCount VF, VPCostContext &Ctx);
+
bool isSingleScalar() const { return IsSingleScalar; }
bool isPredicated() const { return IsPredicated; }
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index e20d5d947ac54..ebdac8b1d7400 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -12,6 +12,7 @@
//===----------------------------------------------------------------------===//
#include "LoopVectorizationPlanner.h"
+#include "VPRecipeBuilder.h"
#include "VPlan.h"
#include "VPlanAnalysis.h"
#include "VPlanCFG.h"
@@ -26,6 +27,7 @@
#include "llvm/Analysis/OptimizationRemarkEmitter.h"
#include "llvm/Analysis/ScalarEvolution.h"
#include "llvm/Analysis/ScalarEvolutionExpressions.h"
+#include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/IR/InstrTypes.h"
#include "llvm/IR/MDBuilder.h"
@@ -37,6 +39,7 @@
using namespace llvm;
using namespace VPlanPatternMatch;
+using namespace SCEVPatternMatch;
namespace {
// Class that is used to build the plain CFG for the incoming IR.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanHelpers.h b/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
index 1b11516c497ab..ff4b2d7f32964 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
@@ -30,6 +30,7 @@ namespace llvm {
class AssumptionCache;
class BasicBlock;
+class CallInst;
class DominatorTree;
class InnerLoopVectorizer;
class IRBuilderBase;
@@ -41,6 +42,10 @@ class VPRegionBlock;
class VPlan;
class Value;
+namespace Intrinsic {
+typedef unsigned ID;
+}
+
/// Returns a calculation for the total number of elements for a given \p VF.
/// For fixed width vectors this value is a constant, whereas for scalable
/// vectors it is an expression determined at runtime.
@@ -324,6 +329,9 @@ struct VPTransformState {
/// Struct to hold various analysis needed for cost computations.
struct VPCostContext {
+ /// Choice for how to widen a call at a given VF.
+ enum class CallWideningKind { Scalarize, Intrinsic, VectorVariant };
+
const TargetTransformInfo &TTI;
const TargetLibraryInfo &TLI;
VPTypeAnalysis Types;
@@ -356,6 +364,17 @@ struct VPCostContext {
/// Forwards to LoopVectorizationCostModel::getPredBlockCostDivisor.
uint64_t getPredBlockCostDivisor(BasicBlock *BB) const;
+ /// Returns true if \p I is known to be scalarized at \p VF.
+ bool willBeScalarized(Instruction *I, ElementCount VF) const;
+
+ /// Forwards to LoopVectorizationCostModel::isMaskRequired.
+ bool isMaskRequired(Instruction *I) const;
+
+ /// Returns the legacy call widening decision for \p CI at \p VF, or
+ /// std::nullopt if none was recorded. Used only in asserts.
+ std::optional<CallWideningKind> getLegacyCallKind(CallInst *CI,
+ ElementCount VF) const;
+
/// Returns the OperandInfo for \p V, if it is a live-in.
TargetTransformInfo::OperandValueInfo getOperandInfo(VPValue *V) const;
@@ -373,6 +392,10 @@ struct VPCostContext {
/// Returns true if an artificially high cost for emulated masked memrefs
/// should be used.
bool useEmulatedMaskMemRefHack(const VPReplicateRecipe *R, ElementCount VF);
+
+ /// Returns true if \p ID is a pseudo intrinsic that is dropped via
+ /// scalarization rather than widened.
+ static bool isFreeScalarIntrinsic(Intrinsic::ID ID);
};
/// This class can be used to assign names to VPValues. For VPValues without
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 2225dfa310c6c..71313334b5196 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -19,6 +19,7 @@
#include "VPlanUtils.h"
#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/SmallVector.h"
+#include "llvm/ADT/SmallVectorExtras.h"
#include "llvm/ADT/Twine.h"
#include "llvm/Analysis/AssumptionCache.h"
#include "llvm/Analysis/IVDescriptors.h"
@@ -1855,6 +1856,11 @@ void VPWidenCallRecipe::execute(VPTransformState &State) {
InstructionCost VPWidenCallRecipe::computeCost(ElementCount VF,
VPCostContext &Ctx) const {
+ return computeVectorCallCost(Variant, Ctx);
+}
+
+InstructionCost VPWidenCallRecipe::computeVectorCallCost(Function *Variant,
+ VPCostContext &Ctx) {
return Ctx.TTI.getCallInstrCost(nullptr, Variant->getReturnType(),
Variant->getFunctionType()->params(),
Ctx.CostKind);
@@ -1940,12 +1946,17 @@ void VPWidenIntrinsicRecipe::execute(VPTransformState &State) {
State.set(this, V);
}
-/// Compute the cost for the intrinsic \p ID with \p Operands, produced by \p R.
-static InstructionCost getCostForIntrinsics(Intrinsic::ID ID,
- ArrayRef<const VPValue *> Operands,
- const VPRecipeWithIRFlags &R,
- ElementCount VF,
- VPCostContext &Ctx) {
+bool VPCostContext::isFreeScalarIntrinsic(Intrinsic::ID ID) {
+ return is_contained({Intrinsic::assume, Intrinsic::lifetime_end,
+ Intrinsic::lifetime_start, Intrinsic::sideeffect,
+ Intrinsic::pseudoprobe,
+ Intrinsic::experimental_noalias_scope_decl},
+ ID);
+}
+
+InstructionCost VPWidenIntrinsicRecipe::computeIntrinsicCost(
+ Intrinsic::ID ID, ArrayRef<const VPValue *> Operands,
+ const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx) {
Type *ScalarRetTy = Ctx.Types.inferScalarType(&R);
// Skip the reverse operation cost for the mask.
// FIXME: Remove this once redundant mask reverse operations can be eliminated
@@ -1973,12 +1984,11 @@ static InstructionCost getCostForIntrinsics(Intrinsic::ID ID,
}
Type *RetTy = VF.isVector() ? toVectorizedTy(ScalarRetTy, VF) : ScalarRetTy;
- SmallVector<Type *> ParamTys;
- for (const VPValue *Op : Operands) {
- ParamTys.push_back(VF.isVector()
- ? toVectorTy(Ctx.Types.inferScalarType(Op), VF)
- : Ctx.Types.inferScalarType(Op));
- }
+ SmallVector<Type *> ParamTys =
+ map_to_vector(Operands, [&](const VPValue *Op) {
+ Type *Ty = Ctx.Types.inferScalarType(Op);
+ return VF.isVector() ? toVectorTy(Ty, VF) : Ty;
+ });
// TODO: Rework TTI interface to avoid reliance on underlying IntrinsicInst.
IntrinsicCostAttributes CostAttrs(
@@ -1991,7 +2001,7 @@ static InstructionCost getCostForIntrinsics(Intrinsic::ID ID,
InstructionCost VPWidenIntrinsicRecipe::computeCost(ElementCount VF,
VPCostContext &Ctx) const {
SmallVector<const VPValue *> ArgOps(operands());
- return getCostForIntrinsics(VectorIntrinsicID, ArgOps, *this, VF, Ctx);
+ return computeIntrinsicCost(VectorIntrinsicID, ArgOps, *this, VF, Ctx);
}
StringRef VPWidenIntrinsicRecipe::getIntrinsicName() const {
@@ -3415,45 +3425,10 @@ InstructionCost VPReplicateRecipe::computeCost(ElementCount VF,
case Instruction::Call: {
auto *CalledFn =
cast<Function>(getOperand(getNumOperands() - 1)->getLiveInIRValue());
-
- SmallVector<const VPValue *> ArgOps(drop_end(operands()));
- SmallVector<Type *, 4> Tys;
- for (const VPValue *ArgOp : ArgOps)
- Tys.push_back(Ctx.Types.inferScalarType(ArgOp));
-
- if (CalledFn->isIntrinsic())
- // Various pseudo-intrinsics with costs of 0 are scalarized instead of
- // vectorized via VPWidenIntrinsicRecipe. Return 0 for them early.
- switch (CalledFn->getIntrinsicID()) {
- case Intrinsic::assume:
- case Intrinsic::lifetime_end:
- case Intrinsic::lifetime_start:
- case Intrinsic::sideeffect:
- case Intrinsic::pseudoprobe:
- case Intrinsic::experimental_noalias_scope_decl: {
- assert(getCostForIntrinsics(CalledFn->getIntrinsicID(), ArgOps, *this,
- ElementCount::getFixed(1), Ctx) == 0 &&
- "scalarizing intrinsic should be free");
- return InstructionCost(0);
- }
- default:
- break;
- }
-
Type *ResultTy = Ctx.Types.inferScalarType(this);
- InstructionCost ScalarCallCost =
- Ctx.TTI.getCallInstrCost(CalledFn, ResultTy, Tys, Ctx.CostKind);
- if (isSingleScalar()) {
- if (CalledFn->isIntrinsic())
- ScalarCallCost = std::min(
- ScalarCallCost,
- getCostForIntrinsics(CalledFn->getIntrinsicID(), ArgOps, *this,
- ElementCount::getFixed(1), Ctx));
- return ScalarCallCost;
- }
-
- return ScalarCallCost * VF.getFixedValue() +
- Ctx.getScalarizationOverhead(ResultTy, ArgOps, VF);
+ SmallVector<const VPValue *> ArgOps(drop_end(operands()));
+ return computeScalarCallCost(CalledFn, ResultTy, ArgOps, isSingleScalar(),
+ VF, Ctx);
}
case Instruction::Add:
case Instruction::Sub:
@@ -3634,6 +3609,39 @@ InstructionCost VPReplicateRecipe::computeCost(ElementCount VF,
return Ctx.getLegacyCost(UI, VF);
}
+InstructionCost VPReplicateRecipe::computeScalarCallCost(
+ Function *CalledFn, Type *ResultTy, ArrayRef<const VPValue *> ArgOps,
+ bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx) {
+ SmallVector<Type *, 4> Tys = map_to_vector<4>(
+ ArgOps, [&](const VPValue *Op) { return Ctx.Types.inferScalarType(Op); });
+
+ Intrinsic::ID IntrinID = CalledFn->getIntrinsicID();
+ auto GetIntrinsicCost = [&] {
+ return Ctx.TTI.getIntrinsicInstrCost(
+ IntrinsicCostAttributes(IntrinID, ResultTy, Tys), Ctx.CostKind);
+ };
+
+ if (IntrinID && VPCostContext::isFreeScalarIntrinsic(IntrinID)) {
+ assert(GetIntrinsicCost() == 0 && "scalarizing intrinsic should be free");
+ return InstructionCost(0);
+ }
+
+ InstructionCost ScalarCallCost =
+ Ctx.TTI.getCallInstrCost(CalledFn, ResultTy, Tys, Ctx.CostKind);
+ if (IsSingleScalar) {
+ if (IntrinID)
+ ScalarCallCost = std::min(ScalarCallCost, GetIntrinsicCost());
+ return ScalarCallCost;
+ }
+
+ // Scalarization overhead is undefined for scalable VFs.
+ if (VF.isScalable())
+ return InstructionCost::getInvalid();
+
+ return ScalarCallCost * VF.getFixedValue() +
+ Ctx.getScalarizationOverhead(ResultTy, ArgOps, VF);
+}
+
#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
void VPReplicateRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
VPSlotTracker &SlotTracker) const {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 262f4798b3d63..85846626e2159 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -6496,3 +6496,189 @@ void VPlanTransforms::makeMemOpWideningDecisions(
ReplaceWith(Recipe);
}
}
+
+/// Returns true if \p Info's parameter kinds are compatible with \p Args.
+static bool areVFParamsOk(const VFInfo &Info, ArrayRef<VPValue *> Args,
+ PredicatedScalarEvolution &PSE, const Loop *L) {
+ return all_of(Info.Shape.Parameters, [&](VFParameter Param) {
+ switch (Param.ParamKind) {
+ case VFParamKind::Vector:
+ case VFParamKind::GlobalPredicate:
+ return true;
+ case VFParamKind::OMP_Uniform:
+ return PSE.getSE()->isLoopInvariant(
+ vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L), L);
+ case VFParamKind::OMP_Linear:
+ return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
+ m_scev_AffineAddRec(
+ m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
+ m_SpecificLoop(L)));
+ default:
+ return false;
+ }
+ });
+}
+
+/// Find a vector variant of \p CI for \p VF, respecting \p MaskRequired.
+/// Returns the variant function and the position of its mask parameter
+/// (if any), or {nullptr, std::nullopt}.
+static std::pair<Function *, std::optional<unsigned>>
+findVectorVariant(CallInst *CI, ArrayRef<VPValue *> Args, ElementCount VF,
+ bool MaskRequired, PredicatedScalarEvolution &PSE,
+ const Loop *L) {
+ if (CI->isNoBuiltin())
+ return {nullptr, std::nullopt};
+ auto Mappings = VFDatabase::getMappings(*CI);
+ const auto *It = find_if(Mappings, [&](const VFInfo &Info) {
+ return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
+ areVFParamsOk(Info, Args, PSE, L);
+ });
+ if (It == Mappings.end())
+ return {nullptr, std::nullopt};
+ if (Function *VecFunc = CI->getModule()->getFunction(It->VectorName))
+ return {VecFunc, It->getParamIndexForOptionalMask()};
+ return {nullptr, std::nullopt};
+}
+
+namespace {
+/// The outcome of choosing how to widen a call at a given VF.
+struct CallWideningDecision {
+ using KindTy = VPCostContext::CallWideningKind;
+ KindTy Kind = KindTy::Scalarize;
+ /// Set when Kind == VectorVariant.
+ Function *Variant = nullptr;
+ /// Position of the mask parameter for \p Variant, if any.
+ std::optional<unsigned> MaskPos;
+};
+} // namespace
+
+/// Pick the cheapest widening for the call \p VPI at \p VF among scalarization,
+/// vector intrinsic, and vector library variant.
+static CallWideningDecision decideCallWidening(VPInstruction &VPI,
+ ArrayRef<VPValue *> Ops,
+ ElementCount VF,
+ VPCostContext &CostCtx) {
+ auto *CI = cast<CallInst>(VPI.getUnderlyingInstr());
+ auto *CalledFn = cast<Function>(
+ VPI.getOperand(VPI.getNumOperandsWithoutMask() - 1)->getLiveInIRValue());
+ Type *ResultTy = CostCtx.Types.inferScalarType(&VPI);
+ Intrinsic::ID ID = getVectorIntrinsicIDForCall(CI, &CostCtx.TLI);
+ bool MaskRequired = CostCtx.isMaskRequired(CI);
+
+ // Pseudo intrinsics (assume, lifetime, ...) are always scalarized.
+ if (ID && VPCostContext::isFreeScalarIntrinsic(ID))
+ return {};
+
+ InstructionCost ScalarCost = VPReplicateRecipe::computeScalarCallCost(
+ CalledFn, ResultTy, Ops,
+ /*IsSingleScalar=*/false, VF, CostCtx);
+
+ auto [VecFunc, MaskPos] =
+ findVectorVariant(CI, Ops, VF, MaskRequired, CostCtx.PSE, CostCtx.L);
+ InstructionCost VecCallCost = InstructionCost::getInvalid();
+ if (VecFunc)
+ VecCallCost = VPWidenCallRecipe::computeVectorCallCost(VecFunc, CostCtx);
+
+ // Prefer the intrinsic if it is at least as cheap as scalarizing and any
+ // available vector variant.
+ if (ID) {
+ InstructionCost IntrinsicCost =
+ VPWidenIntrinsicRecipe::computeIntrinsicCost(ID, Ops, VPI, VF, CostCtx);
+ if (IntrinsicCost.isValid() && ScalarCost >= IntrinsicCost &&
+ (!VecFunc || VecCallCost >= IntrinsicCost))
+ return {CallWideningDecision::KindTy::Intrinsic, nullptr, std::nullopt};
+ }
+
+ // Otherwise, use a vector library variant when it beats scalarizing.
+ if (VecFunc && ScalarCost >= VecCallCost)
+ return {CallWideningDecision::KindTy::VectorVariant, VecFunc, MaskPos};
+
+ return {};
+}
+
+void VPlanTransforms::makeCallWideningDecisions(VPlan &Plan, VFRange &Range,
+ VPRecipeBuilder &RecipeBuilder,
+ VPCostContext &CostCtx) {
+ bool IsScalarVPlan = LoopVectorizationPlanner::getDecisionAndClampRange(
+ [](ElementCount VF) { return VF.isScalar(); }, Range);
+
+ SmallVector<VPInstruction *, 8> ToErase;
+ for (VPBasicBlock *VPBB :
+ VPBlockUtils::blocksOnly<VPBasicBlock>(vp_depth_first_shallow(
+ Plan.getVectorLoopRegion()->getEntryBasicBlock()))) {
+ for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
+ auto *VPI = dyn_cast<VPInstruction>(&R);
+ if (!VPI || !VPI->getUnderlyingValue() ||
+ VPI->getOpcode() != Instruction::Call)
+ continue;
+
+ // Scalar VPlans and known-scalarized calls fall through to replication.
+ auto *CI = cast<CallInst>(VPI->getUnderlyingInstr());
+ bool KeepScalar =
+ IsScalarVPlan ||
+ LoopVectorizationPlanner::getDecisionAndClampRange(
+ [&](ElementCount VF) { return CostCtx.willBeScalarized(CI, VF); },
+ Range);
+
+ VPSingleDefRecipe *Recipe = nullptr;
+ CallWideningDecision Decision;
+ if (!KeepScalar) {
+ SmallVector<VPValue *, 4> Ops(VPI->op_begin(),
+ VPI->op_begin() + CI->arg_size());
+
+ // Pick the cheapest widening at Range.Start, then clamp the range.
+ Decision = decideCallWidening(*VPI, Ops, Range.Start, CostCtx);
+ LoopVectorizationPlanner::getDecisionAndClampRange(
+ [&](ElementCount VF) {
+ CallWideningDecision D =
+ decideCallWidening(*VPI, Ops, VF, CostCtx);
+ return D.Kind == Decision.Kind && D.Variant == Decision.Variant;
+ },
+ Range);
+
+ switch (Decision.Kind) {
+ case CallWideningDecision::KindTy::Intrinsic: {
+ Intrinsic::ID ID = getVectorIntrinsicIDForCall(CI, &CostCtx.TLI);
+ Type *ResultTy = CostCtx.Types.inferScalarType(VPI);
+ Recipe = new VPWidenIntrinsicRecipe(*CI, ID, Ops, ResultTy, *VPI,
+ *VPI, VPI->getDebugLoc());
+ break;
+ }
+ case CallWideningDecision::KindTy::VectorVariant: {
+ if (Decision.MaskPos) {
+ VPValue *Mask = VPI->isMasked() ? VPI->getMask() : Plan.getTrue();
+ Ops.insert(Ops.begin() + *Decision.MaskPos, Mask);
+ }
+ Ops.push_back(VPI->getOperand(VPI->getNumOperandsWithoutMask() - 1));
+ Recipe =
+ new VPWidenCallRecipe(VPI->getUnderlyingValue(), Decision.Variant,
+ Ops, *VPI, *VPI, VPI->getDebugLoc());
+ break;
+ }
+ case CallWideningDecision::KindTy::Scalarize:
+ break;
+ }
+ }
+
+ if (!Recipe)
+ Recipe = RecipeBuilder.handleReplication(VPI, Range);
+
+ assert(all_of(Range,
+ [&](ElementCount VF) {
+ Intrinsic::ID IID =
+ getVectorIntrinsicIDForCall(CI, &CostCtx.TLI);
+ if (IID && VPCostContext::isFreeScalarIntrinsic(IID))
+ return true;
+ auto Legacy = CostCtx.getLegacyCallKind(CI, VF);
+ return !Legacy || *Legacy == Decision.Kind;
+ }) &&
+ "VPlan call widening decision must match legacy decision");
+
+ Recipe->insertBefore(VPI);
+ VPI->replaceAllUsesWith(Recipe->getVPSingleValue());
+ ToErase.push_back(VPI);
+ }
+ }
+ for (VPInstruction *VPI : ToErase)
+ VPI->eraseFromParent();
+}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 6e11de399c406..28a66fd34b7e6 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -539,6 +539,12 @@ struct VPlanTransforms {
/// recipes. Non load/store input instructions are left unchanged.
static void makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
VPRecipeBuilder &RecipeBuilder);
+
+ /// Convert call VPInstructions in \p Plan into widened call, vector
+ /// intrinsic or replicate recipes based on a cost comparison via \p CostCtx.
+ static void makeCallWideningDecisions(VPlan &Plan, VFRange &Range,
+ VPRecipeBuilder &RecipeBuilder,
+ VPCostContext &CostCtx);
};
} // namespace llvm
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll
index 4bc9a8d96e542..a07b4c8723792 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll
@@ -13,6 +13,7 @@
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::introduceMasksAndLinearize
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::createInLoopReductionRecipes
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::makeMemOpWideningDecisions
+; CHECK: VPlan for loop in 'foo' after VPlanTransforms::makeCallWideningDecisions
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::adjustFirstOrderRecurrenceMiddleUsers
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::clearReductionWrapFlags
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::optimizeFindIVReductions
>From 4bc1a840f236d329e7b3fcecdd8476eb694a24e6 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sun, 3 May 2026 12:30:59 +0100
Subject: [PATCH 2/2] [LV] Remove legacy setVectorizedCallDecision & co (NFC).
Remove setVectorizedCallDecision & co after being superseded by
https://github.com/llvm/llvm-project/pull/195518.
Note that we still need to retain some of the call cost logic in the
legacy cost model, to compute if scalarization is profitable.
Depends on https://github.com/llvm/llvm-project/pull/195518 (included in
PR)
---
.../Transforms/Vectorize/LoopVectorize.cpp | 319 ++++--------------
llvm/lib/Transforms/Vectorize/VPlanHelpers.h | 8 -
.../Transforms/Vectorize/VPlanTransforms.cpp | 13 +-
3 files changed, 70 insertions(+), 270 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index e46593eb22172..eec04001ade14 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -847,13 +847,6 @@ class LoopVectorizationCostModel {
/// avoid redundant calculations.
void setCostBasedWideningDecision(ElementCount VF);
- /// A call may be vectorized in different ways depending on whether we have
- /// vectorized variants available and whether the target supports masking.
- /// This function analyzes all calls in the function at the supplied VF,
- /// makes a decision based on the costs of available options, and stores that
- /// decision in a map for use in planning and plan execution.
- void setVectorizedCallDecision(ElementCount VF);
-
/// Collect values we want to ignore in the cost model.
void collectValuesToIgnore();
@@ -928,8 +921,6 @@ class LoopVectorizationCostModel {
CM_Interleave,
CM_GatherScatter,
CM_Scalarize,
- CM_VectorCall,
- CM_IntrinsicCall
};
/// Save vectorization decision \p W and \p Cost taken by the cost model for
@@ -990,31 +981,6 @@ class LoopVectorizationCostModel {
return WideningDecisions[InstOnVF].second;
}
- struct CallWideningDecision {
- InstWidening Kind;
- Function *Variant;
- Intrinsic::ID IID;
- std::optional<unsigned> MaskPos;
- InstructionCost Cost;
- };
-
- void setCallWideningDecision(CallInst *CI, ElementCount VF, InstWidening Kind,
- Function *Variant, Intrinsic::ID IID,
- std::optional<unsigned> MaskPos,
- InstructionCost Cost) {
- assert(!VF.isScalar() && "Expected vector VF");
- CallWideningDecisions[{CI, VF}] = {Kind, Variant, IID, MaskPos, Cost};
- }
-
- CallWideningDecision getCallWideningDecision(CallInst *CI,
- ElementCount VF) const {
- assert(!VF.isScalar() && "Expected vector VF");
- auto I = CallWideningDecisions.find({CI, VF});
- if (I == CallWideningDecisions.end())
- return {CM_Unknown, nullptr, Intrinsic::not_intrinsic, std::nullopt, 0};
- return I->second;
- }
-
/// Return True if instruction \p I is an optimizable truncate whose operand
/// is an induction variable. Such a truncate will be removed by adding a new
/// induction variable with the destination type.
@@ -1058,7 +1024,6 @@ class LoopVectorizationCostModel {
return;
setCostBasedWideningDecision(VF);
collectLoopUniforms(VF);
- setVectorizedCallDecision(VF);
collectLoopScalars(VF);
collectInstsToScalarize(VF);
}
@@ -1279,7 +1244,6 @@ class LoopVectorizationCostModel {
/// Invalidates decisions already taken by the cost model.
void invalidateCostModelingDecisions() {
WideningDecisions.clear();
- CallWideningDecisions.clear();
Uniforms.clear();
Scalars.clear();
}
@@ -1427,20 +1391,13 @@ class LoopVectorizationCostModel {
DecisionList WideningDecisions;
- using CallDecisionList =
- DenseMap<std::pair<CallInst *, ElementCount>, CallWideningDecision>;
-
- CallDecisionList CallWideningDecisions;
-
/// Returns true if \p V is expected to be vectorized and it needs to be
/// extracted.
bool needsExtract(Value *V, ElementCount VF) const {
Instruction *I = dyn_cast<Instruction>(V);
if (VF.isScalar() || !I || !TheLoop->contains(I) ||
TheLoop->isLoopInvariant(I) ||
- getWideningDecision(I, VF) == CM_Scalarize ||
- (isa<CallInst>(I) &&
- getCallWideningDecision(cast<CallInst>(I), VF).Kind == CM_Scalarize))
+ getWideningDecision(I, VF) == CM_Scalarize)
return false;
// Assume we can vectorize V (and hence we need extraction) if the
@@ -2083,32 +2040,74 @@ static unsigned estimateElementCount(ElementCount VF,
return EstimatedVF;
}
+/// Returns true iff \p CI has a library vector variant usable at \p VF: a
+/// mapping with matching VF, masked if required, whose vector function is
+/// declared in the module. Such variants are priced by
+/// VPWidenCallRecipe::computeCost rather than by scalarization.
+static bool hasVectorLibraryVariantFor(const CallInst &CI, ElementCount VF,
+ bool MaskRequired,
+ const TargetLibraryInfo *TLI) {
+ if (!TLI || CI.isNoBuiltin())
+ return false;
+ for (const VFInfo &Info : VFDatabase::getMappings(CI)) {
+ if (Info.Shape.VF != VF)
+ continue;
+ if (MaskRequired && !Info.isMasked())
+ continue;
+ if (CI.getModule()->getFunction(Info.VectorName))
+ return true;
+ }
+ return false;
+}
+
InstructionCost
LoopVectorizationCostModel::getVectorCallCost(CallInst *CI,
ElementCount VF) const {
- // We only need to calculate a cost if the VF is scalar; for actual vectors
- // we should already have a pre-calculated cost at each VF.
- if (!VF.isScalar())
- return getCallWideningDecision(CI, VF).Cost;
-
Type *RetTy = CI->getType();
- if (RecurrenceDescriptor::isFMulAddIntrinsic(CI))
- if (auto RedCost = getReductionPatternCost(CI, VF, RetTy))
- return *RedCost;
-
- SmallVector<Type *, 4> Tys;
- for (auto &ArgOp : CI->args())
- Tys.push_back(ArgOp->getType());
- InstructionCost ScalarCallCost = TTI.getCallInstrCost(
- CI->getCalledFunction(), RetTy, Tys, Config.CostKind);
+ // Scalar VF: pick the cheaper of the scalar call and any matching vector
+ // intrinsic lowering. In-loop fmuladd reductions are priced specially.
+ if (VF.isScalar()) {
+ if (RecurrenceDescriptor::isFMulAddIntrinsic(CI))
+ if (auto RedCost = getReductionPatternCost(CI, VF, RetTy))
+ return *RedCost;
+
+ SmallVector<Type *, 4> Tys;
+ for (Value *Arg : CI->args())
+ Tys.push_back(Arg->getType());
+ InstructionCost ScalarCallCost = TTI.getCallInstrCost(
+ CI->getCalledFunction(), RetTy, Tys, Config.CostKind);
+
+ if (getVectorIntrinsicIDForCall(CI, TLI))
+ return std::min(ScalarCallCost, getVectorIntrinsicCost(CI, VF));
+ return ScalarCallCost;
+ }
+
+ // Vector VF: compare scalarization against any matching vector intrinsic
+ // lowering. Vector library variants are priced by
+ // VPWidenCallRecipe::computeCost and should not reach this function.
+ assert(!hasVectorLibraryVariantFor(*CI, VF, isMaskRequired(CI), TLI) &&
+ "getVectorCallCost does not price vector library variants");
+
+ // Scalarization is only meaningful for fixed VFs.
+ InstructionCost Cost = InstructionCost::getInvalid();
+ if (VF.isFixed()) {
+ SmallVector<Type *, 4> Tys;
+ for (Value *Arg : CI->args())
+ Tys.push_back(Arg->getType());
+ InstructionCost ScalarCallCost = TTI.getCallInstrCost(
+ CI->getCalledFunction(), RetTy, Tys, Config.CostKind);
+ Cost = ScalarCallCost * VF.getKnownMinValue() +
+ getScalarizationOverhead(CI, VF);
+ }
- // If this is an intrinsic we may have a lower cost for it.
if (getVectorIntrinsicIDForCall(CI, TLI)) {
InstructionCost IntrinsicCost = getVectorIntrinsicCost(CI, VF);
- return std::min(ScalarCallCost, IntrinsicCost);
+ if (IntrinsicCost.isValid() && (!Cost.isValid() || IntrinsicCost <= Cost))
+ Cost = IntrinsicCost;
}
- return ScalarCallCost;
+
+ return Cost;
}
static Type *maybeVectorizeType(Type *Ty, ElementCount VF) {
@@ -2372,10 +2371,16 @@ bool LoopVectorizationCostModel::isScalarWithPredication(Instruction *I,
switch(I->getOpcode()) {
default:
return true;
- case Instruction::Call:
+ case Instruction::Call: {
if (VF.isScalar())
return true;
- return getCallWideningDecision(cast<CallInst>(I), VF).Kind == CM_Scalarize;
+ CallInst *CI = cast<CallInst>(I);
+ // A vector intrinsic lowering is always preferred over scalarization.
+ if (getVectorIntrinsicIDForCall(CI, TLI))
+ return false;
+ // A matching vector library variant also avoids scalarization.
+ return !hasVectorLibraryVariantFor(*CI, VF, isMaskRequired(CI), TLI);
+ }
case Instruction::Load:
case Instruction::Store: {
auto *Ptr = getLoadStorePointerOperand(I);
@@ -2925,8 +2930,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
return FixedScalableVFPair::getNone();
}
- assert(WideningDecisions.empty() && CallWideningDecisions.empty() &&
- Uniforms.empty() && Scalars.empty() &&
+ assert(WideningDecisions.empty() && Uniforms.empty() && Scalars.empty() &&
"No cost-modeling decisions should have been taken at this point");
switch (EpilogueLoweringStatus) {
@@ -4065,15 +4069,6 @@ void LoopVectorizationCostModel::collectInstsToScalarize(ElementCount VF) {
computePredInstDiscount(&I, ScalarCosts, VF) >= 0) {
for (const auto &[I, IC] : ScalarCosts)
ScalarCostsVF.insert({I, IC});
- // Check if we decided to scalarize a call. If so, update the widening
- // decision of the call to CM_Scalarize with the computed scalar cost.
- for (const auto &[I, Cost] : ScalarCosts) {
- auto *CI = dyn_cast<CallInst>(I);
- if (!CI || !CallWideningDecisions.contains({CI, VF}))
- continue;
- CallWideningDecisions[{CI, VF}].Kind = CM_Scalarize;
- CallWideningDecisions[{CI, VF}].Cost = Cost;
- }
}
// Remember that BB will remain after vectorization.
PredicatedBBsAfterVectorization[VF].insert(BB);
@@ -4933,163 +4928,6 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
}
}
-void LoopVectorizationCostModel::setVectorizedCallDecision(ElementCount VF) {
- assert(!VF.isScalar() &&
- "Trying to set a vectorization decision for a scalar VF");
-
- auto ForcedScalar = ForcedScalars.find(VF);
- for (BasicBlock *BB : TheLoop->blocks()) {
- // For each instruction in the old loop.
- for (Instruction &I : *BB) {
- CallInst *CI = dyn_cast<CallInst>(&I);
-
- if (!CI)
- continue;
-
- InstructionCost ScalarCost = InstructionCost::getInvalid();
- InstructionCost VectorCost = InstructionCost::getInvalid();
- InstructionCost IntrinsicCost = InstructionCost::getInvalid();
- Function *ScalarFunc = CI->getCalledFunction();
- Type *ScalarRetTy = CI->getType();
- SmallVector<Type *, 4> Tys, ScalarTys;
- for (auto &ArgOp : CI->args())
- ScalarTys.push_back(ArgOp->getType());
-
- // Estimate cost of scalarized vector call. The source operands are
- // assumed to be vectors, so we need to extract individual elements from
- // there, execute VF scalar calls, and then gather the result into the
- // vector return value.
- if (VF.isFixed()) {
- InstructionCost ScalarCallCost = TTI.getCallInstrCost(
- ScalarFunc, ScalarRetTy, ScalarTys, Config.CostKind);
-
- // Compute costs of unpacking argument values for the scalar calls and
- // packing the return values to a vector.
- InstructionCost ScalarizationCost = getScalarizationOverhead(CI, VF);
- ScalarCost = ScalarCallCost * VF.getKnownMinValue() + ScalarizationCost;
- } else {
- // There is no point attempting to calculate the scalar cost for a
- // scalable VF as we know it will be Invalid.
- assert(!getScalarizationOverhead(CI, VF).isValid() &&
- "Unexpected valid cost for scalarizing scalable vectors");
- ScalarCost = InstructionCost::getInvalid();
- }
-
- // Honor ForcedScalars and UniformAfterVectorization decisions.
- // TODO: For calls, it might still be more profitable to widen. Use
- // VPlan-based cost model to compare different options.
- if (VF.isVector() && ((ForcedScalar != ForcedScalars.end() &&
- ForcedScalar->second.contains(CI)) ||
- isUniformAfterVectorization(CI, VF))) {
- setCallWideningDecision(CI, VF, CM_Scalarize, nullptr,
- Intrinsic::not_intrinsic, std::nullopt,
- ScalarCost);
- continue;
- }
-
- bool MaskRequired = isMaskRequired(CI);
- // Compute corresponding vector type for return value and arguments.
- Type *RetTy = toVectorizedTy(ScalarRetTy, VF);
- for (Type *ScalarTy : ScalarTys)
- Tys.push_back(toVectorizedTy(ScalarTy, VF));
-
- // An in-loop reduction using an fmuladd intrinsic is a special case;
- // we don't want the normal cost for that intrinsic.
- if (RecurrenceDescriptor::isFMulAddIntrinsic(CI))
- if (auto RedCost = getReductionPatternCost(CI, VF, RetTy)) {
- setCallWideningDecision(CI, VF, CM_IntrinsicCall, nullptr,
- getVectorIntrinsicIDForCall(CI, TLI),
- std::nullopt, *RedCost);
- continue;
- }
-
- // Find the cost of vectorizing the call, if we can find a suitable
- // vector variant of the function.
- VFInfo FuncInfo;
- Function *VecFunc = nullptr;
- // Search through any available variants for one we can use at this VF.
- for (VFInfo &Info : VFDatabase::getMappings(*CI)) {
- // Must match requested VF.
- if (Info.Shape.VF != VF)
- continue;
-
- // Must take a mask argument if one is required
- if (MaskRequired && !Info.isMasked())
- continue;
-
- // Check that all parameter kinds are supported
- bool ParamsOk = true;
- for (VFParameter Param : Info.Shape.Parameters) {
- switch (Param.ParamKind) {
- case VFParamKind::Vector:
- break;
- case VFParamKind::OMP_Uniform: {
- Value *ScalarParam = CI->getArgOperand(Param.ParamPos);
- // Make sure the scalar parameter in the loop is invariant.
- if (!PSE.getSE()->isLoopInvariant(PSE.getSCEV(ScalarParam),
- TheLoop))
- ParamsOk = false;
- break;
- }
- case VFParamKind::OMP_Linear: {
- Value *ScalarParam = CI->getArgOperand(Param.ParamPos);
- // Find the stride for the scalar parameter in this loop and see if
- // it matches the stride for the variant.
- // TODO: do we need to figure out the cost of an extract to get the
- // first lane? Or do we hope that it will be folded away?
- ScalarEvolution *SE = PSE.getSE();
- if (!match(SE->getSCEV(ScalarParam),
- m_scev_AffineAddRec(
- m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
- m_SpecificLoop(TheLoop))))
- ParamsOk = false;
- break;
- }
- case VFParamKind::GlobalPredicate:
- break;
- default:
- ParamsOk = false;
- break;
- }
- }
-
- if (!ParamsOk)
- continue;
-
- // Found a suitable candidate, stop here.
- VecFunc = CI->getModule()->getFunction(Info.VectorName);
- FuncInfo = Info;
- break;
- }
-
- if (TLI && VecFunc && !CI->isNoBuiltin())
- VectorCost = TTI.getCallInstrCost(nullptr, RetTy, Tys, Config.CostKind);
-
- // Find the cost of an intrinsic; some targets may have instructions that
- // perform the operation without needing an actual call.
- Intrinsic::ID IID = getVectorIntrinsicIDForCall(CI, TLI);
- if (IID != Intrinsic::not_intrinsic)
- IntrinsicCost = getVectorIntrinsicCost(CI, VF);
-
- InstructionCost Cost = ScalarCost;
- InstWidening Decision = CM_Scalarize;
-
- if (VectorCost.isValid() && VectorCost <= Cost) {
- Cost = VectorCost;
- Decision = CM_VectorCall;
- }
-
- if (IntrinsicCost.isValid() && IntrinsicCost <= Cost) {
- Cost = IntrinsicCost;
- Decision = CM_IntrinsicCall;
- }
-
- setCallWideningDecision(CI, VF, Decision, VecFunc, IID,
- FuncInfo.getParamIndexForOptionalMask(), Cost);
- }
- }
-}
-
bool LoopVectorizationCostModel::shouldConsiderInvariant(Value *Op) {
if (!Legal->isInvariant(Op))
return false;
@@ -5472,9 +5310,6 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
return TTI::CastContextHint::Reversed;
case LoopVectorizationCostModel::CM_Unknown:
llvm_unreachable("Instr did not go through cost modelling?");
- case LoopVectorizationCostModel::CM_VectorCall:
- case LoopVectorizationCostModel::CM_IntrinsicCall:
- llvm_unreachable_internal("Instr has invalid widening decision");
}
llvm_unreachable("Unhandled case!");
@@ -5874,22 +5709,6 @@ bool VPCostContext::isMaskRequired(Instruction *I) const {
return CM.isMaskRequired(I);
}
-std::optional<VPCostContext::CallWideningKind>
-VPCostContext::getLegacyCallKind(CallInst *CI, ElementCount VF) const {
- if (VF.isScalar())
- return std::nullopt;
- switch (CM.getCallWideningDecision(CI, VF).Kind) {
- case LoopVectorizationCostModel::CM_Scalarize:
- return CallWideningKind::Scalarize;
- case LoopVectorizationCostModel::CM_IntrinsicCall:
- return CallWideningKind::Intrinsic;
- case LoopVectorizationCostModel::CM_VectorCall:
- return CallWideningKind::VectorVariant;
- default:
- return std::nullopt;
- }
-}
-
InstructionCost
LoopVectorizationPlanner::precomputeCosts(VPlan &Plan, ElementCount VF,
VPCostContext &CostCtx) const {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanHelpers.h b/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
index ff4b2d7f32964..bbe3dee516690 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
@@ -329,9 +329,6 @@ struct VPTransformState {
/// Struct to hold various analysis needed for cost computations.
struct VPCostContext {
- /// Choice for how to widen a call at a given VF.
- enum class CallWideningKind { Scalarize, Intrinsic, VectorVariant };
-
const TargetTransformInfo &TTI;
const TargetLibraryInfo &TLI;
VPTypeAnalysis Types;
@@ -370,11 +367,6 @@ struct VPCostContext {
/// Forwards to LoopVectorizationCostModel::isMaskRequired.
bool isMaskRequired(Instruction *I) const;
- /// Returns the legacy call widening decision for \p CI at \p VF, or
- /// std::nullopt if none was recorded. Used only in asserts.
- std::optional<CallWideningKind> getLegacyCallKind(CallInst *CI,
- ElementCount VF) const;
-
/// Returns the OperandInfo for \p V, if it is a live-in.
TargetTransformInfo::OperandValueInfo getOperandInfo(VPValue *V) const;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 85846626e2159..741eca0509ac8 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -6543,7 +6543,7 @@ findVectorVariant(CallInst *CI, ArrayRef<VPValue *> Args, ElementCount VF,
namespace {
/// The outcome of choosing how to widen a call at a given VF.
struct CallWideningDecision {
- using KindTy = VPCostContext::CallWideningKind;
+ enum class KindTy { Scalarize, Intrinsic, VectorVariant };
KindTy Kind = KindTy::Scalarize;
/// Set when Kind == VectorVariant.
Function *Variant = nullptr;
@@ -6663,17 +6663,6 @@ void VPlanTransforms::makeCallWideningDecisions(VPlan &Plan, VFRange &Range,
if (!Recipe)
Recipe = RecipeBuilder.handleReplication(VPI, Range);
- assert(all_of(Range,
- [&](ElementCount VF) {
- Intrinsic::ID IID =
- getVectorIntrinsicIDForCall(CI, &CostCtx.TLI);
- if (IID && VPCostContext::isFreeScalarIntrinsic(IID))
- return true;
- auto Legacy = CostCtx.getLegacyCallKind(CI, VF);
- return !Legacy || *Legacy == Decision.Kind;
- }) &&
- "VPlan call widening decision must match legacy decision");
-
Recipe->insertBefore(VPI);
VPI->replaceAllUsesWith(Recipe->getVPSingleValue());
ToErase.push_back(VPI);
More information about the llvm-commits
mailing list