[llvm] [VPlan] Move call widening decision to VPlan. (NFCI) (PR #195518)
Graham Hunter via llvm-commits
llvm-commits at lists.llvm.org
Wed May 6 05:40:44 PDT 2026
================
@@ -6548,3 +6552,177 @@ void VPlanTransforms::makeScalarizationDecisions(VPlan &Plan, VFRange &Range) {
}
}
}
+
+/// Returns true if \p Info's parameter kinds are compatible with \p Args.
+static bool areVFParamsOk(const VFInfo &Info, ArrayRef<VPValue *> Args,
+ PredicatedScalarEvolution &PSE, const Loop *L) {
+ return all_of(Info.Shape.Parameters, [&](VFParameter Param) {
+ switch (Param.ParamKind) {
+ case VFParamKind::Vector:
+ case VFParamKind::GlobalPredicate:
+ return true;
+ case VFParamKind::OMP_Uniform:
+ return PSE.getSE()->isLoopInvariant(
+ vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L), L);
+ case VFParamKind::OMP_Linear:
+ return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
+ m_scev_AffineAddRec(
+ m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
+ m_SpecificLoop(L)));
+ default:
+ return false;
+ }
+ });
+}
+
+/// Find a vector variant of \p CI for \p VF, respecting \p MaskRequired.
+/// Returns the variant function and the position of its mask parameter
+/// (if any), or {nullptr, std::nullopt}.
+static std::pair<Function *, std::optional<unsigned>>
+findVectorVariant(CallInst *CI, ArrayRef<VPValue *> Args, ElementCount VF,
+ bool MaskRequired, PredicatedScalarEvolution &PSE,
+ const Loop *L) {
+ if (CI->isNoBuiltin())
+ return {nullptr, std::nullopt};
+ auto Mappings = VFDatabase::getMappings(*CI);
+ const auto *It = find_if(Mappings, [&](const VFInfo &Info) {
+ return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
+ areVFParamsOk(Info, Args, PSE, L);
+ });
+ if (It == Mappings.end())
+ return {nullptr, std::nullopt};
+ if (Function *VecFunc = CI->getModule()->getFunction(It->VectorName))
+ return {VecFunc, It->getParamIndexForOptionalMask()};
+ return {nullptr, std::nullopt};
+}
+
+namespace {
+/// The outcome of choosing how to widen a call at a given VF.
+struct CallWideningDecision {
+ using KindTy = VPCostContext::CallWideningKind;
+ KindTy Kind = KindTy::Scalarize;
+ /// Set when Kind == VectorVariant.
+ Function *Variant = nullptr;
+ /// Position of the mask parameter for \p Variant, if any.
+ std::optional<unsigned> MaskPos;
+};
+} // namespace
+
+/// Pick the cheapest widening for the call \p VPI at \p VF among scalarization,
+/// vector intrinsic, and vector library variant.
+static CallWideningDecision decideCallWidening(VPInstruction &VPI,
+ ArrayRef<VPValue *> Ops,
+ ElementCount VF,
+ VPCostContext &CostCtx) {
+ auto *CI = cast<CallInst>(VPI.getUnderlyingInstr());
+
+ // Scalar VFs and calls forced or known to scalarize always replicate.
+ if (VF.isScalar() || CostCtx.willBeScalarized(CI, VF))
+ return {};
+
+ auto *CalledFn = cast<Function>(
+ VPI.getOperand(VPI.getNumOperandsWithoutMask() - 1)->getLiveInIRValue());
+ Type *ResultTy = CostCtx.Types.inferScalarType(&VPI);
+ Intrinsic::ID ID = getVectorIntrinsicIDForCall(CI, &CostCtx.TLI);
+ bool MaskRequired = CostCtx.isMaskRequired(CI);
+
+ // Pseudo intrinsics (assume, lifetime, ...) are always scalarized.
+ if (ID && VPCostContext::isFreeScalarIntrinsic(ID))
+ return {};
+
+ InstructionCost ScalarCost = VPReplicateRecipe::computeScalarCallCost(
+ CalledFn, ResultTy, Ops,
+ /*IsSingleScalar=*/false, VF, CostCtx);
+
+ auto [VecFunc, MaskPos] =
+ findVectorVariant(CI, Ops, VF, MaskRequired, CostCtx.PSE, CostCtx.L);
+ InstructionCost VecCallCost = InstructionCost::getInvalid();
+ if (VecFunc)
+ VecCallCost = VPWidenCallRecipe::computeVectorCallCost(VecFunc, CostCtx);
+
+ // Prefer the intrinsic if it is at least as cheap as scalarizing and any
+ // available vector variant.
+ if (ID) {
+ InstructionCost IntrinsicCost =
+ VPWidenIntrinsicRecipe::computeIntrinsicCost(ID, Ops, VPI, VF, CostCtx);
+ if (IntrinsicCost.isValid() && ScalarCost >= IntrinsicCost &&
+ (!VecFunc || VecCallCost >= IntrinsicCost))
+ return {CallWideningDecision::KindTy::Intrinsic, nullptr, std::nullopt};
+ }
+
+ // Otherwise, use a vector library variant when it beats scalarizing.
+ if (VecFunc && ScalarCost >= VecCallCost)
+ return {CallWideningDecision::KindTy::VectorVariant, VecFunc, MaskPos};
+
+ return {};
+}
+
+void VPlanTransforms::makeCallWideningDecisions(VPlan &Plan, VFRange &Range,
+ VPRecipeBuilder &RecipeBuilder,
+ VPCostContext &CostCtx) {
+ SmallVector<VPInstruction *, 8> ToErase;
+ for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
+ vp_depth_first_shallow(Plan.getVectorLoopRegion()->getEntry()))) {
+ for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
+ auto *VPI = dyn_cast<VPInstruction>(&R);
+ if (!VPI || !VPI->getUnderlyingValue() ||
+ VPI->getOpcode() != Instruction::Call)
+ continue;
+
+ auto *CI = cast<CallInst>(VPI->getUnderlyingInstr());
+ SmallVector<VPValue *, 4> Ops(VPI->op_begin(),
+ VPI->op_begin() + CI->arg_size());
----------------
huntergr-arm wrote:
MaskPos is there in case any of the vector function variants have the mask as an argument that isn't the last one... we might have been overly flexible when initially creating it, expecting that people might create their own variants with the mask in an arbitrary position.
In practice, we only support variants with the mask as the last argument, and `VFInfo::getParamIndexForOptionalMask` was changed some time ago to assert if that's not the case.
So we could simplify all of this and remove MaskPos, though maybe that should be a separate PR landed first?
https://github.com/llvm/llvm-project/pull/195518
More information about the llvm-commits
mailing list