[llvm] [LV] Emit truncate for widen-induction in VPlan during construction. (PR #225991)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 29 00:07:20 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-vectorizers
@llvm/pr-subscribers-backend-powerpc
Author: Elvis Wang (ElvisWang123)
<details>
<summary>Changes</summary>
Instead of expand the truncate recipe just before execution, this patch emit truncate for VPWidenIntOrFpInductionRecipe directly to the VPlan during construction. This removes the implicit truncate contained in the VPWidenIntOrFpInductionRecipe and helps we model the cost more accurately.
This patch also brings some small improvement from other VPlan optimizations.
---
Patch is 165.79 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/225991.diff
56 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h (+3-2)
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorize.cpp (+14)
- (modified) llvm/lib/Transforms/Vectorize/VPlan.h (+7-27)
- (modified) llvm/lib/Transforms/Vectorize/VPlanEVLTailFolding.cpp (+18)
- (modified) llvm/lib/Transforms/Vectorize/VPlanLowering.cpp (+4-17)
- (modified) llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp (-3)
- (modified) llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp (+22-23)
- (modified) llvm/lib/Transforms/Vectorize/VPlanUtils.cpp (+13-31)
- (modified) llvm/lib/Transforms/Vectorize/VPlanUtils.h (+7-7)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/alias-mask.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/conditional-scalar-assignment.ll (+2-2)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/epilog-iv-live-outs.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/epilog-iv-select-cmp.ll (+5-5)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/epilogue-vectorization-fix-scalar-resume-values.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/induction-costs-sve.ll (+6-12)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/interleave-with-gaps.ll (+2-2)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/optsize_minsize.ll (+3-3)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-dot-product-epilogue.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-dot-product.ll (+6-6)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-fold-tail.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce.ll (+6-6)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/scalable-strict-fadd.ll (+2-2)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/select-index.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/sve-interleaved-accesses.ll (+2-2)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/sve2-histcnt-too-many-deps.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/sve2-histcnt.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/vector-reverse.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/PowerPC/exit-branch-cost.ll (+2-2)
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/induction-costs.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/iv-select-cmp.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/partial-reduce.ll (+2-2)
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-cond-reduction.ll (+8-8)
- (modified) llvm/test/Transforms/LoopVectorize/VPlan/interleave-and-scalarize-only.ll (+16-15)
- (modified) llvm/test/Transforms/LoopVectorize/VPlan/vplan-iv-transforms.ll (+8-7)
- (modified) llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-flags.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing-reductions.ll (+20-20)
- (modified) llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing.ll (+19-17)
- (modified) llvm/test/Transforms/LoopVectorize/X86/conversion-cost.ll (+18-18)
- (modified) llvm/test/Transforms/LoopVectorize/X86/cost-model.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/X86/epilog-vectorization-inductions.ll (+27-26)
- (modified) llvm/test/Transforms/LoopVectorize/X86/induction-costs.ll (+16-16)
- (modified) llvm/test/Transforms/LoopVectorize/X86/pr36524.ll (+2-2)
- (modified) llvm/test/Transforms/LoopVectorize/alias-mask.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/cast-costs.ll (-3)
- (modified) llvm/test/Transforms/LoopVectorize/cast-induction.ll (+3-3)
- (modified) llvm/test/Transforms/LoopVectorize/epilog-iv-select-cmp.ll (+2-2)
- (modified) llvm/test/Transforms/LoopVectorize/epilog-vectorization-reductions.ll (+5-5)
- (modified) llvm/test/Transforms/LoopVectorize/if-pred-stores.ll (+2-2)
- (modified) llvm/test/Transforms/LoopVectorize/induction-cost.ll (+8-8)
- (modified) llvm/test/Transforms/LoopVectorize/induction-step.ll (+2-2)
- (modified) llvm/test/Transforms/LoopVectorize/induction.ll (+15-15)
- (modified) llvm/test/Transforms/LoopVectorize/interleave-with-i65-induction.ll (+4-4)
- (modified) llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/iv-select-cmp-non-const-iv-start.ll (+2-1)
- (modified) llvm/test/Transforms/LoopVectorize/iv-select-cmp-trunc.ll (+6-4)
- (modified) llvm/test/Transforms/LoopVectorize/reduction-small-size.ll (+1-1)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index cb38b0be1808a..b079e905debe2 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -409,9 +409,10 @@ class VPBuilder {
VPDerivedIVRecipe *createDerivedIV(InductionDescriptor::InductionKind Kind,
FPMathOperator *FPBinOp, VPValue *Start,
VPValue *Current, VPValue *Step,
- const VPIRFlags::WrapFlagsTy &Flags = {}) {
+ const VPIRFlags::WrapFlagsTy &Flags = {},
+ DebugLoc DL = DebugLoc::getUnknown()) {
return tryInsertInstruction(
- new VPDerivedIVRecipe(Kind, FPBinOp, Start, Current, Step, Flags));
+ new VPDerivedIVRecipe(Kind, FPBinOp, Start, Current, Step, Flags, DL));
}
VPInstruction *createScalarCast(Instruction::CastOps Opcode, VPValue *Op,
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index d929af8afbd1d..b91c2b9483339 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7565,6 +7565,20 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
}
assert(ResumeV && "Must have a resume value");
VPValue *StartVal = Plan.getOrAddLiveIn(ResumeV);
+ auto *PhiR = dyn_cast<VPWidenIntOrFpInductionRecipe>(&R);
+ // A truncated widen induction has narrower type than resume value, so
+ // create a scalar cast to match the type.
+ if (PhiR && PhiR->getScalarType() != StartVal->getScalarType()) {
+ assert(StartVal->getScalarType()->getScalarSizeInBits() >
+ PhiR->getScalarType()->getScalarSizeInBits() &&
+ "Widen induction type should always narrower or same as resume "
+ "value type.");
+ VPBuilder PHBuilder(Plan.getVectorPreheader(),
+ Plan.getVectorPreheader()->getFirstNonPhi());
+ StartVal = PHBuilder.createScalarCast(Instruction::Trunc, StartVal,
+ PhiR->getScalarType(),
+ PhiR->getDebugLoc());
+ }
cast<VPHeaderPHIRecipe>(&R)->setStartValue(StartVal);
}
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 8f22a362e6fe6..c0837c349ad8d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -2615,8 +2615,6 @@ class VPWidenInductionRecipe : public VPHeaderPHIRecipe {
/// converted to concrete recipes before executing.
class VPWidenIntOrFpInductionRecipe : public VPWidenInductionRecipe,
public VPIRFlags {
- TruncInst *Trunc;
-
// If this recipe is unrolled it will have 2 additional operands.
bool isUnrolled() const { return getNumOperands() == 5; }
@@ -2626,31 +2624,16 @@ class VPWidenIntOrFpInductionRecipe : public VPWidenInductionRecipe,
const VPIRFlags &Flags, DebugLoc DL)
: VPWidenInductionRecipe(VPRecipeBase::VPWidenIntOrFpInductionSC, IV,
Start, Step, IndDesc, DL),
- VPIRFlags(Flags), Trunc(nullptr) {
+ VPIRFlags(Flags) {
addOperand(VF);
}
- VPWidenIntOrFpInductionRecipe(PHINode *IV, VPValue *Start, VPValue *Step,
- VPValue *VF, const InductionDescriptor &IndDesc,
- TruncInst *Trunc, const VPIRFlags &Flags,
- DebugLoc DL)
- : VPWidenInductionRecipe(
- VPRecipeBase::VPWidenIntOrFpInductionSC, IV, Start, Step, IndDesc,
- Trunc ? Trunc->getType() : Start->getScalarType(), DL),
- VPIRFlags(Flags), Trunc(Trunc) {
- addOperand(VF);
- SmallVector<std::pair<unsigned, MDNode *>> Metadata;
- if (Trunc)
- getMetadataToPropagate(Trunc, Metadata);
- assert(Metadata.empty() && "unexpected metadata on Trunc");
- }
-
~VPWidenIntOrFpInductionRecipe() override = default;
VPWidenIntOrFpInductionRecipe *clone() override {
return new VPWidenIntOrFpInductionRecipe(
getPHINode(), getStartValue(), getStepValue(), getVFValue(),
- getInductionDescriptor(), Trunc, *this, getDebugLoc());
+ getInductionDescriptor(), *this, getDebugLoc());
}
VP_CLASSOF_IMPL(VPRecipeBase::VPWidenIntOrFpInductionSC)
@@ -2671,11 +2654,6 @@ class VPWidenIntOrFpInductionRecipe : public VPWidenInductionRecipe,
/// incoming value, its start value.
unsigned getNumIncoming() const override { return 1; }
- /// Returns the first defined value as TruncInst, if it is one or nullptr
- /// otherwise.
- TruncInst *getTruncInst() { return Trunc; }
- const TruncInst *getTruncInst() const { return Trunc; }
-
/// Return the cost of this VPWidenIntOrFpInductionRecipe.
InstructionCost computeCost(ElementCount VF,
VPCostContext &Ctx) const override;
@@ -4209,16 +4187,18 @@ class LLVM_ABI_FOR_TEST VPDerivedIVRecipe : public VPRecipeWithIRFlags {
VPDerivedIVRecipe(InductionDescriptor::InductionKind Kind,
const FPMathOperator *FPBinOp, VPValue *Start,
VPValue *Current, VPValue *Step,
- const VPIRFlags::WrapFlagsTy &Flags = {})
+ const VPIRFlags::WrapFlagsTy &Flags = {},
+ DebugLoc DL = DebugLoc::getUnknown())
: VPRecipeWithIRFlags(VPRecipeBase::VPDerivedIVSC, {Start, Current, Step},
- Start->getScalarType(), Flags),
+ Start->getScalarType(), Flags, DL),
Kind(Kind), FPBinOp(FPBinOp) {}
~VPDerivedIVRecipe() override = default;
VPDerivedIVRecipe *clone() override {
return new VPDerivedIVRecipe(Kind, FPBinOp, getStartValue(), getOperand(1),
- getStepValue(), getNoWrapFlags());
+ getStepValue(), getNoWrapFlags(),
+ getDebugLoc());
}
VP_CLASSOF_IMPL(VPRecipeBase::VPDerivedIVSC)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanEVLTailFolding.cpp b/llvm/lib/Transforms/Vectorize/VPlanEVLTailFolding.cpp
index b6ede1c1edc7f..80aaf04348dc4 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanEVLTailFolding.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanEVLTailFolding.cpp
@@ -384,6 +384,24 @@ static void fixupVFUsersForEVL(VPlan &Plan, VPValue &EVL) {
return isa<VPWidenPointerInductionRecipe>(U);
});
+ // Also replace VF with EVL for truncated widen induction.
+ for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
+ vp_depth_first_deep(Plan.getVectorLoopRegion()->getEntry()))) {
+ for (VPRecipeBase &R : VPBB->phis()) {
+ auto *WidenIV = dyn_cast<VPWidenIntOrFpInductionRecipe>(&R);
+ if (!WidenIV)
+ continue;
+ if (!match(WidenIV->getOperand(2), m_Trunc(m_Specific(&Plan.getVF()))))
+ continue;
+ VPValue *TruncEVL =
+ VPBuilder::getToInsertAfter(EVL.getDefiningRecipe())
+ .createScalarZExtOrTrunc(
+ &EVL, cast<VPSingleDefRecipe>(R).getScalarType(),
+ DebugLoc::getUnknown());
+ R.setOperand(2, TruncEVL);
+ }
+ }
+
// Create a scalar phi to track the previous EVL if fixed-order recurrence is
// contained.
bool ContainsFORs =
diff --git a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
index 230fb5f39d33f..064ecb3b35f59 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
@@ -58,7 +58,7 @@ void VPlanTransforms::replaceWideCanonicalIVWithWideIV(
VPBuilder Builder(WideCanIV);
WideCanIV->replaceAllUsesWith(vputils::createScalarIVSteps(
Plan, InductionDescriptor::IK_IntInduction, Instruction::Add, nullptr,
- nullptr, Plan.getZero(CanIVTy), Plan.getConstantInt(CanIVTy, 1),
+ Plan.getZero(CanIVTy), Plan.getConstantInt(CanIVTy, 1),
WideCanIV->getDebugLoc(), Builder,
{static_cast<bool>(WideCanIV->getNoWrapFlags().HasNUW), false}));
WideCanIV->eraseFromParent();
@@ -259,10 +259,6 @@ expandVPWidenIntOrFpInduction(VPWidenIntOrFpInductionRecipe *WidenIVR) {
VPValue *VF = WidenIVR->getVFValue();
DebugLoc DL = WidenIVR->getDebugLoc();
- // The value from the original loop to which we are mapping the new induction
- // variable.
- Type *Ty = WidenIVR->getScalarType();
-
const InductionDescriptor &ID = WidenIVR->getInductionDescriptor();
Instruction::BinaryOps AddOp;
Instruction::BinaryOps MulOp;
@@ -275,17 +271,9 @@ expandVPWidenIntOrFpInduction(VPWidenIntOrFpInductionRecipe *WidenIVR) {
MulOp = Instruction::FMul;
}
- // If the phi is truncated, truncate the start and step values.
+ // Construct the initial value of the vector IV in the vector loop preheader.
VPBuilder Builder(Plan->getVectorPreheader());
Type *StepTy = Step->getScalarType();
- if (Ty->getScalarSizeInBits() < StepTy->getScalarSizeInBits()) {
- assert(StepTy->isIntegerTy() && "Truncation requires an integer type");
- Step = Builder.createScalarCast(Instruction::Trunc, Step, Ty, DL);
- Start = Builder.createScalarCast(Instruction::Trunc, Start, Ty, DL);
- StepTy = Ty;
- }
-
- // Construct the initial value of the vector IV in the vector loop preheader.
Type *IVIntTy =
IntegerType::get(Plan->getContext(), StepTy->getScalarSizeInBits());
VPValue *Init = Builder.createNaryOp(VPInstruction::StepVector, {}, IVIntTy);
@@ -409,10 +397,9 @@ static void expandVPDerivedIV(VPDerivedIVRecipe *R) {
VPValue *Index = R->getIndex();
Type *StepTy = Step->getScalarType();
Index = StepTy->isIntegerTy()
- ? Builder.createScalarZExtOrTrunc(
- Index, StepTy, DebugLoc::getCompilerGenerated())
+ ? Builder.createScalarZExtOrTrunc(Index, StepTy, R->getDebugLoc())
: Builder.createScalarCast(Instruction::SIToFP, Index, StepTy,
- DebugLoc::getCompilerGenerated());
+ R->getDebugLoc());
VPIRFlags::WrapFlagsTy Flags = R->getNoWrapFlags();
switch (R->getInductionKind()) {
case InductionDescriptor::IK_IntInduction: {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 38c94512f0546..0d5578092f319 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -3013,9 +3013,6 @@ void VPWidenIntOrFpInductionRecipe::printRecipe(
O << " = WIDEN-INDUCTION";
printFlags(O);
printOperands(O, SlotTracker);
-
- if (auto *TI = getTruncInst())
- O << " (truncated to " << *TI->getType() << ")";
}
#endif
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index f26c031c79180..c6dec7c0ecc23 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -677,9 +677,6 @@ static void removeRedundantInductionCasts(VPlan &Plan) {
for (VPWidenIntOrFpInductionRecipe &IV :
make_isa_range<VPWidenIntOrFpInductionRecipe>(
Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis())) {
- if (IV.getTruncInst())
- continue;
-
// A sequence of IR Casts has potentially been recorded for IV, which
// *must be bypassed* when the IV is vectorized, because the vectorized IV
// will produce the desired casted value. This sequence forms a def-use
@@ -853,8 +850,8 @@ static void legalizeAndOptimizeInductions(VPlan &Plan) {
VPScalarIVStepsRecipe *Steps = vputils::createScalarIVSteps(
Plan, ID.getKind(), ID.getInductionOpcode(),
dyn_cast_or_null<FPMathOperator>(ID.getInductionBinOp()),
- WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
- WideIV->getDebugLoc(), Builder, WrapFlags);
+ WideIV->getStartValue(), WideIV->getStepValue(), WideIV->getDebugLoc(),
+ Builder, WrapFlags);
// Update scalar users of IV to use Step instead.
if (!HasOnlyVectorVFs) {
@@ -880,10 +877,10 @@ static VPWidenInductionRecipe *
getOptimizableIVOf(VPValue *VPV, PredicatedScalarEvolution &PSE) {
auto *WideIV = dyn_cast<VPWidenInductionRecipe>(VPV);
if (WideIV) {
- // VPV itself is a wide induction, separately compute the end value for exit
- // users if it is not a truncated IV.
- auto *IntOrFpIV = dyn_cast<VPWidenIntOrFpInductionRecipe>(WideIV);
- return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
+ if (any_of(WideIV->operands(),
+ [](VPValue *V) { return match(V, m_Trunc(m_VPValue())); }))
+ return nullptr;
+ return WideIV;
}
// Check if VPV is an optimizable induction increment.
@@ -978,17 +975,11 @@ static VPValue *optimizeEarlyExitInductionUser(VPlan &Plan, VPValue *Op,
return EndValue;
}
-/// Compute the end value for \p WideIV, unless it is truncated. Creates a
-/// VPDerivedIVRecipe for non-canonical inductions.
+/// Compute the end value for \p WideIV. Creates a VPDerivedIVRecipe for
+/// non-canonical inductions.
static VPValue *tryToComputeEndValueForInduction(VPWidenInductionRecipe *WideIV,
VPBuilder &VectorPHBuilder,
VPValue *VectorTC) {
- auto *WideIntOrFp = dyn_cast<VPWidenIntOrFpInductionRecipe>(WideIV);
- // Truncated wide inductions resume from the last lane of their vector value
- // in the last vector iteration which is handled elsewhere.
- if (WideIntOrFp && WideIntOrFp->getTruncInst())
- return nullptr;
-
VPValue *Start = WideIV->getStartValue();
VPValue *Step = WideIV->getStepValue();
const InductionDescriptor &ID = WideIV->getInductionDescriptor();
@@ -1102,7 +1093,7 @@ void VPlanTransforms::optimizeInductionLiveOutUsers(
// Compute end values for all inductions.
VPRegionBlock *VectorRegion = Plan.getVectorLoopRegion();
auto *VectorPH = cast<VPBasicBlock>(VectorRegion->getSinglePredecessor());
- VPBuilder VectorPHBuilder(VectorPH, VectorPH->getFirstNonPhi());
+ VPBuilder VectorPHBuilder(VectorPH);
DenseMap<VPValue *, VPValue *> EndValues;
VPValue *ResumeTC =
Plan.hasTailFolded() ? Plan.getTripCount() : &Plan.getVectorTripCount();
@@ -2043,9 +2034,6 @@ static bool optimizeVectorInductionWidthForTCAndVFUF(VPlan &Plan,
m_Broadcast(m_Specific(Plan.getBackedgeTakenCount())))))
continue;
- // Update IV operands and comparison bound to use new narrower type.
- assert(!WideIV->getTruncInst() &&
- "canonical IV is not expected to have a truncation");
auto *NewWideIV = new VPWidenIntOrFpInductionRecipe(
WideIV->getPHINode(), Plan.getZero(NewIVTy),
Plan.getConstantInt(NewIVTy, 1), WideIV->getVFValue(),
@@ -5977,11 +5965,22 @@ void VPlanTransforms::narrowInductionTruncates(VPlan &Plan, VFRange &Range,
IsNarrowingProfitable, Range))
continue;
+ VPBuilder PHBuilder(Plan.getVectorPreheader());
+ auto *NewStart = PHBuilder.createScalarCast(
+ Instruction::Trunc, WideIV->getStartValue(), VPI.getScalarType(),
+ VPI.getDebugLoc());
+ auto *NewStep =
+ PHBuilder.createScalarCast(Instruction::Trunc, WideIV->getStepValue(),
+ VPI.getScalarType(), VPI.getDebugLoc());
+ auto *NewVF =
+ PHBuilder.createScalarCast(Instruction::Trunc, WideIV->getVFValue(),
+ VPI.getScalarType(), VPI.getDebugLoc());
+
// Wrap flags of the original induction do not hold in the truncated
// type, so do not propagate them.
auto *NarrowIV = new VPWidenIntOrFpInductionRecipe(
- WideIV->getPHINode(), WideIV->getStartValue(), WideIV->getStepValue(),
- WideIV->getVFValue(), WideIV->getInductionDescriptor(), Trunc,
+ WideIV->getPHINode(), NewStart, NewStep, NewVF,
+ WideIV->getInductionDescriptor(),
VPIRFlags::WrapFlagsTy(false, false), VPI.getDebugLoc());
NarrowIV->insertBefore(*HeaderVPBB, HeaderVPBB->getFirstNonPhi());
VPI.replaceAllUsesWith(NarrowIV);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 9a07697f66762..4f17bedc5a855 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -322,8 +322,6 @@ const SCEV *vputils::getSCEVExprForVPValue(const VPValue *V,
getSCEVExprForVPValue(R->getStartValue(), PSE, L);
const SCEV *AddRec =
SE.getAddRecExpr(Start, Step, L, SCEV::FlagNone);
- if (R->getTruncInst())
- return SE.getTruncateExpr(AddRec, R->getScalarType());
return AddRec;
})
.Case([&SE, &PSE,
@@ -626,37 +624,21 @@ vputils::getEarlyExits(const VPlan &Plan, const VPBlockBase *MiddleVPBB) {
VPScalarIVStepsRecipe *vputils::createScalarIVSteps(
VPlan &Plan, InductionDescriptor::InductionKind Kind,
Instruction::BinaryOps InductionOpcode, FPMathOperator *FPBinOp,
- Instruction *TruncI, VPValue *StartV, VPValue *Step, DebugLoc DL,
- VPBuilder &Builder, const VPIRFlags::WrapFlagsTy &Flags) {
+ VPValue *StartV, VPValue *Step, DebugLoc DL, VPBuilder &Builder,
+ const VPIRFlags::WrapFlagsTy &Flags) {
VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
- VPBasicBlock *HeaderVPBB = LoopRegion->getEntryBasicBlock();
VPValue *CanonicalIV = LoopRegion->getCanonicalIV();
- VPSingleDefRecipe *BaseIV =
- Builder.createDerivedIV(Kind, FPBinOp, StartV, CanonicalIV, Step, Flags);
-
- // Truncate base induction if needed.
- Type *ResultTy = BaseIV->getScalarType();
- if (TruncI) {
- Type *TruncTy = TruncI->getType();
- assert(ResultTy->getScalarSizeInBits() > TruncTy->getScalarSizeInBits() &&
- "Not truncating.");
- assert(ResultTy->isIntegerTy() && "Truncation requires an integer type");
- BaseIV = Builder.createScalarCast(Instruction::Trunc, BaseIV, TruncTy, DL);
- ResultTy = TruncTy;
- }
-
- // Truncate step if needed.
+ Type *CanonicalIVTy = CanonicalIV->getScalarType();
Type *StepTy = Step->getScalarType();
- if (ResultTy != StepTy) {
- assert(StepTy->getScalarSizeInBits() > ResultTy->getScalarSizeInBits() &&
- "Not truncating.");
- assert(StepTy->isIntegerTy() && "Truncation requires an integer type");
- auto *VecPreheader =
- cast<VPBasicBlock>(HeaderVPBB->getSingleHierarchicalPredecessor());
- VPBuilder::InsertPointGuard Guard(Builder);
- Builder.setInsertPoint(VecPreheader);
- Step = Builder.createScalarCast(Instruction::Trunc, Step, ResultTy, DL);
- }
+ if (CanonicalIVTy->isIntegerTy() && StepTy->isIntegerTy() &&
+ CanonicalIVTy->getScalarSizeInBits() > StepTy->getScalarSizeInBits())
+ CanonicalIV = Builder.createScalarZExtOrTrunc(CanonicalIV, StepTy, DL);
+ VPSingleDefRecipe *BaseIV = Builder.createDerivedIV(
+ Kind, FPBinOp, StartV, CanonicalIV, Step, Flags, DL);
+
+ assert(BaseIV->getScalarType()->getScalarSizeInBits() ==
+ Step->getScalarType()->getScalarSizeInBits() &&
+ "IV should already be truncated");
return Builder.createScalarIVSteps(InductionOpcode, FPBinOp, BaseIV, Step,
&Plan.getVF(), DL);
}
@@ -669,7 +651,7 @@ vputils::scalarizeVPWidenPointerInduction(VPWidenPointerInductionRecipe *PtrIV,
VPValue *StepV = PtrIV->getOperand(1);
VPScalarIVStepsRecipe *Steps = createScalarIVSteps(
Plan, InductionDescriptor::IK_IntInduction, Instruction::Add, nullptr,
- nullptr, StartV, StepV, PtrIV->getDebugLoc(), Builder);
+ StartV, StepV, PtrIV->getDebugLoc(), Builder);
return Builder.createPtrAdd(PtrIV->getStartValue(), Steps,
PtrIV->getDebugLoc(), "next.gep");
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.h b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
index 8b5309c3e56fb..b6...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/225991
More information about the llvm-commits
mailing list