[llvm] 6c95a92 - [LV] Narrow truncated inductions in a VPlan transform (#220730)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 22 20:04:57 PDT 2026
Author: Vedant Paranjape
Date: 2026-09-22T20:04:50-07:00
New Revision: 6c95a9274885a7c0490cd8bc67ae7549f479d825
URL: https://github.com/llvm/llvm-project/commit/6c95a9274885a7c0490cd8bc67ae7549f479d825
DIFF: https://github.com/llvm/llvm-project/commit/6c95a9274885a7c0490cd8bc67ae7549f479d825.diff
LOG: [LV] Narrow truncated inductions in a VPlan transform (#220730)
VPRecipeBuilder::tryToOptimizeInductionTruncate matched a truncate of an
induction phi on the underlying IR, and it built the narrowed
VPWidenIntOrFpInductionRecipe from the recipe behind the truncate's
operand.
VPlanTransforms already answers the same question on VPValues in
getOptimizableIVOf, which returns the header IV whether the operand is
the IV
itself or an add of the IV and its step.
Add VPlanTransforms::narrowInductionTruncates and do the match there,
reusing
getOptimizableIVOf and restricting it to the phi for now. The pass runs
right
after makeCallWideningDecisions, which preserves the ordering against
makeScalarizationDecisions that the recipe builder relied on. Building
the
recipe in the transform also removes the need for the conversion loop to
special case a truncate and move the recipe into the header phi section,
because the transform inserts it there directly.
Assisted-By: Claude Opus 5
Added:
Modified:
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h
llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
llvm/lib/Transforms/Vectorize/VPlanTransforms.h
llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
llvm/test/Transforms/LoopVectorize/cast-induction.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index dc533d938bb6e..805a57f8dc4ab 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -6087,38 +6087,6 @@ VPRecipeBase *VPRecipeBuilder::tryToWidenMemory(VPInstruction *VPI,
*VPI, Store->getDebugLoc());
}
-VPWidenIntOrFpInductionRecipe *
-VPRecipeBuilder::tryToOptimizeInductionTruncate(VPInstruction *VPI,
- VFRange &Range) {
- auto *I = cast<TruncInst>(VPI->getUnderlyingInstr());
- // Optimize the special case where the source is a constant integer
- // induction variable. Notice that we can only optimize the 'trunc' case
- // because (a) FP conversions lose precision, (b) sext/zext may wrap, and
- // (c) other casts depend on pointer size.
-
- // Determine whether \p K is a truncation based on an induction variable that
- // can be optimized.
- if (!LoopVectorizationPlanner::getDecisionAndClampRange(
- bind_front(&LoopVectorizationCostModel::isOptimizableIVTruncate, CM,
- I),
- Range))
- return nullptr;
-
- auto *WidenIV = cast<VPWidenIntOrFpInductionRecipe>(
- VPI->getOperand(0)->getDefiningRecipe());
- PHINode *Phi = WidenIV->getPHINode();
- VPValue *Start = WidenIV->getStartValue();
- const InductionDescriptor &IndDesc = WidenIV->getInductionDescriptor();
-
- // Wrap flags from the original induction do not apply to the truncated type,
- // so do not propagate them.
- VPIRFlags Flags = VPIRFlags::WrapFlagsTy(false, false);
- VPValue *Step =
- vputils::getOrCreateVPValueForSCEVExpr(Plan, IndDesc.getStep());
- return new VPWidenIntOrFpInductionRecipe(
- Phi, Start, Step, &Plan.getVF(), IndDesc, I, Flags, VPI->getDebugLoc());
-}
-
bool VPRecipeBuilder::shouldWiden(Instruction *I, VFRange &Range) const {
assert((!isa<UncondBrInst, CondBrInst, PHINode, LoadInst, StoreInst>(I)) &&
"Instruction should have been handled earlier");
@@ -6312,17 +6280,10 @@ VPRecipeBase *
VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
VFRange &Range) {
assert(!R->isPhi() && "phis must be handled earlier");
- // First, check for specific widening recipes that deal with optimizing
- // truncates and memory operations.
auto *VPI = cast<VPInstruction>(R);
assert(VPI->getOpcode() != Instruction::Call &&
"Call should have been handled by makeCallWideningDecisions");
- VPRecipeBase *Recipe;
- if (VPI->getOpcode() == Instruction::Trunc &&
- (Recipe = tryToOptimizeInductionTruncate(VPI, Range)))
- return Recipe;
-
// All widen recipes below deal only with VF > 1.
if (LoopVectorizationPlanner::getDecisionAndClampRange(
[&](ElementCount VF) { return VF.isScalar(); }, Range))
@@ -6674,6 +6635,9 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
RUN_VPLAN_PASS(VPlanTransforms::makeCallWideningDecisions, *Plan, Range,
RecipeBuilder, CostCtx);
+ RUN_VPLAN_PASS(VPlanTransforms::narrowInductionTruncates, *Plan, Range, TTI,
+ PSE);
+
// Convert remaining VPInstructions to widen or replicate recipes.
// TODO: This legacy code should eventually be migrated to VPlan.
VPBasicBlock *HeaderVPBB = LoopRegion->getEntryBasicBlock();
@@ -6705,17 +6669,10 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
VPRecipeBase *Recipe =
RecipeBuilder.tryToCreateWidenNonPhiRecipe(&VPI, Range);
+ if (!Recipe)
+ Recipe = RecipeBuilder.handleReplication(&VPI, Range);
+ Builder.insert(Recipe);
- if (isa_and_nonnull<VPWidenIntOrFpInductionRecipe>(Recipe) &&
- VPI.getOpcode() == Instruction::Trunc) {
- // Optimized a truncate to VPWidenIntOrFpInductionRecipe. It needs to be
- // moved to the phi section in the header.
- Recipe->insertBefore(*HeaderVPBB, HeaderVPBB->getFirstNonPhi());
- } else {
- if (!Recipe)
- Recipe = RecipeBuilder.handleReplication(&VPI, Range);
- Builder.insert(Recipe);
- }
if (Recipe->getNumDefinedValues() == 1) {
VPI.replaceAllUsesWith(Recipe->getVPSingleValue());
} else {
diff --git a/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h b/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h
index 1303b62d5faf4..3af91e68ef427 100644
--- a/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h
+++ b/llvm/lib/Transforms/Vectorize/VPRecipeBuilder.h
@@ -38,11 +38,6 @@ class VPRecipeBuilder {
/// Range. The function should not be called for memory instructions or calls.
bool shouldWiden(Instruction *I, VFRange &Range) const;
- /// Optimize the special case where the operand of \p VPI is a constant
- /// integer induction variable.
- VPWidenIntOrFpInductionRecipe *
- tryToOptimizeInductionTruncate(VPInstruction *VPI, VFRange &Range);
-
/// Check if \p VPI has an opcode that can be widened and return a
/// widened recipe if it can. The function should only be called if the
/// cost-model indicates that widening should be performed.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index a2f2a2997086e..220497c1f1655 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -5886,6 +5886,67 @@ void VPlanTransforms::makeCallWideningDecisions(VPlan &Plan, VFRange &Range,
}
}
+void VPlanTransforms::narrowInductionTruncates(VPlan &Plan, VFRange &Range,
+ const TargetTransformInfo &TTI,
+ PredicatedScalarEvolution &PSE) {
+ VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
+ VPBasicBlock *HeaderVPBB = LoopRegion->getEntryBasicBlock();
+ for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
+ vp_depth_first_shallow(LoopRegion->getEntry()))) {
+ for (VPInstruction &VPI :
+ make_early_inc_range(make_isa_range<VPInstruction>(*VPBB))) {
+ // Only truncates are handled, as sext/zext may wrap, FP conversions lose
+ // precision and other casts depend on the pointer size.
+ if (VPI.getOpcode() != Instruction::Trunc)
+ continue;
+
+ // Underlying Trunc is necessary to create VPWidenIntOrFpInductionRecipe.
+ auto *Trunc = cast_or_null<TruncInst>(VPI.getUnderlyingValue());
+ if (!Trunc)
+ continue;
+
+ // A truncate that is not widened is left to the scalarization decisions
+ // made earlier.
+ if (vputils::onlyFirstLaneUsed(&VPI))
+ continue;
+
+ VPValue *Op = VPI.getOperand(0);
+ auto *WideIV = getOptimizableIVOf(Op, PSE);
+ if (!WideIV)
+ continue;
+
+ // getOptimizableIVOf also matches an add of the IV and its step, which
+ // is not handled here.
+ // TODO: Also narrow truncates of the incremented IV.
+ if (Op != WideIV)
+ continue;
+
+ // Replacing a free truncate would add an induction update instruction to
+ // each iteration of the loop. The canonical induction is exempt, as it
+ // needs an update instruction regardless.
+ auto IsNarrowingProfitable = [&](ElementCount VF) {
+ return match(WideIV, m_CanonicalWidenIV()) ||
+ !TTI.isTruncateFree(
+ toVectorTy(VPI.getOperand(0)->getScalarType(), VF),
+ toVectorTy(VPI.getScalarType(), VF));
+ };
+ if (!LoopVectorizationPlanner::getDecisionAndClampRange(
+ IsNarrowingProfitable, Range))
+ continue;
+
+ // Wrap flags of the original induction do not hold in the truncated
+ // type, so do not propagate them.
+ auto *NarrowIV = new VPWidenIntOrFpInductionRecipe(
+ WideIV->getPHINode(), WideIV->getStartValue(), WideIV->getStepValue(),
+ WideIV->getVFValue(), WideIV->getInductionDescriptor(), Trunc,
+ VPIRFlags::WrapFlagsTy(false, false), VPI.getDebugLoc());
+ NarrowIV->insertBefore(*HeaderVPBB, HeaderVPBB->getFirstNonPhi());
+ VPI.replaceAllUsesWith(NarrowIV);
+ VPI.eraseFromParent();
+ }
+ }
+}
+
void VPlanTransforms::convertToStridedAccesses(VPlan &Plan,
PredicatedScalarEvolution &PSE,
Loop &L, VPCostContext &Ctx,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 2eacb3629a050..6d01c893a88c8 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -636,6 +636,15 @@ struct VPlanTransforms {
static void makeCallWideningDecisions(VPlan &Plan, VFRange &Range,
VPRecipeBuilder &RecipeBuilder,
VPCostContext &CostCtx);
+
+ /// Replace truncates of a wide induction, or of that induction's increment,
+ /// by a VPWidenIntOrFpInductionRecipe producing the truncated type directly.
+ /// The canonical induction is narrowed even when the target reports the
+ /// truncate as free. If narrowing is only profitable for a subset of VFs in
+ /// \p Range, Range.End is updated.
+ static void narrowInductionTruncates(VPlan &Plan, VFRange &Range,
+ const TargetTransformInfo &TTI,
+ PredicatedScalarEvolution &PSE);
};
} // namespace llvm
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
index fe71cee7c3492..5fd844e186e44 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
@@ -31,6 +31,7 @@
; CHECK-AFTER: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::makeMemOpWideningDecisions
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::makeScalarizationDecisions
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::makeCallWideningDecisions
+; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::narrowInductionTruncates
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::adjustFirstOrderRecurrenceMiddleUsers
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::clearReductionWrapFlags
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::optimizeFindIVReductions
diff --git a/llvm/test/Transforms/LoopVectorize/cast-induction.ll b/llvm/test/Transforms/LoopVectorize/cast-induction.ll
index e76a13ad4e003..0fdd5f11c537b 100644
--- a/llvm/test/Transforms/LoopVectorize/cast-induction.ll
+++ b/llvm/test/Transforms/LoopVectorize/cast-induction.ll
@@ -565,3 +565,66 @@ loop:
exit:
ret i64 %acc.next
}
+
+; The truncate feeds off an identity operation on the induction, which VPlan
+; folds away before the narrowing runs.
+define void @cast_induction_through_identity_op(ptr %dst) {
+; VF4-LABEL: define void @cast_induction_through_identity_op(
+; VF4-SAME: ptr [[DST:%.*]]) {
+; VF4-NEXT: [[ENTRY:.*:]]
+; VF4-NEXT: br label %[[VECTOR_PH:.*]]
+; VF4: [[VECTOR_PH]]:
+; VF4-NEXT: br label %[[VECTOR_BODY:.*]]
+; VF4: [[VECTOR_BODY]]:
+; VF4-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; VF4-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; VF4-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[DST]], i64 [[INDEX]]
+; VF4-NEXT: store <4 x i32> [[VEC_IND]], ptr [[TMP0]], align 4
+; VF4-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; VF4-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
+; VF4-NEXT: [[TMP1:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; VF4-NEXT: br i1 [[TMP1]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; VF4: [[MIDDLE_BLOCK]]:
+; VF4-NEXT: br label %[[EXIT:.*]]
+; VF4: [[EXIT]]:
+; VF4-NEXT: ret void
+;
+; IC2-LABEL: define void @cast_induction_through_identity_op(
+; IC2-SAME: ptr [[DST:%.*]]) {
+; IC2-NEXT: [[ENTRY:.*:]]
+; IC2-NEXT: br label %[[VECTOR_PH:.*]]
+; IC2: [[VECTOR_PH]]:
+; IC2-NEXT: br label %[[VECTOR_BODY:.*]]
+; IC2: [[VECTOR_BODY]]:
+; IC2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; IC2-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; IC2-NEXT: [[TMP1:%.*]] = trunc i64 [[INDEX]] to i32
+; IC2-NEXT: [[TMP2:%.*]] = add i32 [[TMP1]], 1
+; IC2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[DST]], i64 [[INDEX]]
+; IC2-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[DST]], i64 [[TMP0]]
+; IC2-NEXT: store i32 [[TMP1]], ptr [[TMP3]], align 4
+; IC2-NEXT: store i32 [[TMP2]], ptr [[TMP4]], align 4
+; IC2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 2
+; IC2-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; IC2-NEXT: br i1 [[TMP5]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; IC2: [[MIDDLE_BLOCK]]:
+; IC2-NEXT: br label %[[EXIT:.*]]
+; IC2: [[EXIT]]:
+; IC2-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %identity = mul i64 %iv, 1
+ %trunc = trunc i64 %identity to i32
+ %gep = getelementptr inbounds i32, ptr %dst, i64 %iv
+ store i32 %trunc, ptr %gep, align 4
+ %iv.next = add i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
More information about the llvm-commits
mailing list