[llvm] [VPlan] Split legalizeAndOptimizeIVs in 2 phases. (PR #217765)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 24 01:38:30 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/217765
>From 81cccf9f33410b36d9e7bf05deb9bb99dfcd6317 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 20 Aug 2026 14:31:48 +0100
Subject: [PATCH] [VPlan] Split legalizeAndOptimizeIVs in 2 phases.
Split the transform into 2 phases:
1. narrow all users of all IVs
2. replace wide IVs if all users are scalar.
Together with removing the old, wide recipes, this allows us to catch
slightly more cases.
---
.../Transforms/Vectorize/VPlanTransforms.cpp | 27 +++++++++++--------
.../vector-call-linear-args-no-wide-iv.ll | 22 ++++++---------
.../VPlan/vplan-narrow-iv-users.ll | 13 ++++-----
3 files changed, 29 insertions(+), 33 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index baf1716c078cd..6eaf8a4ac2e68 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -732,16 +732,17 @@ void VPlanTransforms::removeDeadRecipes(VPlan &Plan) {
static void legalizeAndOptimizeInductions(VPlan &Plan) {
VPBasicBlock *HeaderVPBB = Plan.getVectorLoopRegion()->getEntryBasicBlock();
bool HasOnlyVectorVFs = !Plan.hasScalarVFOnly();
- VPBuilder Builder(HeaderVPBB, HeaderVPBB->getFirstNonPhi());
- for (VPRecipeBase &Phi : HeaderVPBB->phis()) {
- auto *PhiR = dyn_cast<VPWidenInductionRecipe>(&Phi);
- if (!PhiR)
- continue;
- // Try to narrow wide and replicating recipes to uniform recipes, based on
- // VPlan analysis.
- // TODO: Apply to all recipes in the future, to replace legacy uniformity
- // analysis.
+ SmallVector<VPWidenInductionRecipe *> WideIVs;
+ for (VPRecipeBase &Phi : HeaderVPBB->phis())
+ if (auto *PhiR = dyn_cast<VPWidenInductionRecipe>(&Phi))
+ WideIVs.push_back(PhiR);
+
+ // Try to narrow wide and replicating recipes to uniform recipes, based on
+ // VPlan analysis.
+ // TODO: Apply to all recipes in the future, to replace legacy uniformity
+ // analysis.
+ for (VPWidenInductionRecipe *PhiR : WideIVs) {
auto Users = vputils::collectUsersRecursively(PhiR);
for (VPUser *U : reverse(Users)) {
auto *Def = dyn_cast<VPRecipeWithIRFlags>(U);
@@ -767,11 +768,15 @@ static void legalizeAndOptimizeInductions(VPlan &Plan) {
Def->getUnderlyingInstr());
Clone->insertAfter(Def);
Def->replaceAllUsesWith(Clone);
+ Def->eraseFromParent();
}
+ }
+ VPBuilder Builder(HeaderVPBB, HeaderVPBB->getFirstNonPhi());
+ for (VPWidenInductionRecipe *PhiR : WideIVs) {
// Replace wide pointer inductions which have only their scalars used by
// PtrAdd(IndStart, ScalarIVSteps (0, Step)).
- if (auto *PtrIV = dyn_cast<VPWidenPointerInductionRecipe>(&Phi)) {
+ if (auto *PtrIV = dyn_cast<VPWidenPointerInductionRecipe>(PhiR)) {
if (!Plan.hasScalarVFOnly() &&
!PtrIV->onlyScalarsGenerated(Plan.hasScalableVF()))
continue;
@@ -784,7 +789,7 @@ static void legalizeAndOptimizeInductions(VPlan &Plan) {
// Replace widened induction with scalar steps for users that only use
// scalars.
- auto *WideIV = cast<VPWidenIntOrFpInductionRecipe>(&Phi);
+ auto *WideIV = cast<VPWidenIntOrFpInductionRecipe>(PhiR);
if (HasOnlyVectorVFs && none_of(WideIV->users(), [WideIV](VPUser *U) {
return U->usesScalars(WideIV);
}))
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/vector-call-linear-args-no-wide-iv.ll b/llvm/test/Transforms/LoopVectorize/AArch64/vector-call-linear-args-no-wide-iv.ll
index 49d6d49622a0c..384d3b19ff80b 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/vector-call-linear-args-no-wide-iv.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/vector-call-linear-args-no-wide-iv.ll
@@ -15,23 +15,17 @@ define void @narrow_linear_call_arg(ptr noalias %a, i64 %n) {
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 1000, [[TMP1]]
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 1000, [[N_MOD_VF]]
-; CHECK-NEXT: [[TMP2:%.*]] = call <vscale x 4 x i64> @llvm.stepvector.nxv4i64()
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 4 x i64> poison, i64 [[TMP1]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 4 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 4 x i64> poison, <vscale x 4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <vscale x 4 x i64> [ [[TMP2]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP3:%.*]] = extractelement <vscale x 4 x i64> [[VEC_IND]], i64 0
-; CHECK-NEXT: [[TMP4:%.*]] = trunc i64 [[TMP3]] to i32
-; CHECK-NEXT: [[TMP5:%.*]] = mul i32 [[TMP4]], 3
-; CHECK-NEXT: [[TMP6:%.*]] = call <vscale x 4 x i32> @vec_bar_linear3_nomask_sve(i32 [[TMP5]])
-; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT: store <vscale x 4 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = trunc i64 [[INDEX]] to i32
+; CHECK-NEXT: [[TMP3:%.*]] = mul i32 [[TMP2]], 3
+; CHECK-NEXT: [[TMP4:%.*]] = call <vscale x 4 x i32> @vec_bar_linear3_nomask_sve(i32 [[TMP3]])
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: store <vscale x 4 x i32> [[TMP4]], ptr [[TMP5]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <vscale x 4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
-; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 1000, [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
@@ -42,7 +36,7 @@ define void @narrow_linear_call_arg(ptr noalias %a, i64 %n) {
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[IV_TRUNC:%.*]] = trunc i64 [[IV]] to i32
; CHECK-NEXT: [[TREBLED:%.*]] = mul i32 [[IV_TRUNC]], 3
-; CHECK-NEXT: [[DATA:%.*]] = call i32 @bar(i32 [[TREBLED]]) #[[ATTR3:[0-9]+]]
+; CHECK-NEXT: [[DATA:%.*]] = call i32 @bar(i32 [[TREBLED]]) #[[ATTR2:[0-9]+]]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: store i32 [[DATA]], ptr [[GEP_A]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-narrow-iv-users.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-narrow-iv-users.ll
index 8e7cda8f59ed4..7cb871ab734e0 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-narrow-iv-users.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-narrow-iv-users.ll
@@ -25,10 +25,9 @@ define void @narrow_iv_user_chain(ptr noalias %A, ptr noalias %B) {
; CHECK-NEXT: CLONE ir<%and> = and vp<[[VP5]]>, ir<-4>
; CHECK-NEXT: CLONE ir<%gep.ld> = getelementptr inbounds ir<%A>, ir<%and>
; CHECK-NEXT: CLONE ir<%ld> = load ir<%gep.ld>
-; CHECK-NEXT: WIDEN ir<%calc> = add nsw ir<%ld>, ir<42>
-; CHECK-NEXT: CLONE ir<%calc>.1 = add nsw ir<%ld>, ir<42>
+; CHECK-NEXT: CLONE ir<%calc> = add nsw ir<%ld>, ir<42>
; CHECK-NEXT: CLONE ir<%gep.st> = getelementptr inbounds ir<%B>, ir<%and>
-; CHECK-NEXT: REPLICATE store ir<%calc>.1, ir<%gep.st>
+; CHECK-NEXT: REPLICATE store ir<%calc>, ir<%gep.st>
; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
; CHECK-NEXT: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
; CHECK-NEXT: No successors
@@ -81,12 +80,10 @@ define void @narrow_iv_user_chain_multiple_levels(ptr noalias %A, ptr noalias %B
; CHECK-NEXT: CLONE ir<%and> = and vp<[[VP5]]>, ir<-4>
; CHECK-NEXT: CLONE ir<%gep.ld> = getelementptr inbounds ir<%A>, ir<%and>
; CHECK-NEXT: CLONE ir<%ld> = load ir<%gep.ld>
-; CHECK-NEXT: WIDEN ir<%calc> = add nsw ir<%ld>, ir<42>
-; CHECK-NEXT: CLONE ir<%calc>.1 = add nsw ir<%ld>, ir<42>
-; CHECK-NEXT: WIDEN ir<%calc2> = mul nsw ir<%calc>.1, ir<3>
-; CHECK-NEXT: CLONE ir<%calc2>.1 = mul nsw ir<%calc>.1, ir<3>
+; CHECK-NEXT: CLONE ir<%calc> = add nsw ir<%ld>, ir<42>
+; CHECK-NEXT: CLONE ir<%calc2> = mul nsw ir<%calc>, ir<3>
; CHECK-NEXT: CLONE ir<%gep.st> = getelementptr inbounds ir<%B>, ir<%and>
-; CHECK-NEXT: REPLICATE store ir<%calc2>.1, ir<%gep.st>
+; CHECK-NEXT: REPLICATE store ir<%calc2>, ir<%gep.st>
; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
; CHECK-NEXT: EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
; CHECK-NEXT: No successors
More information about the llvm-commits
mailing list