[llvm] [VPlan] Split legalizeAndOptimizeIVs in 2 phases. (PR #217765)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Mon Aug 24 01:38:30 PDT 2026


https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/217765

>From 81cccf9f33410b36d9e7bf05deb9bb99dfcd6317 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 20 Aug 2026 14:31:48 +0100
Subject: [PATCH] [VPlan] Split legalizeAndOptimizeIVs in 2 phases.

Split the transform into 2 phases:

1. narrow all users of all IVs
2. replace wide IVs if all users are scalar.

Together with removing the old, wide recipes, this allows us to catch
slightly more cases.
---
 .../Transforms/Vectorize/VPlanTransforms.cpp  | 27 +++++++++++--------
 .../vector-call-linear-args-no-wide-iv.ll     | 22 ++++++---------
 .../VPlan/vplan-narrow-iv-users.ll            | 13 ++++-----
 3 files changed, 29 insertions(+), 33 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index baf1716c078cd..6eaf8a4ac2e68 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -732,16 +732,17 @@ void VPlanTransforms::removeDeadRecipes(VPlan &Plan) {
 static void legalizeAndOptimizeInductions(VPlan &Plan) {
   VPBasicBlock *HeaderVPBB = Plan.getVectorLoopRegion()->getEntryBasicBlock();
   bool HasOnlyVectorVFs = !Plan.hasScalarVFOnly();
-  VPBuilder Builder(HeaderVPBB, HeaderVPBB->getFirstNonPhi());
-  for (VPRecipeBase &Phi : HeaderVPBB->phis()) {
-    auto *PhiR = dyn_cast<VPWidenInductionRecipe>(&Phi);
-    if (!PhiR)
-      continue;
 
-    // Try to narrow wide and replicating recipes to uniform recipes, based on
-    // VPlan analysis.
-    // TODO: Apply to all recipes in the future, to replace legacy uniformity
-    // analysis.
+  SmallVector<VPWidenInductionRecipe *> WideIVs;
+  for (VPRecipeBase &Phi : HeaderVPBB->phis())
+    if (auto *PhiR = dyn_cast<VPWidenInductionRecipe>(&Phi))
+      WideIVs.push_back(PhiR);
+
+  // Try to narrow wide and replicating recipes to uniform recipes, based on
+  // VPlan analysis.
+  // TODO: Apply to all recipes in the future, to replace legacy uniformity
+  // analysis.
+  for (VPWidenInductionRecipe *PhiR : WideIVs) {
     auto Users = vputils::collectUsersRecursively(PhiR);
     for (VPUser *U : reverse(Users)) {
       auto *Def = dyn_cast<VPRecipeWithIRFlags>(U);
@@ -767,11 +768,15 @@ static void legalizeAndOptimizeInductions(VPlan &Plan) {
           Def->getUnderlyingInstr());
       Clone->insertAfter(Def);
       Def->replaceAllUsesWith(Clone);
+      Def->eraseFromParent();
     }
+  }
 
+  VPBuilder Builder(HeaderVPBB, HeaderVPBB->getFirstNonPhi());
+  for (VPWidenInductionRecipe *PhiR : WideIVs) {
     // Replace wide pointer inductions which have only their scalars used by
     // PtrAdd(IndStart, ScalarIVSteps (0, Step)).
-    if (auto *PtrIV = dyn_cast<VPWidenPointerInductionRecipe>(&Phi)) {
+    if (auto *PtrIV = dyn_cast<VPWidenPointerInductionRecipe>(PhiR)) {
       if (!Plan.hasScalarVFOnly() &&
           !PtrIV->onlyScalarsGenerated(Plan.hasScalableVF()))
         continue;
@@ -784,7 +789,7 @@ static void legalizeAndOptimizeInductions(VPlan &Plan) {
 
     // Replace widened induction with scalar steps for users that only use
     // scalars.
-    auto *WideIV = cast<VPWidenIntOrFpInductionRecipe>(&Phi);
+    auto *WideIV = cast<VPWidenIntOrFpInductionRecipe>(PhiR);
     if (HasOnlyVectorVFs && none_of(WideIV->users(), [WideIV](VPUser *U) {
           return U->usesScalars(WideIV);
         }))
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/vector-call-linear-args-no-wide-iv.ll b/llvm/test/Transforms/LoopVectorize/AArch64/vector-call-linear-args-no-wide-iv.ll
index 49d6d49622a0c..384d3b19ff80b 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/vector-call-linear-args-no-wide-iv.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/vector-call-linear-args-no-wide-iv.ll
@@ -15,23 +15,17 @@ define void @narrow_linear_call_arg(ptr noalias %a, i64 %n) {
 ; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 1000, [[TMP1]]
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 1000, [[N_MOD_VF]]
-; CHECK-NEXT:    [[TMP2:%.*]] = call <vscale x 4 x i64> @llvm.stepvector.nxv4i64()
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 4 x i64> poison, i64 [[TMP1]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 4 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 4 x i64> poison, <vscale x 4 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 4 x i64> [ [[TMP2]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <vscale x 4 x i64> [[VEC_IND]], i64 0
-; CHECK-NEXT:    [[TMP4:%.*]] = trunc i64 [[TMP3]] to i32
-; CHECK-NEXT:    [[TMP5:%.*]] = mul i32 [[TMP4]], 3
-; CHECK-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i32> @vec_bar_linear3_nomask_sve(i32 [[TMP5]])
-; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT:    store <vscale x 4 x i32> [[TMP6]], ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = trunc i64 [[INDEX]] to i32
+; CHECK-NEXT:    [[TMP3:%.*]] = mul i32 [[TMP2]], 3
+; CHECK-NEXT:    [[TMP4:%.*]] = call <vscale x 4 x i32> @vec_bar_linear3_nomask_sve(i32 [[TMP3]])
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    store <vscale x 4 x i32> [[TMP4]], ptr [[TMP5]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <vscale x 4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
-; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 1000, [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
@@ -42,7 +36,7 @@ define void @narrow_linear_call_arg(ptr noalias %a, i64 %n) {
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[IV_TRUNC:%.*]] = trunc i64 [[IV]] to i32
 ; CHECK-NEXT:    [[TREBLED:%.*]] = mul i32 [[IV_TRUNC]], 3
-; CHECK-NEXT:    [[DATA:%.*]] = call i32 @bar(i32 [[TREBLED]]) #[[ATTR3:[0-9]+]]
+; CHECK-NEXT:    [[DATA:%.*]] = call i32 @bar(i32 [[TREBLED]]) #[[ATTR2:[0-9]+]]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
 ; CHECK-NEXT:    store i32 [[DATA]], ptr [[GEP_A]], align 4
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-narrow-iv-users.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-narrow-iv-users.ll
index 8e7cda8f59ed4..7cb871ab734e0 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-narrow-iv-users.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-narrow-iv-users.ll
@@ -25,10 +25,9 @@ define void @narrow_iv_user_chain(ptr noalias %A, ptr noalias %B) {
 ; CHECK-NEXT:      CLONE ir<%and> = and vp<[[VP5]]>, ir<-4>
 ; CHECK-NEXT:      CLONE ir<%gep.ld> = getelementptr inbounds ir<%A>, ir<%and>
 ; CHECK-NEXT:      CLONE ir<%ld> = load ir<%gep.ld>
-; CHECK-NEXT:      WIDEN ir<%calc> = add nsw ir<%ld>, ir<42>
-; CHECK-NEXT:      CLONE ir<%calc>.1 = add nsw ir<%ld>, ir<42>
+; CHECK-NEXT:      CLONE ir<%calc> = add nsw ir<%ld>, ir<42>
 ; CHECK-NEXT:      CLONE ir<%gep.st> = getelementptr inbounds ir<%B>, ir<%and>
-; CHECK-NEXT:      REPLICATE store ir<%calc>.1, ir<%gep.st>
+; CHECK-NEXT:      REPLICATE store ir<%calc>, ir<%gep.st>
 ; CHECK-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
 ; CHECK-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; CHECK-NEXT:    No successors
@@ -81,12 +80,10 @@ define void @narrow_iv_user_chain_multiple_levels(ptr noalias %A, ptr noalias %B
 ; CHECK-NEXT:      CLONE ir<%and> = and vp<[[VP5]]>, ir<-4>
 ; CHECK-NEXT:      CLONE ir<%gep.ld> = getelementptr inbounds ir<%A>, ir<%and>
 ; CHECK-NEXT:      CLONE ir<%ld> = load ir<%gep.ld>
-; CHECK-NEXT:      WIDEN ir<%calc> = add nsw ir<%ld>, ir<42>
-; CHECK-NEXT:      CLONE ir<%calc>.1 = add nsw ir<%ld>, ir<42>
-; CHECK-NEXT:      WIDEN ir<%calc2> = mul nsw ir<%calc>.1, ir<3>
-; CHECK-NEXT:      CLONE ir<%calc2>.1 = mul nsw ir<%calc>.1, ir<3>
+; CHECK-NEXT:      CLONE ir<%calc> = add nsw ir<%ld>, ir<42>
+; CHECK-NEXT:      CLONE ir<%calc2> = mul nsw ir<%calc>, ir<3>
 ; CHECK-NEXT:      CLONE ir<%gep.st> = getelementptr inbounds ir<%B>, ir<%and>
-; CHECK-NEXT:      REPLICATE store ir<%calc2>.1, ir<%gep.st>
+; CHECK-NEXT:      REPLICATE store ir<%calc2>, ir<%gep.st>
 ; CHECK-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
 ; CHECK-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; CHECK-NEXT:    No successors



More information about the llvm-commits mailing list