[llvm] [VPlan] Fix VPScalarIVStepsRecipe for FSub inductions. (PR #219214)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 06:44:29 PDT 2026


https://github.com/fhahn created https://github.com/llvm/llvm-project/pull/219214

The lane offsets passed to the new VPScalarIVStepsRecipe always count upwards (based on the canonical IV), independent of the induction opcode.

Remove code incorrectly negating the lane offset for inductions with FPSub opcode. The double negation caused incorrect results.

>From c7405479de89c687b05f6cb036592486a5eb7f19 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 27 Aug 2026 12:03:06 +0100
Subject: [PATCH] [VPlan] Fix VPScalarIVStepsRecipe for FSub inductions.

The lane offsets passed to the new VPScalarIVStepsRecipe always count
upwards (based on the canonical IV), independent of the induction
opcode.

Remove code incorrectly negating the lane offset for inductions with
FPSub opcode. The double negation caused incorrect results.
---
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp |  9 ++++-----
 .../LoopVectorize/float-induction.ll          | 20 +++++++++----------
 2 files changed, 14 insertions(+), 15 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 21ea2a6b74ff5..14f4bcb2be467 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -570,11 +570,10 @@ static void addLaneToStartIndex(VPScalarIVStepsRecipe *Steps, unsigned Lane,
   // TODO: Retrieve the flags from Steps unconditionally.
   VPIRFlags Flags;
   if (BaseIVTy->isFloatingPointTy()) {
-    int SignedLane = static_cast<int>(Lane);
-    if (!OldStartIndex && Steps->getInductionOpcode() == Instruction::FSub)
-      SignedLane = -SignedLane;
-    LaneOffset = Plan.getOrAddLiveIn(ConstantFP::get(BaseIVTy, SignedLane));
-    AddOpcode = Steps->getInductionOpcode();
+    // The start index counts upwards, so accumulate with FAdd regardless of the
+    // induction opcode; see VPScalarIVStepsRecipe.
+    LaneOffset = Plan.getOrAddLiveIn(ConstantFP::get(BaseIVTy, Lane));
+    AddOpcode = Instruction::FAdd;
     Flags = VPIRFlags(FastMathFlags());
   } else {
     unsigned BaseIVBits = BaseIVTy->getScalarSizeInBits();
diff --git a/llvm/test/Transforms/LoopVectorize/float-induction.ll b/llvm/test/Transforms/LoopVectorize/float-induction.ll
index e8ed35ddbd224..cde00b90aa015 100644
--- a/llvm/test/Transforms/LoopVectorize/float-induction.ll
+++ b/llvm/test/Transforms/LoopVectorize/float-induction.ll
@@ -2173,11 +2173,11 @@ define void @fp_iv_used_in_gep_fsub(float %init, ptr noalias nocapture %A, float
 ; VEC4_INTERL1-NEXT:    [[DOTCAST5:%.*]] = sitofp i64 [[INDEX]] to float
 ; VEC4_INTERL1-NEXT:    [[TMP7:%.*]] = fmul fast float [[FPINC]], [[DOTCAST5]]
 ; VEC4_INTERL1-NEXT:    [[OFFSET_IDX:%.*]] = fsub fast float [[INIT]], [[TMP7]]
-; VEC4_INTERL1-NEXT:    [[TMP9:%.*]] = fmul fast float -1.000000e+00, [[FPINC]]
+; VEC4_INTERL1-NEXT:    [[TMP9:%.*]] = fmul fast float 1.000000e+00, [[FPINC]]
 ; VEC4_INTERL1-NEXT:    [[TMP10:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP9]]
-; VEC4_INTERL1-NEXT:    [[TMP11:%.*]] = fmul fast float -2.000000e+00, [[FPINC]]
+; VEC4_INTERL1-NEXT:    [[TMP11:%.*]] = fmul fast float 2.000000e+00, [[FPINC]]
 ; VEC4_INTERL1-NEXT:    [[TMP12:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP11]]
-; VEC4_INTERL1-NEXT:    [[TMP18:%.*]] = fmul fast float -3.000000e+00, [[FPINC]]
+; VEC4_INTERL1-NEXT:    [[TMP18:%.*]] = fmul fast float 3.000000e+00, [[FPINC]]
 ; VEC4_INTERL1-NEXT:    [[TMP20:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP18]]
 ; VEC4_INTERL1-NEXT:    [[TMP13:%.*]] = fptoui <4 x float> [[VEC_IND]] to <4 x i32>
 ; VEC4_INTERL1-NEXT:    [[TMP14:%.*]] = extractelement <4 x i32> [[TMP13]], i64 0
@@ -2246,19 +2246,19 @@ define void @fp_iv_used_in_gep_fsub(float %init, ptr noalias nocapture %A, float
 ; VEC4_INTERL2-NEXT:    [[DOTCAST3:%.*]] = sitofp i64 [[INDEX]] to float
 ; VEC4_INTERL2-NEXT:    [[TMP7:%.*]] = fmul fast float [[FPINC]], [[DOTCAST3]]
 ; VEC4_INTERL2-NEXT:    [[OFFSET_IDX:%.*]] = fsub fast float [[INIT]], [[TMP7]]
-; VEC4_INTERL2-NEXT:    [[TMP9:%.*]] = fmul fast float -1.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP9:%.*]] = fmul fast float 1.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP10:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP9]]
-; VEC4_INTERL2-NEXT:    [[TMP11:%.*]] = fmul fast float -2.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP11:%.*]] = fmul fast float 2.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP12:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP11]]
-; VEC4_INTERL2-NEXT:    [[TMP13:%.*]] = fmul fast float -3.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP13:%.*]] = fmul fast float 3.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP14:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP13]]
 ; VEC4_INTERL2-NEXT:    [[TMP15:%.*]] = fmul fast float 4.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP16:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP15]]
-; VEC4_INTERL2-NEXT:    [[TMP17:%.*]] = fmul fast float 3.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP17:%.*]] = fmul fast float 5.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP18:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP17]]
-; VEC4_INTERL2-NEXT:    [[TMP30:%.*]] = fmul fast float 2.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP30:%.*]] = fmul fast float 6.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP32:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP30]]
-; VEC4_INTERL2-NEXT:    [[TMP33:%.*]] = fmul fast float 1.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP33:%.*]] = fmul fast float 7.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP19:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP33]]
 ; VEC4_INTERL2-NEXT:    [[TMP20:%.*]] = fptoui <4 x float> [[VEC_IND]] to <4 x i32>
 ; VEC4_INTERL2-NEXT:    [[TMP25:%.*]] = fptoui <4 x float> [[STEP_ADD]] to <4 x i32>
@@ -2393,7 +2393,7 @@ define void @fp_iv_used_in_gep_fsub(float %init, ptr noalias nocapture %A, float
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[DOTCAST5:%.*]] = sitofp i64 [[INDEX]] to float
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP7:%.*]] = fmul fast float [[FPINC]], [[DOTCAST5]]
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[OFFSET_IDX:%.*]] = fsub fast float [[INIT]], [[TMP7]]
-; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP12:%.*]] = fmul fast float -1.000000e+00, [[FPINC]]
+; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP12:%.*]] = fmul fast float 1.000000e+00, [[FPINC]]
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP8:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP12]]
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP9:%.*]] = fptoui <2 x float> [[VEC_IND]] to <2 x i32>
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP10:%.*]] = extractelement <2 x i32> [[TMP9]], i64 0



More information about the llvm-commits mailing list