[llvm] [VPlan] Fix VPScalarIVStepsRecipe for FSub inductions. (PR #219214)

via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 06:45:05 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Florian Hahn (fhahn)

<details>
<summary>Changes</summary>

The lane offsets passed to the new VPScalarIVStepsRecipe always count upwards (based on the canonical IV), independent of the induction opcode.

Remove code incorrectly negating the lane offset for inductions with FPSub opcode. The double negation caused incorrect results.

---
Full diff: https://github.com/llvm/llvm-project/pull/219214.diff


2 Files Affected:

- (modified) llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp (+4-5) 
- (modified) llvm/test/Transforms/LoopVectorize/float-induction.ll (+10-10) 


``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 21ea2a6b74ff5..14f4bcb2be467 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -570,11 +570,10 @@ static void addLaneToStartIndex(VPScalarIVStepsRecipe *Steps, unsigned Lane,
   // TODO: Retrieve the flags from Steps unconditionally.
   VPIRFlags Flags;
   if (BaseIVTy->isFloatingPointTy()) {
-    int SignedLane = static_cast<int>(Lane);
-    if (!OldStartIndex && Steps->getInductionOpcode() == Instruction::FSub)
-      SignedLane = -SignedLane;
-    LaneOffset = Plan.getOrAddLiveIn(ConstantFP::get(BaseIVTy, SignedLane));
-    AddOpcode = Steps->getInductionOpcode();
+    // The start index counts upwards, so accumulate with FAdd regardless of the
+    // induction opcode; see VPScalarIVStepsRecipe.
+    LaneOffset = Plan.getOrAddLiveIn(ConstantFP::get(BaseIVTy, Lane));
+    AddOpcode = Instruction::FAdd;
     Flags = VPIRFlags(FastMathFlags());
   } else {
     unsigned BaseIVBits = BaseIVTy->getScalarSizeInBits();
diff --git a/llvm/test/Transforms/LoopVectorize/float-induction.ll b/llvm/test/Transforms/LoopVectorize/float-induction.ll
index e8ed35ddbd224..cde00b90aa015 100644
--- a/llvm/test/Transforms/LoopVectorize/float-induction.ll
+++ b/llvm/test/Transforms/LoopVectorize/float-induction.ll
@@ -2173,11 +2173,11 @@ define void @fp_iv_used_in_gep_fsub(float %init, ptr noalias nocapture %A, float
 ; VEC4_INTERL1-NEXT:    [[DOTCAST5:%.*]] = sitofp i64 [[INDEX]] to float
 ; VEC4_INTERL1-NEXT:    [[TMP7:%.*]] = fmul fast float [[FPINC]], [[DOTCAST5]]
 ; VEC4_INTERL1-NEXT:    [[OFFSET_IDX:%.*]] = fsub fast float [[INIT]], [[TMP7]]
-; VEC4_INTERL1-NEXT:    [[TMP9:%.*]] = fmul fast float -1.000000e+00, [[FPINC]]
+; VEC4_INTERL1-NEXT:    [[TMP9:%.*]] = fmul fast float 1.000000e+00, [[FPINC]]
 ; VEC4_INTERL1-NEXT:    [[TMP10:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP9]]
-; VEC4_INTERL1-NEXT:    [[TMP11:%.*]] = fmul fast float -2.000000e+00, [[FPINC]]
+; VEC4_INTERL1-NEXT:    [[TMP11:%.*]] = fmul fast float 2.000000e+00, [[FPINC]]
 ; VEC4_INTERL1-NEXT:    [[TMP12:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP11]]
-; VEC4_INTERL1-NEXT:    [[TMP18:%.*]] = fmul fast float -3.000000e+00, [[FPINC]]
+; VEC4_INTERL1-NEXT:    [[TMP18:%.*]] = fmul fast float 3.000000e+00, [[FPINC]]
 ; VEC4_INTERL1-NEXT:    [[TMP20:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP18]]
 ; VEC4_INTERL1-NEXT:    [[TMP13:%.*]] = fptoui <4 x float> [[VEC_IND]] to <4 x i32>
 ; VEC4_INTERL1-NEXT:    [[TMP14:%.*]] = extractelement <4 x i32> [[TMP13]], i64 0
@@ -2246,19 +2246,19 @@ define void @fp_iv_used_in_gep_fsub(float %init, ptr noalias nocapture %A, float
 ; VEC4_INTERL2-NEXT:    [[DOTCAST3:%.*]] = sitofp i64 [[INDEX]] to float
 ; VEC4_INTERL2-NEXT:    [[TMP7:%.*]] = fmul fast float [[FPINC]], [[DOTCAST3]]
 ; VEC4_INTERL2-NEXT:    [[OFFSET_IDX:%.*]] = fsub fast float [[INIT]], [[TMP7]]
-; VEC4_INTERL2-NEXT:    [[TMP9:%.*]] = fmul fast float -1.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP9:%.*]] = fmul fast float 1.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP10:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP9]]
-; VEC4_INTERL2-NEXT:    [[TMP11:%.*]] = fmul fast float -2.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP11:%.*]] = fmul fast float 2.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP12:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP11]]
-; VEC4_INTERL2-NEXT:    [[TMP13:%.*]] = fmul fast float -3.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP13:%.*]] = fmul fast float 3.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP14:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP13]]
 ; VEC4_INTERL2-NEXT:    [[TMP15:%.*]] = fmul fast float 4.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP16:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP15]]
-; VEC4_INTERL2-NEXT:    [[TMP17:%.*]] = fmul fast float 3.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP17:%.*]] = fmul fast float 5.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP18:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP17]]
-; VEC4_INTERL2-NEXT:    [[TMP30:%.*]] = fmul fast float 2.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP30:%.*]] = fmul fast float 6.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP32:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP30]]
-; VEC4_INTERL2-NEXT:    [[TMP33:%.*]] = fmul fast float 1.000000e+00, [[FPINC]]
+; VEC4_INTERL2-NEXT:    [[TMP33:%.*]] = fmul fast float 7.000000e+00, [[FPINC]]
 ; VEC4_INTERL2-NEXT:    [[TMP19:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP33]]
 ; VEC4_INTERL2-NEXT:    [[TMP20:%.*]] = fptoui <4 x float> [[VEC_IND]] to <4 x i32>
 ; VEC4_INTERL2-NEXT:    [[TMP25:%.*]] = fptoui <4 x float> [[STEP_ADD]] to <4 x i32>
@@ -2393,7 +2393,7 @@ define void @fp_iv_used_in_gep_fsub(float %init, ptr noalias nocapture %A, float
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[DOTCAST5:%.*]] = sitofp i64 [[INDEX]] to float
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP7:%.*]] = fmul fast float [[FPINC]], [[DOTCAST5]]
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[OFFSET_IDX:%.*]] = fsub fast float [[INIT]], [[TMP7]]
-; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP12:%.*]] = fmul fast float -1.000000e+00, [[FPINC]]
+; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP12:%.*]] = fmul fast float 1.000000e+00, [[FPINC]]
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP8:%.*]] = fsub fast float [[OFFSET_IDX]], [[TMP12]]
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP9:%.*]] = fptoui <2 x float> [[VEC_IND]] to <2 x i32>
 ; VEC2_INTERL1_PRED_STORE-NEXT:    [[TMP10:%.*]] = extractelement <2 x i32> [[TMP9]], i64 0

``````````

</details>


https://github.com/llvm/llvm-project/pull/219214


More information about the llvm-commits mailing list