[llvm] [VPlan] Only retain NSW when possible in getFlagsFromIndDesc. (PR #226315)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 25 01:27:34 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/226315
>From 8c6e40494e74392c459075521c1d8033fc5ed3da Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 24 Sep 2026 20:53:24 +0100
Subject: [PATCH] [VPlan] Only retain NSW when possible in getFlagsFromIndDesc.
getFlagsFromIndDesc is used to get the flags to use for wide inductions.
Those are normalized to Adds, with the step negated if the induction
binop is Sub.
Only retain all flags for Add. For Sub, never retain NUW and only keep
NSW if the step is known to not be signed min.
Alive2 Proofs showing incorrect NUW/NSW transfer and transfer of NSW if
step is != signed INT_MIN: https://alive2.llvm.org/ce/z/zGKeZg
Fixes https://github.com/llvm/llvm-project/issues/224024.
---
llvm/lib/Transforms/Vectorize/VPlanUtils.h | 19 +++++++++++----
.../x86-interleaved-accesses-masked-group.ll | 4 ++--
.../LoopVectorize/induction-wrapflags.ll | 23 +++++++++----------
3 files changed, 27 insertions(+), 19 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.h b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
index 8b5309c3e56fb8..1f49fb33bdfe4f 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
@@ -137,15 +137,24 @@ getOpcodeOrIntrinsicID(const VPValue *V);
std::optional<MemoryLocation> getMemoryLocation(const VPRecipeBase &R);
/// Extracts and returns NoWrap and FastMath flags from the induction binop in
-/// \p ID.
+/// \p ID, for use on a wide induction, which adds the step.
inline VPIRFlags getFlagsFromIndDesc(const InductionDescriptor &ID) {
if (ID.getKind() == InductionDescriptor::IK_FpInduction)
return ID.getInductionBinOp()->getFastMathFlags();
- if (auto *OBO = dyn_cast_if_present<OverflowingBinaryOperator>(
- ID.getInductionBinOp()))
- return VPIRFlags::WrapFlagsTy(OBO->hasNoUnsignedWrap(),
- OBO->hasNoSignedWrap());
+ if (auto *AddO = dyn_cast_if_present<AddOperator>(ID.getInductionBinOp())) {
+ return VPIRFlags::WrapFlagsTy(AddO->hasNoUnsignedWrap(),
+ AddO->hasNoSignedWrap());
+ }
+
+ // The step of a sub induction is negated, so NUW cannot be preserved. NSW
+ // can, if the step is not the signed minimum.
+ if (auto *SubO = dyn_cast_if_present<SubOperator>(ID.getInductionBinOp())) {
+ ConstantInt *Step = ID.getConstIntStepValue();
+ return VPIRFlags::WrapFlagsTy(false,
+ SubO->hasNoSignedWrap() && Step &&
+ !Step->isMinValue(/*IsSigned=*/true));
+ }
assert(ID.getKind() == InductionDescriptor::IK_IntInduction &&
"Expected int induction");
diff --git a/llvm/test/Transforms/LoopVectorize/X86/x86-interleaved-accesses-masked-group.ll b/llvm/test/Transforms/LoopVectorize/X86/x86-interleaved-accesses-masked-group.ll
index acbdf269934dfe..6f5b9d4403af12 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/x86-interleaved-accesses-masked-group.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/x86-interleaved-accesses-masked-group.ll
@@ -1925,7 +1925,7 @@ define void @masked_strided2_reverse(ptr noalias nocapture readonly %p, ptr noal
; DISABLED_MASKED_STRIDED-NEXT: br label %[[PRED_STORE_CONTINUE60]]
; DISABLED_MASKED_STRIDED: [[PRED_STORE_CONTINUE60]]:
; DISABLED_MASKED_STRIDED-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
-; DISABLED_MASKED_STRIDED-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <8 x i32> [[VEC_IND]], splat (i32 -8)
+; DISABLED_MASKED_STRIDED-NEXT: [[VEC_IND_NEXT]] = add nsw <8 x i32> [[VEC_IND]], splat (i32 -8)
; DISABLED_MASKED_STRIDED-NEXT: [[TMP157:%.*]] = icmp eq i32 [[INDEX_NEXT]], 1024
; DISABLED_MASKED_STRIDED-NEXT: br i1 [[TMP157]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
; DISABLED_MASKED_STRIDED: [[MIDDLE_BLOCK]]:
@@ -2232,7 +2232,7 @@ define void @masked_strided2_reverse(ptr noalias nocapture readonly %p, ptr noal
; ENABLED_MASKED_STRIDED-NEXT: br label %[[PRED_STORE_CONTINUE60]]
; ENABLED_MASKED_STRIDED: [[PRED_STORE_CONTINUE60]]:
; ENABLED_MASKED_STRIDED-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
-; ENABLED_MASKED_STRIDED-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <8 x i32> [[VEC_IND]], splat (i32 -8)
+; ENABLED_MASKED_STRIDED-NEXT: [[VEC_IND_NEXT]] = add nsw <8 x i32> [[VEC_IND]], splat (i32 -8)
; ENABLED_MASKED_STRIDED-NEXT: [[TMP157:%.*]] = icmp eq i32 [[INDEX_NEXT]], 1024
; ENABLED_MASKED_STRIDED-NEXT: br i1 [[TMP157]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
; ENABLED_MASKED_STRIDED: [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/induction-wrapflags.ll b/llvm/test/Transforms/LoopVectorize/induction-wrapflags.ll
index 3216689f8b28c2..fc0f331c039d5f 100644
--- a/llvm/test/Transforms/LoopVectorize/induction-wrapflags.ll
+++ b/llvm/test/Transforms/LoopVectorize/induction-wrapflags.ll
@@ -110,7 +110,6 @@ exit:
ret i32 0
}
-; FIXME: Currently we incorrectly add NUW & NSW to the vector induction.
; Test for https://github.com/llvm/llvm-project/issues/224024.
define i32 @sub_induction_nuw_nsw(i32 %start, i32 %step, i32 %n) {
; CHECK-LABEL: define i32 @sub_induction_nuw_nsw(
@@ -128,9 +127,9 @@ define i32 @sub_induction_nuw_nsw(i32 %start, i32 %step, i32 %n) {
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i32> poison, i32 [[TMP0]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT1]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP4:%.*]] = mul nuw nsw <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[BROADCAST_SPLAT2]]
-; CHECK-NEXT: [[INDUCTION:%.*]] = add nuw nsw <4 x i32> [[BROADCAST_SPLAT]], [[TMP4]]
-; CHECK-NEXT: [[TMP5:%.*]] = shl nuw nsw i32 [[TMP0]], 2
+; CHECK-NEXT: [[TMP4:%.*]] = mul <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[BROADCAST_SPLAT2]]
+; CHECK-NEXT: [[INDUCTION:%.*]] = add <4 x i32> [[BROADCAST_SPLAT]], [[TMP4]]
+; CHECK-NEXT: [[TMP5:%.*]] = shl i32 [[TMP0]], 2
; CHECK-NEXT: [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <4 x i32> poison, i32 [[TMP5]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT4:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT3]], <4 x i32> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
@@ -140,7 +139,7 @@ define i32 @sub_induction_nuw_nsw(i32 %start, i32 %step, i32 %n) {
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ [[INDUCTION]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP6]] = add <4 x i32> [[VEC_PHI]], [[VEC_IND]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i32> [[VEC_IND]], [[BROADCAST_SPLAT4]]
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], [[BROADCAST_SPLAT4]]
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -183,7 +182,7 @@ define void @sub_induction_nuw_nsw_constant_step(ptr noalias %dst, i32 %start, i
; CHECK-NEXT: [[TMP4:%.*]] = add i32 [[START]], [[TMP3]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[START]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[INDUCTION:%.*]] = add nuw nsw <4 x i32> [[BROADCAST_SPLAT]], <i32 0, i32 -3, i32 -6, i32 -9>
+; CHECK-NEXT: [[INDUCTION:%.*]] = add nsw <4 x i32> [[BROADCAST_SPLAT]], <i32 0, i32 -3, i32 -6, i32 -9>
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
@@ -191,7 +190,7 @@ define void @sub_induction_nuw_nsw_constant_step(ptr noalias %dst, i32 %start, i
; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[DST]], i32 [[INDEX]]
; CHECK-NEXT: store <4 x i32> [[VEC_IND]], ptr [[TMP5]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i32> [[VEC_IND]], splat (i32 -12)
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add nsw <4 x i32> [[VEC_IND]], splat (i32 -12)
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -233,7 +232,7 @@ define void @sub_induction_nsw_signed_min_step(ptr noalias %dst, i32 %start, i32
; CHECK-NEXT: [[TMP4:%.*]] = add i32 [[START]], [[TMP3]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[START]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[INDUCTION:%.*]] = add nsw <4 x i32> [[BROADCAST_SPLAT]], <i32 0, i32 -2147483648, i32 0, i32 -2147483648>
+; CHECK-NEXT: [[INDUCTION:%.*]] = add <4 x i32> [[BROADCAST_SPLAT]], <i32 0, i32 -2147483648, i32 0, i32 -2147483648>
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
@@ -286,19 +285,19 @@ define void @sub_induction_nsw_interleave(ptr noalias %dst, i32 %start, i32 %ste
; CHECK-NEXT: [[TMP6:%.*]] = mul <4 x i32> splat (i32 4), [[BROADCAST_SPLAT]]
; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i32> poison, i32 [[START]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT1]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP7:%.*]] = mul nsw <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[BROADCAST_SPLAT]]
-; CHECK-NEXT: [[INDUCTION:%.*]] = add nsw <4 x i32> [[BROADCAST_SPLAT2]], [[TMP7]]
+; CHECK-NEXT: [[TMP7:%.*]] = mul <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[INDUCTION:%.*]] = add <4 x i32> [[BROADCAST_SPLAT2]], [[TMP7]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ [[INDUCTION]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[STEP_ADD:%.*]] = add nsw <4 x i32> [[VEC_IND]], [[TMP6]]
+; CHECK-NEXT: [[STEP_ADD:%.*]] = add <4 x i32> [[VEC_IND]], [[TMP6]]
; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[DST]], i32 [[INDEX]]
; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP8]], i64 4
; CHECK-NEXT: store <4 x i32> [[VEC_IND]], ptr [[TMP8]], align 4
; CHECK-NEXT: store <4 x i32> [[STEP_ADD]], ptr [[TMP9]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add nsw <4 x i32> [[STEP_ADD]], [[TMP6]]
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[STEP_ADD]], [[TMP6]]
; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
More information about the llvm-commits
mailing list