[llvm] [VPlan] Create SCEV before any VPIRInstructions to check for overflow (PR #177911)
Jim Lin via llvm-commits
llvm-commits at lists.llvm.org
Tue Jan 27 18:49:37 PST 2026
https://github.com/tclin914 updated https://github.com/llvm/llvm-project/pull/177911
>From 61a159cd98e60cf56051fca49c1e055f6f33e135 Mon Sep 17 00:00:00 2001
From: Jim Lin <jim at andestech.com>
Date: Mon, 26 Jan 2026 14:45:18 +0800
Subject: [PATCH 1/4] [VPlan] Create SCEV before any VPIRInstructions to check
for overflow
This PR tried to fix the assertion fail at VPlanTransforms.cpp:4862
since SCEV was created after VPIRInstructions.
The tripcount in scalable-predication.ll was changed from constant value
256 to non-constant value %n to avoid VPIRInstructions optimized out,
which cannot trigger the assertion fail.
The orders in ir-bb<entry> from:
ir-bb<entry>:
EMIT vp<%2> = EXPAND SCEV (1 umax %n)
EMIT vp<%3> = sub ir<-1>, vp<%2>
EMIT vp<%4> = EXPAND SCEV (4 * vscale)<nuw>
EMIT vp<%5> = icmp ult vp<%3>, vp<%4>
EMIT branch-on-cond vp<%5>
Successor(s): scalar.ph, vector.ph
to:
ir-bb<entry>:
EMIT vp<%2> = EXPAND SCEV (1 umax %n)
EMIT vp<%3> = EXPAND SCEV (4 * vscale)<nuw>
EMIT vp<%4> = sub ir<-1>, vp<%2>
EMIT vp<%5> = icmp ult vp<%4>, vp<%3>
EMIT branch-on-cond vp<%5>
Successor(s): scalar.ph, vector.ph
---
llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp | 4 +++-
llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp | 2 +-
.../LoopVectorize/scalable-predication.ll | 13 ++++++++-----
3 files changed, 12 insertions(+), 7 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 1f8243d5f6c72..37e4a9cb9e0e0 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1040,6 +1040,8 @@ void VPlanTransforms::addMinimumIterationCheck(
// an overflow to zero when updating induction variables and so an
// additional overflow check is required before entering the vector loop.
+ VPValue *StepV = Builder.createExpandSCEV(Step);
+
// Get the maximum unsigned value for the type.
VPValue *MaxUIntTripCount =
Plan.getConstantInt(cast<IntegerType>(TripCountTy)->getMask());
@@ -1051,7 +1053,7 @@ void VPlanTransforms::addMinimumIterationCheck(
// FIXME: Should only check VF * UF, but currently checks Step=max(VF*UF,
// minProfitableTripCount).
TripCountCheck = Builder.createICmp(ICmpInst::ICMP_ULT, DistanceToMax,
- Builder.createExpandSCEV(Step), DL);
+ StepV, DL);
} else {
// TripCountCheck = false, folding tail implies positive vector trip
// count.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index a39b171ab4cd6..fa34de81bd001 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4861,7 +4861,7 @@ VPlanTransforms::expandSCEVs(VPlan &Plan, ScalarEvolution &SE) {
}
assert(none_of(*Entry, IsaPred<VPExpandSCEVRecipe>) &&
"VPExpandSCEVRecipes must be at the beginning of the entry block, "
- "after any VPIRInstructions");
+ "before any VPIRInstructions");
// Add IR instructions in the entry basic block but not in the VPIRBasicBlock
// to the VPIRBasicBlock.
auto EI = Entry->begin();
diff --git a/llvm/test/Transforms/LoopVectorize/scalable-predication.ll b/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
index f5b445641097c..3747792416039 100644
--- a/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
+++ b/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
@@ -5,18 +5,21 @@
; deliberately doesn't correspond to an in-tree backend since those
; *do* have vscale as power-of-two) exercises the code required for the
; minimum iteration check in the non-power-of-two case.
-define void @foo(i32 %val, ptr dereferenceable(1024) %ptr) {
+
+define void @foo(i32 %val, ptr dereferenceable(1024) %ptr, i64 %n) {
; CHECK-LABEL: @foo(
; CHECK-NEXT: entry:
+; CHECK-NEXT: [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N:%.*]], i64 1)
; CHECK-NEXT: [[TMP6:%.*]] = call i64 @llvm.vscale.i64()
; CHECK-NEXT: [[TMP7:%.*]] = shl nuw i64 [[TMP6]], 2
-; CHECK-NEXT: [[TMP8:%.*]] = icmp ult i64 -257, [[TMP7]]
+; CHECK-NEXT: [[TMP3:%.*]] = sub i64 -1, [[UMAX]]
+; CHECK-NEXT: [[TMP8:%.*]] = icmp ult i64 [[TMP3]], [[TMP7]]
; CHECK-NEXT: br i1 [[TMP8]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
; CHECK: vector.ph:
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
; CHECK-NEXT: [[TMP2:%.*]] = sub i64 [[TMP1]], 1
-; CHECK-NEXT: [[N_RND_UP:%.*]] = add i64 256, [[TMP2]]
+; CHECK-NEXT: [[N_RND_UP:%.*]] = add i64 [[UMAX]], [[TMP2]]
; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N_RND_UP]], [[TMP1]]
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N_RND_UP]], [[N_MOD_VF]]
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
@@ -34,7 +37,7 @@ define void @foo(i32 %val, ptr dereferenceable(1024) %ptr) {
; CHECK-NEXT: [[GEP:%.*]] = getelementptr i32, ptr [[PTR:%.*]], i64 [[INDEX]]
; CHECK-NEXT: [[LD1:%.*]] = load i32, ptr [[GEP]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nsw i64 [[INDEX]], 1
-; CHECK-NEXT: [[CMP10:%.*]] = icmp ult i64 [[INDEX_NEXT]], 256
+; CHECK-NEXT: [[CMP10:%.*]] = icmp ult i64 [[INDEX_NEXT]], [[N]]
; CHECK-NEXT: br i1 [[CMP10]], label [[WHILE_BODY]], label [[WHILE_END_LOOPEXIT]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: while.end.loopexit:
; CHECK-NEXT: ret void
@@ -47,7 +50,7 @@ while.body: ; preds = %while.body, %entry
%gep = getelementptr i32, ptr %ptr, i64 %index
%ld1 = load i32, ptr %gep, align 4
%index.next = add nsw i64 %index, 1
- %cmp10 = icmp ult i64 %index.next, 256
+ %cmp10 = icmp ult i64 %index.next, %n
br i1 %cmp10, label %while.body, label %while.end.loopexit, !llvm.loop !0
while.end.loopexit: ; preds = %while.body
>From b8a9084c5c8e425f7c566714f431a43c3dd4a522 Mon Sep 17 00:00:00 2001
From: Jim Lin <jim at andestech.com>
Date: Mon, 26 Jan 2026 16:52:33 +0800
Subject: [PATCH 2/4] Apply clang-format
---
llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 37e4a9cb9e0e0..b1cb9c44fa90d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1052,8 +1052,8 @@ void VPlanTransforms::addMinimumIterationCheck(
// Don't execute the vector loop if (UMax - n) < (VF * UF).
// FIXME: Should only check VF * UF, but currently checks Step=max(VF*UF,
// minProfitableTripCount).
- TripCountCheck = Builder.createICmp(ICmpInst::ICMP_ULT, DistanceToMax,
- StepV, DL);
+ TripCountCheck =
+ Builder.createICmp(ICmpInst::ICMP_ULT, DistanceToMax, StepV, DL);
} else {
// TripCountCheck = false, folding tail implies positive vector trip
// count.
>From df73ba5e593ff1432212def2ba73de4e23e1cfc5 Mon Sep 17 00:00:00 2001
From: Jim Lin <jim at andestech.com>
Date: Tue, 27 Jan 2026 16:51:41 +0800
Subject: [PATCH 3/4] Add a new function which causes a crash
---
.../LoopVectorize/scalable-predication.ll | 61 +++++++++++++++++--
1 file changed, 55 insertions(+), 6 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/scalable-predication.ll b/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
index 3747792416039..f3bff5aba89ac 100644
--- a/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
+++ b/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
@@ -6,20 +6,18 @@
; *do* have vscale as power-of-two) exercises the code required for the
; minimum iteration check in the non-power-of-two case.
-define void @foo(i32 %val, ptr dereferenceable(1024) %ptr, i64 %n) {
+define void @foo(i32 %val, ptr dereferenceable(1024) %ptr) {
; CHECK-LABEL: @foo(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N:%.*]], i64 1)
; CHECK-NEXT: [[TMP6:%.*]] = call i64 @llvm.vscale.i64()
; CHECK-NEXT: [[TMP7:%.*]] = shl nuw i64 [[TMP6]], 2
-; CHECK-NEXT: [[TMP3:%.*]] = sub i64 -1, [[UMAX]]
-; CHECK-NEXT: [[TMP8:%.*]] = icmp ult i64 [[TMP3]], [[TMP7]]
+; CHECK-NEXT: [[TMP8:%.*]] = icmp ult i64 -257, [[TMP7]]
; CHECK-NEXT: br i1 [[TMP8]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
; CHECK: vector.ph:
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
; CHECK-NEXT: [[TMP2:%.*]] = sub i64 [[TMP1]], 1
-; CHECK-NEXT: [[N_RND_UP:%.*]] = add i64 [[UMAX]], [[TMP2]]
+; CHECK-NEXT: [[N_RND_UP:%.*]] = add i64 256, [[TMP2]]
; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N_RND_UP]], [[TMP1]]
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N_RND_UP]], [[N_MOD_VF]]
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
@@ -37,7 +35,7 @@ define void @foo(i32 %val, ptr dereferenceable(1024) %ptr, i64 %n) {
; CHECK-NEXT: [[GEP:%.*]] = getelementptr i32, ptr [[PTR:%.*]], i64 [[INDEX]]
; CHECK-NEXT: [[LD1:%.*]] = load i32, ptr [[GEP]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nsw i64 [[INDEX]], 1
-; CHECK-NEXT: [[CMP10:%.*]] = icmp ult i64 [[INDEX_NEXT]], [[N]]
+; CHECK-NEXT: [[CMP10:%.*]] = icmp ult i64 [[INDEX_NEXT]], 256
; CHECK-NEXT: br i1 [[CMP10]], label [[WHILE_BODY]], label [[WHILE_END_LOOPEXIT]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: while.end.loopexit:
; CHECK-NEXT: ret void
@@ -45,6 +43,57 @@ define void @foo(i32 %val, ptr dereferenceable(1024) %ptr, i64 %n) {
entry:
br label %while.body
+while.body: ; preds = %while.body, %entry
+ %index = phi i64 [ %index.next, %while.body ], [ 0, %entry ]
+ %gep = getelementptr i32, ptr %ptr, i64 %index
+ %ld1 = load i32, ptr %gep, align 4
+ %index.next = add nsw i64 %index, 1
+ %cmp10 = icmp ult i64 %index.next, 256
+ br i1 %cmp10, label %while.body, label %while.end.loopexit, !llvm.loop !0
+
+while.end.loopexit: ; preds = %while.body
+ ret void
+}
+
+define void @foo2(i32 %val, ptr dereferenceable(1024) %ptr, i64 %n) {
+; CHECK-LABEL: @foo2(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N:%.*]], i64 1)
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-NEXT: [[TMP2:%.*]] = sub i64 -1, [[UMAX]]
+; CHECK-NEXT: [[TMP3:%.*]] = icmp ult i64 [[TMP2]], [[TMP1]]
+; CHECK-NEXT: br i1 [[TMP3]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 2
+; CHECK-NEXT: [[TMP6:%.*]] = sub i64 [[TMP5]], 1
+; CHECK-NEXT: [[N_RND_UP:%.*]] = add i64 [[UMAX]], [[TMP6]]
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N_RND_UP]], [[TMP5]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N_RND_UP]], [[N_MOD_VF]]
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX1:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT2:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT2]] = add i64 [[INDEX1]], [[TMP5]]
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT2]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP7]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: br label [[WHILE_END_LOOPEXIT:%.*]]
+; CHECK: scalar.ph:
+; CHECK-NEXT: br label [[WHILE_BODY:%.*]]
+; CHECK: while.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ [[INDEX_NEXT:%.*]], [[WHILE_BODY]] ], [ 0, [[SCALAR_PH]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr i32, ptr [[PTR:%.*]], i64 [[INDEX]]
+; CHECK-NEXT: [[LD1:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nsw i64 [[INDEX]], 1
+; CHECK-NEXT: [[CMP10:%.*]] = icmp ult i64 [[INDEX_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP10]], label [[WHILE_BODY]], label [[WHILE_END_LOOPEXIT]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: while.end.loopexit:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %while.body
+
while.body: ; preds = %while.body, %entry
%index = phi i64 [ %index.next, %while.body ], [ 0, %entry ]
%gep = getelementptr i32, ptr %ptr, i64 %index
>From 71ef25a9789db03f6e4b0b7eed7a7120db23eb98 Mon Sep 17 00:00:00 2001
From: Jim Lin <jim at andestech.com>
Date: Wed, 28 Jan 2026 09:20:00 +0800
Subject: [PATCH 4/4] Address comment
---
llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp | 4 ++--
llvm/test/Transforms/LoopVectorize/scalable-predication.ll | 1 +
2 files changed, 3 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index b1cb9c44fa90d..bc3fce1191824 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1040,7 +1040,7 @@ void VPlanTransforms::addMinimumIterationCheck(
// an overflow to zero when updating induction variables and so an
// additional overflow check is required before entering the vector loop.
- VPValue *StepV = Builder.createExpandSCEV(Step);
+ VPValue *StepVPV = Builder.createExpandSCEV(Step);
// Get the maximum unsigned value for the type.
VPValue *MaxUIntTripCount =
@@ -1053,7 +1053,7 @@ void VPlanTransforms::addMinimumIterationCheck(
// FIXME: Should only check VF * UF, but currently checks Step=max(VF*UF,
// minProfitableTripCount).
TripCountCheck =
- Builder.createICmp(ICmpInst::ICMP_ULT, DistanceToMax, StepV, DL);
+ Builder.createICmp(ICmpInst::ICMP_ULT, DistanceToMax, StepVPV, DL);
} else {
// TripCountCheck = false, folding tail implies positive vector trip
// count.
diff --git a/llvm/test/Transforms/LoopVectorize/scalable-predication.ll b/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
index f3bff5aba89ac..65d3e7e7cbdf4 100644
--- a/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
+++ b/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
@@ -55,6 +55,7 @@ while.end.loopexit: ; preds = %while.body
ret void
}
+; Same as @foo, but with variable trip count.
define void @foo2(i32 %val, ptr dereferenceable(1024) %ptr, i64 %n) {
; CHECK-LABEL: @foo2(
; CHECK-NEXT: entry:
More information about the llvm-commits
mailing list