[llvm] 0ed8e72 - [VPlan] Create SCEV before any VPIRInstructions to check for overflow (#177911)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Jan 27 19:16:55 PST 2026
Author: Jim Lin
Date: 2026-01-28T03:16:50Z
New Revision: 0ed8e7230f82ab0da92730200268ccada64e8484
URL: https://github.com/llvm/llvm-project/commit/0ed8e7230f82ab0da92730200268ccada64e8484
DIFF: https://github.com/llvm/llvm-project/commit/0ed8e7230f82ab0da92730200268ccada64e8484.diff
LOG: [VPlan] Create SCEV before any VPIRInstructions to check for overflow (#177911)
This PR tried to fix the assertion fail at VPlanTransforms.cpp:4862
since SCEV was created after VPIRInstructions.
The tripcount in scalable-predication.ll was changed from constant value
256 to non-constant value %n to avoid VPIRInstructions optimized out,
which cannot trigger the assertion fail.
The orders in ir-bb<entry> from:
ir-bb<entry>:
EMIT vp<%2> = EXPAND SCEV (1 umax %n)
EMIT vp<%3> = sub ir<-1>, vp<%2>
EMIT vp<%4> = EXPAND SCEV (4 * vscale)<nuw>
EMIT vp<%5> = icmp ult vp<%3>, vp<%4>
EMIT branch-on-cond vp<%5>
Successor(s): scalar.ph, vector.ph
to:
ir-bb<entry>:
EMIT vp<%2> = EXPAND SCEV (1 umax %n)
EMIT vp<%3> = EXPAND SCEV (4 * vscale)<nuw>
EMIT vp<%4> = sub ir<-1>, vp<%2>
EMIT vp<%5> = icmp ult vp<%4>, vp<%3>
EMIT branch-on-cond vp<%5>
Successor(s): scalar.ph, vector.ph
Added:
Modified:
llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
llvm/test/Transforms/LoopVectorize/scalable-predication.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 4fc7b2fe1c2de..4846abdab3c55 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1040,6 +1040,8 @@ void VPlanTransforms::addMinimumIterationCheck(
// an overflow to zero when updating induction variables and so an
// additional overflow check is required before entering the vector loop.
+ VPValue *StepVPV = Builder.createExpandSCEV(Step);
+
// Get the maximum unsigned value for the type.
VPValue *MaxUIntTripCount =
Plan.getConstantInt(cast<IntegerType>(TripCountTy)->getMask());
@@ -1050,8 +1052,8 @@ void VPlanTransforms::addMinimumIterationCheck(
// Don't execute the vector loop if (UMax - n) < (VF * UF).
// FIXME: Should only check VF * UF, but currently checks Step=max(VF*UF,
// minProfitableTripCount).
- TripCountCheck = Builder.createICmp(ICmpInst::ICMP_ULT, DistanceToMax,
- Builder.createExpandSCEV(Step), DL);
+ TripCountCheck =
+ Builder.createICmp(ICmpInst::ICMP_ULT, DistanceToMax, StepVPV, DL);
} else {
// TripCountCheck = false, folding tail implies positive vector trip
// count.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 90e32b9e1ce1d..c09e0d4e1ad94 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4861,7 +4861,7 @@ VPlanTransforms::expandSCEVs(VPlan &Plan, ScalarEvolution &SE) {
}
assert(none_of(*Entry, IsaPred<VPExpandSCEVRecipe>) &&
"VPExpandSCEVRecipes must be at the beginning of the entry block, "
- "after any VPIRInstructions");
+ "before any VPIRInstructions");
// Add IR instructions in the entry basic block but not in the VPIRBasicBlock
// to the VPIRBasicBlock.
auto EI = Entry->begin();
diff --git a/llvm/test/Transforms/LoopVectorize/scalable-predication.ll b/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
index f5b445641097c..65d3e7e7cbdf4 100644
--- a/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
+++ b/llvm/test/Transforms/LoopVectorize/scalable-predication.ll
@@ -5,6 +5,7 @@
; deliberately doesn't correspond to an in-tree backend since those
; *do* have vscale as power-of-two) exercises the code required for the
; minimum iteration check in the non-power-of-two case.
+
define void @foo(i32 %val, ptr dereferenceable(1024) %ptr) {
; CHECK-LABEL: @foo(
; CHECK-NEXT: entry:
@@ -54,6 +55,58 @@ while.end.loopexit: ; preds = %while.body
ret void
}
+; Same as @foo, but with variable trip count.
+define void @foo2(i32 %val, ptr dereferenceable(1024) %ptr, i64 %n) {
+; CHECK-LABEL: @foo2(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N:%.*]], i64 1)
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-NEXT: [[TMP2:%.*]] = sub i64 -1, [[UMAX]]
+; CHECK-NEXT: [[TMP3:%.*]] = icmp ult i64 [[TMP2]], [[TMP1]]
+; CHECK-NEXT: br i1 [[TMP3]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 2
+; CHECK-NEXT: [[TMP6:%.*]] = sub i64 [[TMP5]], 1
+; CHECK-NEXT: [[N_RND_UP:%.*]] = add i64 [[UMAX]], [[TMP6]]
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N_RND_UP]], [[TMP5]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N_RND_UP]], [[N_MOD_VF]]
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX1:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT2:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT2]] = add i64 [[INDEX1]], [[TMP5]]
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT2]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP7]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: br label [[WHILE_END_LOOPEXIT:%.*]]
+; CHECK: scalar.ph:
+; CHECK-NEXT: br label [[WHILE_BODY:%.*]]
+; CHECK: while.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ [[INDEX_NEXT:%.*]], [[WHILE_BODY]] ], [ 0, [[SCALAR_PH]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr i32, ptr [[PTR:%.*]], i64 [[INDEX]]
+; CHECK-NEXT: [[LD1:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nsw i64 [[INDEX]], 1
+; CHECK-NEXT: [[CMP10:%.*]] = icmp ult i64 [[INDEX_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP10]], label [[WHILE_BODY]], label [[WHILE_END_LOOPEXIT]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: while.end.loopexit:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %while.body
+
+while.body: ; preds = %while.body, %entry
+ %index = phi i64 [ %index.next, %while.body ], [ 0, %entry ]
+ %gep = getelementptr i32, ptr %ptr, i64 %index
+ %ld1 = load i32, ptr %gep, align 4
+ %index.next = add nsw i64 %index, 1
+ %cmp10 = icmp ult i64 %index.next, %n
+ br i1 %cmp10, label %while.body, label %while.end.loopexit, !llvm.loop !0
+
+while.end.loopexit: ; preds = %while.body
+ ret void
+}
+
!0 = distinct !{!0, !1, !2, !3, !4}
!1 = !{!"llvm.loop.vectorize.predicate.enable", i1 true}
!2 = !{!"llvm.loop.vectorize.scalable.enable", i1 true}
More information about the llvm-commits
mailing list