[llvm] [LV] Scalarize induction compare live-outs (PR #222521)
Pengcheng Wang via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 23:42:11 PDT 2026
https://github.com/wangpc-pp created https://github.com/llvm/llvm-project/pull/222521
Compute the last scalar value of an affine induction with SCEV and
rebuild integer comparisons in the middle block. This avoids keeping
a vector induction solely for a dynamic last-lane extraction.
Assisted-by: TRAE CLI (GPT-5)
>From 99bb08777e9eb802ac1d478561fa7ccfd0406725 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 10 Sep 2026 11:08:01 +0800
Subject: [PATCH 1/2] [LV] Precommit test for induction compare live-out
Add coverage for an induction-derived comparison that is only used
outside an EVL-tail-folded vector loop. The checks preserve the
redundant vector induction and dynamic last-lane extraction.
Assisted-by: TRAE CLI (GPT-5)
---
.../RISCV/tail-folding-iv-outside-user.ll | 56 +++++++++++++++++++
1 file changed, 56 insertions(+)
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-iv-outside-user.ll b/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-iv-outside-user.ll
index 4e083dd9ffe76..eea8e0091496a 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-iv-outside-user.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-iv-outside-user.ll
@@ -64,3 +64,59 @@ loop:
exit:
ret i32 %iv2
}
+
+; Check an induction-derived compare that is only used after the loop.
+define i1 @induction_compare_live_out(ptr %p, i32 %start, i64 %n) {
+; CHECK-LABEL: define i1 @induction_compare_live_out(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[START:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP6:%.*]] = call <vscale x 4 x i32> @llvm.stepvector.nxv4i32()
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 4 x i32> poison, i32 [[START]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 4 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 4 x i32> poison, <vscale x 4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP7:%.*]] = sub nsw <vscale x 4 x i32> [[BROADCAST_SPLAT]], [[TMP6]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX1:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <vscale x 4 x i32> [ [[TMP7]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[AVL:%.*]] = phi i64 [ [[TMP0]], %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 4, i1 true)
+; CHECK-NEXT: [[TMP3:%.*]] = zext i32 [[TMP1]] to i64
+; CHECK-NEXT: [[TMP5:%.*]] = sub nsw i32 0, [[TMP1]]
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <vscale x 4 x i32> poison, i32 [[TMP5]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT3:%.*]] = shufflevector <vscale x 4 x i32> [[BROADCAST_SPLATINSERT2]], <vscale x 4 x i32> poison, <vscale x 4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 [[INDEX1]]
+; CHECK-NEXT: call void @llvm.vp.store.nxv4i32.p0(<vscale x 4 x i32> zeroinitializer, ptr align 4 [[TMP2]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP1]])
+; CHECK-NEXT: [[CURRENT_ITERATION_NEXT]] = add i64 [[TMP3]], [[INDEX1]]
+; CHECK-NEXT: [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP3]]
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add nsw <vscale x 4 x i32> [[VEC_IND]], [[BROADCAST_SPLAT3]]
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP8:%.*]] = add nsw <vscale x 4 x i32> [[VEC_IND]], splat (i32 -1)
+; CHECK-NEXT: [[TMP9:%.*]] = icmp ne <vscale x 4 x i32> [[TMP8]], zeroinitializer
+; CHECK-NEXT: [[TMP10:%.*]] = sub i64 [[TMP3]], 1
+; CHECK-NEXT: [[RES:%.*]] = extractelement <vscale x 4 x i1> [[TMP9]], i64 [[TMP10]]
+; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret i1 [[RES]]
+;
+entry:
+ br label %loop
+
+loop:
+ %index = phi i64 [ 0, %entry ], [ %index.next, %loop ]
+ %iv = phi i32 [ %start, %entry ], [ %iv.next, %loop ]
+ %iv.next = add nsw i32 %iv, -1
+ %out = icmp ne i32 %iv.next, 0
+ %gep = getelementptr inbounds i32, ptr %p, i64 %index
+ store i32 0, ptr %gep, align 4
+ %index.next = add nuw i64 %index, 1
+ %cmp = icmp ult i64 %index.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret i1 %out
+}
>From 6e0836a4c0997d1ff26d461c2e7c648368db8207 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 10 Sep 2026 11:08:31 +0800
Subject: [PATCH 2/2] [LV] Scalarize induction compare live-outs
Compute the last scalar value of an affine induction with SCEV and
rebuild integer comparisons in the middle block. This avoids keeping
a vector induction solely for a dynamic last-lane extraction.
Assisted-by: TRAE CLI (GPT-5)
---
.../Transforms/Vectorize/VPlanTransforms.cpp | 83 ++++++++++++++++---
.../LoopVectorize/RISCV/pointer-induction.ll | 19 ++---
.../RISCV/tail-folding-iv-outside-user.ll | 25 ++----
3 files changed, 85 insertions(+), 42 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 25a78bc13644f..20f81c2797a54 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -1056,20 +1056,13 @@ optimizeLatchExitInductionUser(VPlan &Plan, VPValue *Op,
return nullptr;
}
-static VPValue *optimizeLatchExitIVUserViaSCEV(VPlan &Plan, VPValue *Op,
- PredicatedScalarEvolution &PSE,
- VPValue *ResumeTC,
- const Loop *L) {
- VPValue *Incoming;
- if (!match(Op, m_CombineOr(m_ExtractLastLaneOfLastPart(m_VPValue(Incoming)),
- m_ExtractLane(m_LastActiveLane(m_HeaderMask()),
- m_VPValue(Incoming)))))
- return nullptr;
-
- const SCEV *IncomingSCEV = vputils::getSCEVExprForVPValue(Incoming, PSE, L);
+static VPValue *computeLoopExitValueForSCEV(VPlan &Plan, VPValue *Op,
+ PredicatedScalarEvolution &PSE,
+ VPValue *ResumeTC, const Loop *L,
+ const SCEV *S) {
const SCEV *Start, *Step;
- if (!match(IncomingSCEV, m_scev_AffineAddRec(m_SCEV(Start), m_SCEV(Step),
- m_SpecificLoop(L))))
+ if (!match(S, m_scev_AffineAddRec(m_SCEV(Start), m_SCEV(Step),
+ m_SpecificLoop(L))))
return nullptr;
auto *ExtractR = cast<VPInstruction>(Op);
@@ -1092,6 +1085,67 @@ static VPValue *optimizeLatchExitIVUserViaSCEV(VPlan &Plan, VPValue *Op,
StepVPV);
}
+static VPValue *optimizeLatchExitIVUserViaSCEV(VPlan &Plan, VPValue *Op,
+ PredicatedScalarEvolution &PSE,
+ VPValue *ResumeTC,
+ const Loop *L) {
+ VPValue *Incoming;
+ if (!match(Op, m_CombineOr(m_ExtractLastLaneOfLastPart(m_VPValue(Incoming)),
+ m_ExtractLane(m_LastActiveLane(m_HeaderMask()),
+ m_VPValue(Incoming)))))
+ return nullptr;
+
+ const SCEV *IncomingSCEV = vputils::getSCEVExprForVPValue(Incoming, PSE, L);
+ return computeLoopExitValueForSCEV(Plan, Op, PSE, ResumeTC, L, IncomingSCEV);
+}
+
+/// Try to scalarize an integer comparison extracted from the last active lane
+/// by computing the scalar value of an induction-derived operand.
+static VPValue *
+optimizeLatchExitInductionCompare(VPlan &Plan, VPValue *Op,
+ PredicatedScalarEvolution &PSE,
+ VPValue *ResumeTC, const Loop *L) {
+ VPValue *Incoming;
+ if (!match(Op, m_CombineOr(m_ExtractLastLaneOfLastPart(m_VPValue(Incoming)),
+ m_ExtractLane(m_LastActiveLane(m_HeaderMask()),
+ m_VPValue(Incoming)))))
+ return nullptr;
+
+ CmpPredicate Pred;
+ VPValue *LHS, *RHS;
+ if (!match(Incoming, m_ICmp(Pred, m_VPValue(LHS), m_VPValue(RHS))))
+ return nullptr;
+
+ const SCEV *LHSSCEV = vputils::getSCEVExprForVPValue(LHS, PSE, L);
+ const SCEV *RHSSCEV = vputils::getSCEVExprForVPValue(RHS, PSE, L);
+ bool LHSIsAddRec = isa<SCEVAddRecExpr>(LHSSCEV);
+ bool RHSIsAddRec = isa<SCEVAddRecExpr>(RHSSCEV);
+ if (LHSIsAddRec == RHSIsAddRec ||
+ isa<SCEVCouldNotCompute>(LHSIsAddRec ? RHSSCEV : LHSSCEV))
+ return nullptr;
+
+ const SCEV *AddRecSCEV = LHSIsAddRec ? LHSSCEV : RHSSCEV;
+ VPValue *Invariant = LHSIsAddRec ? RHS : LHS;
+ const SCEV *InvariantSCEV = LHSIsAddRec ? RHSSCEV : LHSSCEV;
+ // The invariant operand is reused in the middle block, so restrict it to a
+ // live-in that is known to be loop-invariant.
+ if (!isa<VPIRValue>(Invariant) ||
+ !PSE.getSE()->isLoopInvariant(InvariantSCEV, L))
+ return nullptr;
+
+ VPValue *ScalarAddRec =
+ computeLoopExitValueForSCEV(Plan, Op, PSE, ResumeTC, L, AddRecSCEV);
+ if (!ScalarAddRec)
+ return nullptr;
+
+ auto *ExtractR = cast<VPInstruction>(Op);
+ VPBuilder Builder(ExtractR);
+ if (!LHSIsAddRec)
+ std::swap(ScalarAddRec, Invariant);
+ return Builder.createICmp(Pred, ScalarAddRec, Invariant,
+ ExtractR->getDebugLoc());
+}
+
void VPlanTransforms::optimizeInductionLiveOutUsers(
VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L) {
// Compute end values for all inductions.
@@ -1135,6 +1189,9 @@ void VPlanTransforms::optimizeInductionLiveOutUsers(
if (!Escape)
Escape = optimizeLatchExitIVUserViaSCEV(
Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
+ if (!Escape)
+ Escape = optimizeLatchExitInductionCompare(
+ Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
} else {
Escape = optimizeEarlyExitInductionUser(
Plan, ExitIRI->getOperand(Idx), PSE);
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll b/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll
index 667612a193dd5..251d2b373865d 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll
@@ -74,18 +74,11 @@ define i1 @scalarize_ptr_induction(ptr %start, ptr %end, ptr noalias %dst, i1 %c
; CHECK-NEXT: [[SCEVGEP6:%.*]] = getelementptr i8, ptr [[START]], i64 4
; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x ptr> poison, ptr [[DST]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x ptr> [[BROADCAST_SPLATINSERT]], <vscale x 2 x ptr> poison, <vscale x 2 x i32> zeroinitializer
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT6:%.*]] = insertelement <vscale x 2 x ptr> poison, ptr [[END]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT7:%.*]] = shufflevector <vscale x 2 x ptr> [[BROADCAST_SPLATINSERT6]], <vscale x 2 x ptr> poison, <vscale x 2 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_MEMCHECK]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[POINTER_PHI:%.*]] = phi ptr [ [[START]], %[[VECTOR_MEMCHECK]] ], [ [[PTR_IND:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[AVL:%.*]] = phi i64 [ [[TMP3]], %[[VECTOR_MEMCHECK]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP13:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
-; CHECK-NEXT: [[TMP14:%.*]] = mul <vscale x 2 x i64> [[TMP13]], splat (i64 12)
-; CHECK-NEXT: [[VECTOR_GEP:%.*]] = getelementptr i8, ptr [[POINTER_PHI]], <vscale x 2 x i64> [[TMP14]]
; CHECK-NEXT: [[TMP11:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 2, i1 true)
-; CHECK-NEXT: [[TMP26:%.*]] = zext i32 [[TMP11]] to i64
; CHECK-NEXT: [[TMP12:%.*]] = mul i64 [[INDEX]], 12
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[SCEVGEP6]], i64 [[TMP12]]
; CHECK-NEXT: [[TMP18:%.*]] = call <vscale x 2 x i32> @llvm.experimental.vp.strided.load.nxv2i32.p0.i64(ptr align 4 [[TMP15]], i64 12, <vscale x 2 x i1> splat (i1 true), i32 [[TMP11]])
@@ -93,17 +86,17 @@ define i1 @scalarize_ptr_induction(ptr %start, ptr %end, ptr noalias %dst, i1 %c
; CHECK-NEXT: [[TMP20:%.*]] = mul <vscale x 2 x i64> [[TMP19]], splat (i64 -7070675565921424023)
; CHECK-NEXT: [[TMP21:%.*]] = add <vscale x 2 x i64> [[TMP20]], splat (i64 -4)
; CHECK-NEXT: call void @llvm.vp.scatter.nxv2i64.nxv2p0(<vscale x 2 x i64> [[TMP21]], <vscale x 2 x ptr> align 1 [[BROADCAST_SPLAT]], <vscale x 2 x i1> splat (i1 true), i32 [[TMP11]])
+; CHECK-NEXT: [[TMP26:%.*]] = zext i32 [[TMP11]] to i64
; CHECK-NEXT: [[CURRENT_ITERATION_NEXT]] = add i64 [[TMP26]], [[INDEX]]
; CHECK-NEXT: [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP26]]
-; CHECK-NEXT: [[TMP27:%.*]] = mul i64 12, [[TMP26]]
-; CHECK-NEXT: [[PTR_IND]] = getelementptr i8, ptr [[POINTER_PHI]], i64 [[TMP27]]
; CHECK-NEXT: [[TMP28:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
; CHECK-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: [[TMP30:%.*]] = getelementptr nusw i8, <vscale x 2 x ptr> [[VECTOR_GEP]], i64 12
-; CHECK-NEXT: [[TMP17:%.*]] = icmp eq <vscale x 2 x ptr> [[TMP30]], [[BROADCAST_SPLAT7]]
-; CHECK-NEXT: [[TMP29:%.*]] = sub i64 [[TMP26]], 1
-; CHECK-NEXT: [[TMP25:%.*]] = extractelement <vscale x 2 x i1> [[TMP17]], i64 [[TMP29]]
+; CHECK-NEXT: [[TMP16:%.*]] = getelementptr nuw i8, ptr [[START]], i64 12
+; CHECK-NEXT: [[TMP17:%.*]] = sub nuw i64 [[TMP3]], 1
+; CHECK-NEXT: [[TMP22:%.*]] = mul i64 [[TMP17]], 12
+; CHECK-NEXT: [[TMP23:%.*]] = getelementptr i8, ptr [[TMP16]], i64 [[TMP22]]
+; CHECK-NEXT: [[TMP25:%.*]] = icmp eq ptr [[TMP23]], [[END]]
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: ret i1 [[TMP25]]
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-iv-outside-user.ll b/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-iv-outside-user.ll
index eea8e0091496a..feac6a8b95a6d 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-iv-outside-user.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-iv-outside-user.ll
@@ -65,40 +65,33 @@ exit:
ret i32 %iv2
}
-; Check an induction-derived compare that is only used after the loop.
+; Make sure an induction-derived compare that is only used after the loop does
+; not require a vector induction just to extract its last active lane.
define i1 @induction_compare_live_out(ptr %p, i32 %start, i64 %n) {
; CHECK-LABEL: define i1 @induction_compare_live_out(
-; CHECK-SAME: ptr [[P:%.*]], i32 [[START:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-SAME: ptr [[P:%.*]], i32 [[START:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP6:%.*]] = call <vscale x 4 x i32> @llvm.stepvector.nxv4i32()
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 4 x i32> poison, i32 [[START]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 4 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 4 x i32> poison, <vscale x 4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP7:%.*]] = sub nsw <vscale x 4 x i32> [[BROADCAST_SPLAT]], [[TMP6]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX1:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <vscale x 4 x i32> [ [[TMP7]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[AVL:%.*]] = phi i64 [ [[TMP0]], %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP1:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 4, i1 true)
-; CHECK-NEXT: [[TMP3:%.*]] = zext i32 [[TMP1]] to i64
-; CHECK-NEXT: [[TMP5:%.*]] = sub nsw i32 0, [[TMP1]]
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <vscale x 4 x i32> poison, i32 [[TMP5]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT3:%.*]] = shufflevector <vscale x 4 x i32> [[BROADCAST_SPLATINSERT2]], <vscale x 4 x i32> poison, <vscale x 4 x i32> zeroinitializer
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 [[INDEX1]]
; CHECK-NEXT: call void @llvm.vp.store.nxv4i32.p0(<vscale x 4 x i32> zeroinitializer, ptr align 4 [[TMP2]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP1]])
+; CHECK-NEXT: [[TMP3:%.*]] = zext i32 [[TMP1]] to i64
; CHECK-NEXT: [[CURRENT_ITERATION_NEXT]] = add i64 [[TMP3]], [[INDEX1]]
; CHECK-NEXT: [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP3]]
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add nsw <vscale x 4 x i32> [[VEC_IND]], [[BROADCAST_SPLAT3]]
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: [[TMP8:%.*]] = add nsw <vscale x 4 x i32> [[VEC_IND]], splat (i32 -1)
-; CHECK-NEXT: [[TMP9:%.*]] = icmp ne <vscale x 4 x i32> [[TMP8]], zeroinitializer
-; CHECK-NEXT: [[TMP10:%.*]] = sub i64 [[TMP3]], 1
-; CHECK-NEXT: [[RES:%.*]] = extractelement <vscale x 4 x i1> [[TMP9]], i64 [[TMP10]]
+; CHECK-NEXT: [[TMP5:%.*]] = add i32 [[START]], -1
+; CHECK-NEXT: [[TMP6:%.*]] = sub nuw i64 [[TMP0]], 1
+; CHECK-NEXT: [[TMP7:%.*]] = trunc i64 [[TMP6]] to i32
+; CHECK-NEXT: [[TMP8:%.*]] = sub i32 [[TMP5]], [[TMP7]]
+; CHECK-NEXT: [[RES:%.*]] = icmp ne i32 [[TMP8]], 0
; CHECK-NEXT: br label %[[EXIT:.*]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret i1 [[RES]]
More information about the llvm-commits
mailing list