[llvm] [LV] Use SCEV to compute final value of complex induction variables (PR #195059)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Apr 30 04:02:26 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-vectorizers
Author: Mel Chen (Mel-Chen)
<details>
<summary>Changes</summary>
Extend optimizeLatchExitInductionUser to handle complex induction variable by using SCEV analysis. When an induction variable forms an affine AddRec {Start,+,Step}, compute the final value as Start + (ResumeTripCount - 1) * Step using VPDerivedIVRecipe.
This patch eliminates unnecessary vector widening for induction variables only used outside the loop.
Pre-commit test #<!-- -->19505
---
Patch is 28.76 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/195059.diff
11 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorize.cpp (+2-2)
- (modified) llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp (+68-37)
- (modified) llvm/lib/Transforms/Vectorize/VPlanTransforms.h (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll (-3)
- (modified) llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing.ll (+3-5)
- (modified) llvm/test/Transforms/LoopVectorize/X86/gep-use-outside-loop.ll (+1-1)
- (modified) llvm/test/Transforms/LoopVectorize/X86/pr51366-sunk-instruction-used-outside-of-loop.ll (+6-6)
- (added) llvm/test/Transforms/LoopVectorize/cast-iv-outside-user.ll (+64)
- (modified) llvm/test/Transforms/LoopVectorize/instruction-only-used-outside-of-loop.ll (+1-4)
- (modified) llvm/test/Transforms/LoopVectorize/iv_outside_user.ll (+31-65)
- (modified) llvm/test/Transforms/LoopVectorize/no_outside_user.ll (+1-4)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 6c9298a2cf98d..52747991ad56a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7097,7 +7097,7 @@ LoopVectorizationPlanner::tryToBuildVPlanWithVPRecipes(VPlanPtr Plan,
RUN_VPLAN_PASS(VPlanTransforms::optimizeFindIVReductions, *Plan, PSE,
*OrigLoop);
RUN_VPLAN_PASS(VPlanTransforms::optimizeInductionLiveOutUsers, *Plan, PSE,
- CM.foldTailByMasking());
+ OrigLoop, CM.foldTailByMasking());
// Apply mandatory transformation to handle reductions with multiple in-loop
// uses if possible, bail out otherwise.
@@ -7192,7 +7192,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VFRange &Range) {
return nullptr;
// Optimize induction live-out users to use precomputed end values.
- VPlanTransforms::optimizeInductionLiveOutUsers(*Plan, PSE,
+ VPlanTransforms::optimizeInductionLiveOutUsers(*Plan, PSE, OrigLoop,
/*FoldTail=*/false);
assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 3b3b01973225e..1e70a1e9c86f2 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -1119,50 +1119,81 @@ static VPValue *tryToComputeEndValueForInduction(VPWidenInductionRecipe *WideIV,
/// exit block coming from the latch in the original scalar loop.
static VPValue *optimizeLatchExitInductionUser(
VPlan &Plan, VPTypeAnalysis &TypeInfo, VPBlockBase *PredVPBB, VPValue *Op,
- DenseMap<VPValue *, VPValue *> &EndValues, PredicatedScalarEvolution &PSE) {
+ DenseMap<VPValue *, VPValue *> &EndValues, PredicatedScalarEvolution &PSE,
+ VPValue *ResumeTC, const Loop *L) {
VPValue *Incoming;
- VPWidenInductionRecipe *WideIV = nullptr;
- if (match(Op, m_ExtractLastLaneOfLastPart(m_VPValue(Incoming))))
- WideIV = getOptimizableIVOf(Incoming, PSE);
-
- if (!WideIV)
+ if (!match(Op, m_ExtractLastLaneOfLastPart(m_VPValue(Incoming))))
return nullptr;
- VPValue *EndValue = EndValues.lookup(WideIV);
- assert(EndValue && "Must have computed the end value up front");
+ if (VPWidenInductionRecipe *WideIV = getOptimizableIVOf(Incoming, PSE)) {
+ VPValue *EndValue = EndValues.lookup(WideIV);
+ assert(EndValue && "Must have computed the end value up front");
- // `getOptimizableIVOf()` always returns the pre-incremented IV, so if it
- // changed it means the exit is using the incremented value, so we don't
- // need to subtract the step.
- if (Incoming != WideIV)
- return EndValue;
+ // `getOptimizableIVOf()` always returns the pre-incremented IV, so if it
+ // changed it means the exit is using the incremented value, so we don't
+ // need to subtract the step.
+ if (Incoming != WideIV)
+ return EndValue;
- // Otherwise, subtract the step from the EndValue.
- VPBuilder B(cast<VPBasicBlock>(PredVPBB)->getTerminator());
- VPValue *Step = WideIV->getStepValue();
- Type *ScalarTy = TypeInfo.inferScalarType(WideIV);
- if (ScalarTy->isIntegerTy())
- return B.createSub(EndValue, Step, DebugLoc::getUnknown(), "ind.escape");
- if (ScalarTy->isPointerTy()) {
- Type *StepTy = TypeInfo.inferScalarType(Step);
- auto *Zero = Plan.getZero(StepTy);
- return B.createPtrAdd(EndValue, B.createSub(Zero, Step),
- DebugLoc::getUnknown(), "ind.escape");
- }
- if (ScalarTy->isFloatingPointTy()) {
- const auto &ID = WideIV->getInductionDescriptor();
- return B.createNaryOp(
- ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
- ? Instruction::FSub
- : Instruction::FAdd,
- {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
- }
- llvm_unreachable("all possible induction types must be handled");
- return nullptr;
+ // Otherwise, subtract the step from the EndValue.
+ VPBuilder B(cast<VPBasicBlock>(PredVPBB)->getTerminator());
+ VPValue *Step = WideIV->getStepValue();
+ Type *ScalarTy = TypeInfo.inferScalarType(WideIV);
+ if (ScalarTy->isIntegerTy())
+ return B.createSub(EndValue, Step, DebugLoc::getUnknown(), "ind.escape");
+ if (ScalarTy->isPointerTy()) {
+ Type *StepTy = TypeInfo.inferScalarType(Step);
+ auto *Zero = Plan.getZero(StepTy);
+ return B.createPtrAdd(EndValue, B.createSub(Zero, Step),
+ DebugLoc::getUnknown(), "ind.escape");
+ }
+ if (ScalarTy->isFloatingPointTy()) {
+ const auto &ID = WideIV->getInductionDescriptor();
+ return B.createNaryOp(
+ ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
+ ? Instruction::FSub
+ : Instruction::FAdd,
+ {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
+ }
+ llvm_unreachable("all possible induction types must be handled");
+ }
+
+ const SCEV *IncomingSCEV = vputils::getSCEVExprForVPValue(Incoming, PSE, L);
+ const SCEV *Start, *Step;
+ if (!match(IncomingSCEV, m_scev_AffineAddRec(m_SCEV(Start), m_SCEV(Step),
+ m_SpecificLoop(L))))
+ return nullptr;
+
+ VPValue *StartVPV = vputils::getOrCreateVPValueForSCEVExpr(Plan, Start);
+ auto *StartIRV = dyn_cast<VPIRValue>(StartVPV);
+ if (!StartIRV) {
+ VPRecipeBase *Def = StartVPV->getDefiningRecipe();
+ assert(Def && "The value must be defined by VPExpandSCEVRecipe");
+ assert(StartVPV->getNumUsers() == 0 &&
+ "Newly created VPExpandSCEVRecipe should have no users");
+ Def->eraseFromParent();
+ return nullptr;
+ }
+
+ Type *StartTy = StartIRV->getType();
+ assert(StartTy->isIntOrPtrTy() && "The type must be SCEVable");
+ InductionDescriptor::InductionKind Kind =
+ StartTy->isPointerTy() ? InductionDescriptor::IK_PtrInduction
+ : InductionDescriptor::IK_IntInduction;
+ VPValue *StepVPV = vputils::getOrCreateVPValueForSCEVExpr(Plan, Step);
+ VPBuilder Builder(cast<VPInstruction>(Op));
+ Type *TCTy = TypeInfo.inferScalarType(ResumeTC);
+ VPValue *It = Builder.createSub(ResumeTC, Plan.getConstantInt(TCTy, 1),
+ DebugLoc::getUnknown());
+ Type *StepTy = TypeInfo.inferScalarType(StepVPV);
+ It =
+ Builder.createScalarZExtOrTrunc(It, StepTy, TCTy, DebugLoc::getUnknown());
+ return Builder.createDerivedIV(Kind, /*FPBinOp=*/nullptr, StartIRV, It,
+ StepVPV);
}
void VPlanTransforms::optimizeInductionLiveOutUsers(
- VPlan &Plan, PredicatedScalarEvolution &PSE, bool FoldTail) {
+ VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L, bool FoldTail) {
// Compute end values for all inductions.
VPTypeAnalysis TypeInfo(Plan);
VPRegionBlock *VectorRegion = Plan.getVectorLoopRegion();
@@ -1202,7 +1233,7 @@ void VPlanTransforms::optimizeInductionLiveOutUsers(
if (PredVPBB == MiddleVPBB)
Escape = optimizeLatchExitInductionUser(Plan, TypeInfo, PredVPBB,
ExitIRI->getOperand(Idx),
- EndValues, PSE);
+ EndValues, PSE, ResumeTC, L);
else
Escape = optimizeEarlyExitInductionUser(
Plan, TypeInfo, PredVPBB, ExitIRI->getOperand(Idx), PSE);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 3d7a67635ce7e..981f4ab9ca274 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -389,7 +389,7 @@ struct VPlanTransforms {
/// one step backwards.
static void optimizeInductionLiveOutUsers(VPlan &Plan,
PredicatedScalarEvolution &PSE,
- bool FoldTail);
+ const Loop *L, bool FoldTail);
/// Add explicit broadcasts for live-ins and VPValues defined in \p Plan's entry block if they are used as vectors.
static void materializeBroadcasts(VPlan &Plan);
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll
index e5886d83c0182..fe8735755cbc6 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll
@@ -10,14 +10,11 @@ define i64 @reduction(i64 %arg) #0 {
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP1:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[STEP_ADD:%.*]] = add nuw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP5]] = or <4 x i32> [[VEC_PHI]], splat (i32 1)
; CHECK-NEXT: [[TMP1]] = or <4 x i32> [[VEC_PHI2]], splat (i32 1)
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add nsw <4 x i64> [[STEP_ADD]], splat (i64 4)
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 96
; CHECK-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing.ll
index d9eaab8d9a000..06d4806206190 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-printing.ll
@@ -54,7 +54,7 @@ define void @print_call_and_memory(i64 %n, ptr noalias %y, ptr noalias %x) {
; CHECK-NEXT: IR %iv = phi i64 [ %iv.next, %for.body ], [ 0, %for.body.preheader ] (extra operand: vp<%bc.resume.val> from scalar.ph)
; CHECK-NEXT: IR %arrayidx = getelementptr inbounds float, ptr %y, i64 %iv
; CHECK-NEXT: IR %lv = load float, ptr %arrayidx, align 4
-; CHECK-NEXT: IR %call = tail call float @llvm.sqrt.f32(float %lv)
+; CHECK-NEXT: IR %call = tail call float @llvm.sqrt.f32(float %lv) #2
; CHECK-NEXT: IR %arrayidx2 = getelementptr inbounds float, ptr %x, i64 %iv
; CHECK-NEXT: IR store float %call, ptr %arrayidx2, align 4
; CHECK-NEXT: IR %iv.next = add i64 %iv, 1
@@ -569,7 +569,6 @@ define i32 @print_exit_value(ptr %ptr, i32 %off) {
; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
; CHECK-EMPTY:
; CHECK-NEXT: vector.body:
-; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION nsw ir<0>, ir<1>, vp<[[VP0]]>
; CHECK-NEXT: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
; CHECK-NEXT: CLONE ir<%gep> = getelementptr inbounds ir<%ptr>, vp<[[VP4]]>
; CHECK-NEXT: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds ir<%gep>
@@ -581,9 +580,8 @@ define i32 @print_exit_value(ptr %ptr, i32 %off) {
; CHECK-NEXT: Successor(s): middle.block
; CHECK-EMPTY:
; CHECK-NEXT: middle.block:
-; CHECK-NEXT: WIDEN ir<%add> = add ir<%iv>, ir<%off>
-; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = extract-last-part ir<%add>
-; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = extract-last-lane vp<[[VP7]]>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = sub vp<[[VP2]]>, ir<1>
+; CHECK-NEXT: vp<[[VP8:%[0-9]+]]> = DERIVED-IV ir<%off> + vp<[[VP7]]> * ir<1>
; CHECK-NEXT: EMIT vp<%cmp.n> = icmp eq ir<1000>, vp<[[VP2]]>
; CHECK-NEXT: EMIT branch-on-cond vp<%cmp.n>
; CHECK-NEXT: Successor(s): ir-bb<exit>, scalar.ph
diff --git a/llvm/test/Transforms/LoopVectorize/X86/gep-use-outside-loop.ll b/llvm/test/Transforms/LoopVectorize/X86/gep-use-outside-loop.ll
index 8fdd60f2dc1b1..c3ea2a3f4fd76 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/gep-use-outside-loop.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/gep-use-outside-loop.ll
@@ -82,10 +82,10 @@ define void @gep_use_outside_loop(ptr noalias %dst, ptr %src) {
; CHECK-NEXT: [[TMP0:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr i16, ptr [[DST]], <4 x i64> [[VEC_IND]]
-; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x ptr> [[TMP1]], i64 0
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr i16, ptr [[SRC]], i64 [[TMP0]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP2]], align 2
; CHECK-NEXT: [[TMP5:%.*]] = icmp ne <4 x i16> [[WIDE_LOAD]], splat (i16 10)
+; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x ptr> [[TMP1]], i64 0
; CHECK-NEXT: call void @llvm.masked.store.v4i16.p0(<4 x i16> zeroinitializer, ptr align 2 [[TMP6]], <4 x i1> [[TMP5]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[TMP0]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i64> [[VEC_IND]], splat (i64 4)
diff --git a/llvm/test/Transforms/LoopVectorize/X86/pr51366-sunk-instruction-used-outside-of-loop.ll b/llvm/test/Transforms/LoopVectorize/X86/pr51366-sunk-instruction-used-outside-of-loop.ll
index b0cac6bab7d44..347b8cf75c05e 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/pr51366-sunk-instruction-used-outside-of-loop.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/pr51366-sunk-instruction-used-outside-of-loop.ll
@@ -11,15 +11,12 @@ define ptr @test(ptr noalias %src, ptr noalias %dst) {
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_LOAD_CONTINUE2:.*]] ]
; CHECK-NEXT: [[VEC_IND:%.*]] = phi <2 x i64> [ <i64 0, i64 1>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[PRED_LOAD_CONTINUE2]] ]
-; CHECK-NEXT: [[TMP1:%.*]] = add i64 [[INDEX]], 1
-; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[TMP1]]
-; CHECK-NEXT: [[TMP16:%.*]] = insertelement <2 x ptr> poison, ptr [[TMP6]], i32 0
-; CHECK-NEXT: [[TMP18:%.*]] = insertelement <2 x ptr> [[TMP16]], ptr [[TMP2]], i32 1
; CHECK-NEXT: [[TMP4:%.*]] = icmp ne <2 x i64> [[VEC_IND]], zeroinitializer
; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x i1> [[TMP4]], i64 0
; CHECK-NEXT: br i1 [[TMP5]], label %[[PRED_LOAD_IF:.*]], label %[[PRED_LOAD_CONTINUE:.*]]
; CHECK: [[PRED_LOAD_IF]]:
+; CHECK-NEXT: [[TMP3:%.*]] = add i64 [[INDEX]], 0
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[TMP3]]
; CHECK-NEXT: [[TMP7:%.*]] = load i32, ptr [[TMP6]], align 4
; CHECK-NEXT: [[TMP8:%.*]] = insertelement <2 x i32> poison, i32 [[TMP7]], i64 0
; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE]]
@@ -28,6 +25,8 @@ define ptr @test(ptr noalias %src, ptr noalias %dst) {
; CHECK-NEXT: [[TMP10:%.*]] = extractelement <2 x i1> [[TMP4]], i64 1
; CHECK-NEXT: br i1 [[TMP10]], label %[[PRED_LOAD_IF1:.*]], label %[[PRED_LOAD_CONTINUE2]]
; CHECK: [[PRED_LOAD_IF1]]:
+; CHECK-NEXT: [[TMP13:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[TMP13]]
; CHECK-NEXT: [[TMP11:%.*]] = load i32, ptr [[TMP2]], align 4
; CHECK-NEXT: [[TMP12:%.*]] = insertelement <2 x i32> [[TMP9]], i32 [[TMP11]], i64 1
; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE2]]
@@ -41,9 +40,10 @@ define ptr @test(ptr noalias %src, ptr noalias %dst) {
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1000
; CHECK-NEXT: br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP16:%.*]] = getelementptr i8, ptr [[SRC]], i64 3996
; CHECK-NEXT: br label %[[EXIT:.*]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: ret ptr [[TMP2]]
+; CHECK-NEXT: ret ptr [[TMP16]]
;
entry:
br label %loop.header
diff --git a/llvm/test/Transforms/LoopVectorize/cast-iv-outside-user.ll b/llvm/test/Transforms/LoopVectorize/cast-iv-outside-user.ll
new file mode 100644
index 0000000000000..58cc1df853149
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/cast-iv-outside-user.ll
@@ -0,0 +1,64 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt < %s -S -passes=loop-vectorize -force-vector-width=4 | FileCheck %s
+
+define i32 @incremented_iv_live_out(ptr %arr, i32 %n) {
+; CHECK-LABEL: define i32 @incremented_iv_live_out(
+; CHECK-SAME: ptr [[ARR:%.*]], i32 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = zext i32 [[N]] to i64
+; CHECK-NEXT: [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[TMP0]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[UMAX]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[UMAX]], 4
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[UMAX]], [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr i8, ptr [[ARR]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = add <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; CHECK-NEXT: store <4 x i8> [[TMP2]], ptr [[TMP1]], align 1
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP4:%.*]] = sub i64 [[N_VEC]], 1
+; CHECK-NEXT: [[TMP5:%.*]] = trunc i64 [[TMP4]] to i32
+; CHECK-NEXT: [[TMP6:%.*]] = add i32 1, [[TMP5]]
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[UMAX]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr i8, ptr [[ARR]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-NEXT: [[VAL_INC:%.*]] = add i8 [[VAL]], 1
+; CHECK-NEXT: store i8 [[VAL_INC]], ptr [[GEP]], align 1
+; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT: [[IV_TRUNC:%.*]] = trunc i64 [[IV_NEXT]] to i32
+; CHECK-NEXT: [[COND:%.*]] = icmp ult i32 [[IV_TRUNC]], [[N]]
+; CHECK-NEXT: br i1 [[COND]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[IV_TRUNC_LCSSA:%.*]] = phi i32 [ [[IV_TRUNC]], %[[LOOP]] ], [ [[TMP6]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i32 [[IV_TRUNC_LCSSA]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep = getelementptr i8, ptr %arr, i64 %iv
+ %val = load i8, ptr %gep, align 1
+ %val.inc = add i8 %val, 1
+ store i8 %val.inc, ptr %gep, align 1
+ %iv.next = add i64 %iv, 1
+ %iv.trunc = trunc i64 %iv.next to i32
+ %cond = icmp ult i32 %iv.trunc, %n
+ br i1 %cond, label %loop, label %exit
+
+exit:
+ ret i32 %iv.trunc
+}
diff --git a/llvm/test/Transforms/LoopVectorize/instruction-only-used-outside-of-loop.ll b/llvm/test/Transforms/LoopVectorize/instruction-only-used-outside-of-loop.ll
index 1bec39fbae92f..7ffeb8b3c2908 100644
--- a/llvm/test/Transforms/LoopVectorize/instruction-only-used-outside-of-loop.ll
+++ b/llvm/test/Transforms/LoopVectorize/instruction-only-used-outside-of-loop.ll
@@ -15...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/195059
More information about the llvm-commits
mailing list