[llvm] [VPlan] Model first memory runtime checks as VPlan recipes. (PR #221483)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 28 14:07:25 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/221483
>From 53c5f377e7d07d08b65104a048a0f938f6f736cb Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 3 Sep 2026 14:40:45 +0100
Subject: [PATCH 1/8] [VPlan] Model first memory runtime checks as VPlan
recipes.
Add VPlanTransforms::addMemoryRuntimeChecks to generate memory runtime
checks directly as VPlan recipes, re-using VPSCEVExpander. This enables
more CSE/simplification opportunities in VPlan, and in the long term
allows us to remove the current slightly hacky runtime check generation
code, which builds checks up front in IR, and removes them again if we
do not vectorize. This whole dance requires quite a bit of code to make
the removal step transparent if the loop was not vectorized.
The initial patch excludes the following checks from VPlan expansion (to
be added in follow-ups), to keep the initial logic as simple as
possible:
* checks in nested loops (VPlan expansion cannot hoist out of the outer
loop)
* diff checks
* bounds requiring expanding AddRecs (currently AddRecs must be
expanded in the Plan's entry)
The IR block built by GeneratedRTChecks::create() is still needed to cost the
checks and to detect that they folded to a constant. Once all kinds of
runtime checks can be created directly in VPlan, we can compute their
cost via the VPlan-based cost model, like other skeleton blocks.
---
.../Vectorize/LoopVectorizationPlanner.h | 5 +-
.../Transforms/Vectorize/LoopVectorize.cpp | 106 ++++++++++++++----
.../Vectorize/VPlanConstruction.cpp | 45 +++++++-
.../Transforms/Vectorize/VPlanTransforms.h | 8 ++
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 29 ++++-
.../LoopVectorize/AArch64/predicated-costs.ll | 3 +-
.../LoopVectorize/AArch64/reduction-cost.ll | 3 +-
.../RISCV/blocks-with-dead-instructions.ll | 4 +-
.../LoopVectorize/RISCV/dead-ops-cost.ll | 10 +-
.../LoopVectorize/RISCV/induction-costs.ll | 12 +-
.../truncate-to-minimal-bitwidth-cost.ll | 7 +-
.../RISCV/type-info-cache-evl-crash.ll | 3 +-
.../LoopVectorize/VPlan/memory-checks.ll | 87 +++++++-------
.../LoopVectorize/X86/cost-model.ll | 5 +-
.../LoopVectorize/X86/interleave-cost.ll | 3 +-
.../LoopVectorize/consecutive-ptr-uniforms.ll | 20 ++--
.../Transforms/LoopVectorize/if-conversion.ll | 5 +-
.../Transforms/LoopVectorize/induction.ll | 30 ++---
.../interleaved-accesses-metadata.ll | 6 +-
.../invariant-store-vectorization-2.ll | 9 +-
.../invariant-store-vectorization.ll | 15 +--
.../test/Transforms/LoopVectorize/metadata.ll | 12 +-
.../Transforms/LoopVectorize/opaque-ptr.ll | 24 ++--
.../pointer-select-runtime-checks.ll | 36 +++---
.../pr59319-loop-access-info-invalidation.ll | 10 +-
.../Transforms/LoopVectorize/runtime-check.ll | 12 +-
26 files changed, 293 insertions(+), 216 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index e14c335701f62..a848f4c4484f3 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -1009,9 +1009,10 @@ class LoopVectorizationPlanner {
void addMinimumIterationCheck(VPlan &Plan, ElementCount VF, unsigned UF,
ElementCount MinProfitableTripCount) const;
- /// Attach the runtime checks of \p RTChecks to \p Plan.
+ /// Attach the runtime checks of \p RTChecks to \p Plan. Generates the memory
+ /// checks as recipes if \p UseVPlanMemChecks is true and they are supported.
void attachRuntimeChecks(VPlan &Plan, GeneratedRTChecks &RTChecks,
- bool HasBranchWeights) const;
+ bool HasBranchWeights, bool UseVPlanMemChecks) const;
/// Update loop metadata and profile info for both the scalar remainder loop
/// and \p VectorLoop, if it exists. Keeps all loop hints from the original
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 6e27d1e209daf..c6062de3042b8 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1546,6 +1546,10 @@ class GeneratedRTChecks {
/// If it is nullptr no memory runtime checks have been generated.
Value *MemRuntimeCheckCond = nullptr;
+ /// Set in create() if memory checks were generated and did not fold away.
+ /// Unlike MemRuntimeCheckCond, stays set when the pre-built block is dropped.
+ bool HasMemChecks = false;
+
DominatorTree *DT;
LoopInfo *LI;
TargetTransformInfo *TTI;
@@ -1647,6 +1651,7 @@ class GeneratedRTChecks {
assert(MemRuntimeCheckCond &&
"no RT checks generated although RtPtrChecking "
"claimed checks are required");
+ HasMemChecks = getMemRuntimeChecks().first != nullptr;
}
SCEVExp.eraseDeadInstructions(SCEVCheckCond);
@@ -1773,32 +1778,17 @@ class GeneratedRTChecks {
/// unused.
~GeneratedRTChecks() {
SCEVExpanderCleaner SCEVCleaner(SCEVExp);
- SCEVExpanderCleaner MemCheckCleaner(MemCheckExp);
bool SCEVChecksUsed = !SCEVCheckBlock || !pred_empty(SCEVCheckBlock);
- bool MemChecksUsed = !MemCheckBlock || !pred_empty(MemCheckBlock);
if (SCEVChecksUsed)
SCEVCleaner.markResultUsed();
- if (MemChecksUsed) {
- MemCheckCleaner.markResultUsed();
- } else {
- auto &SE = *MemCheckExp.getSE();
- // Memory runtime check generation creates compares that use expanded
- // values. Remove them before running the SCEVExpanderCleaners.
- for (auto &I : make_early_inc_range(reverse(*MemCheckBlock))) {
- if (MemCheckExp.isInsertedInstruction(&I))
- continue;
- SE.forgetValue(&I);
- I.eraseFromParent();
- }
- }
- MemCheckCleaner.cleanup();
+ if (MemCheckBlock && pred_empty(MemCheckBlock))
+ eraseMemCheckBlock();
+
SCEVCleaner.cleanup();
if (!SCEVChecksUsed)
SCEVCheckBlock->eraseFromParent();
- if (!MemChecksUsed)
- MemCheckBlock->eraseFromParent();
}
/// Retrieves the SCEVCheckCond and SCEVCheckBlock that were generated as IR
@@ -1821,8 +1811,33 @@ class GeneratedRTChecks {
}
/// Return true if any runtime checks have been added
- bool hasChecks() const {
- return getSCEVChecks().first || getMemRuntimeChecks().first;
+ bool hasChecks() const { return getSCEVChecks().first || HasMemChecks; }
+
+ /// Drop the pre-built memory check block in favour of VPlan recipes.
+ /// TODO: Remove once the checks can be costed in VPlan, before VF selection.
+ void dropMemRuntimeChecks() {
+ assert(MemCheckBlock && pred_empty(MemCheckBlock) &&
+ "cannot drop memory checks that are missing or already connected");
+ eraseMemCheckBlock();
+ }
+
+private:
+ /// Erase the memory check block, its instructions and their SCEV expansions.
+ void eraseMemCheckBlock() {
+ SCEVExpanderCleaner MemCheckCleaner(MemCheckExp);
+ auto &SE = *MemCheckExp.getSE();
+ // Memory runtime check generation creates compares that use expanded
+ // values. Remove them before running the SCEVExpanderCleaner.
+ for (auto &I : make_early_inc_range(reverse(*MemCheckBlock))) {
+ if (MemCheckExp.isInsertedInstruction(&I))
+ continue;
+ SE.forgetValue(&I);
+ I.eraseFromParent();
+ }
+ MemCheckCleaner.cleanup();
+ MemCheckBlock->eraseFromParent();
+ MemCheckBlock = nullptr;
+ MemRuntimeCheckCond = nullptr;
}
};
} // namespace
@@ -7001,8 +7016,37 @@ void LoopVectorizationPlanner::addReductionResultComputation(
RUN_VPLAN_PASS(VPlanTransforms::clearReductionWrapFlags, *Plan);
}
+/// Return true if \p CG's bounds can be expanded in the check block.
+/// VPSCEVExpander only expands AddRecs of loops enclosing the plan's scope.
+static bool boundsAreVPlanExpandable(const RuntimeCheckingPtrGroup &CG,
+ ScalarEvolution &SE) {
+ return !SE.containsAddRecurrence(CG.Low) &&
+ !SE.containsAddRecurrence(CG.High);
+}
+
+/// Return true if \p RtPtrChecking's memory checks for \p OrigLoop can be
+/// modelled as VPlan recipes.
+static bool
+canModelMemChecksInVPlan(const RuntimePointerChecking &RtPtrChecking,
+ const Loop &OrigLoop, ScalarEvolution &SE) {
+ // Diff checks are not modelled in VPlan yet.
+ if (RtPtrChecking.getDiffChecks())
+ return false;
+
+ // The VPlan expander cannot hoist bounds out of an enclosing loop.
+ if (OrigLoop.getParentLoop())
+ return false;
+
+ ArrayRef<RuntimePointerCheck> Checks = RtPtrChecking.getChecks();
+ return !Checks.empty() && all_of(Checks, [&SE](const RuntimePointerCheck &C) {
+ return boundsAreVPlanExpandable(*C.first, SE) &&
+ boundsAreVPlanExpandable(*C.second, SE);
+ });
+}
+
void LoopVectorizationPlanner::attachRuntimeChecks(
- VPlan &Plan, GeneratedRTChecks &RTChecks, bool HasBranchWeights) const {
+ VPlan &Plan, GeneratedRTChecks &RTChecks, bool HasBranchWeights,
+ bool UseVPlanMemChecks) const {
const auto &[SCEVCheckCond, SCEVCheckBlock] = RTChecks.getSCEVChecks();
if (SCEVCheckBlock && SCEVCheckBlock->hasNPredecessors(0)) {
assert((!Config.OptForSize ||
@@ -7033,6 +7077,17 @@ void LoopVectorizationPlanner::attachRuntimeChecks(
"(e.g., adding 'restrict').";
});
}
+ const RuntimePointerChecking &RtPtrChecking =
+ *Legal->getLAI()->getRuntimePointerChecking();
+ ScalarEvolution &SE = *PSE.getSE();
+ if (UseVPlanMemChecks &&
+ canModelMemChecksInVPlan(RtPtrChecking, *OrigLoop, SE)) {
+ RTChecks.dropMemRuntimeChecks();
+ RUN_VPLAN_PASS(VPlanTransforms::addMemoryRuntimeChecks, Plan,
+ RtPtrChecking.getChecks(), SE, OrigLoop->getStartLoc(),
+ HasBranchWeights);
+ return;
+ }
RUN_VPLAN_PASS(VPlanTransforms::attachCheckBlock, Plan, MemCheckCond,
MemCheckBlock, HasBranchWeights);
}
@@ -8175,7 +8230,11 @@ bool LoopVectorizePass::processLoop(Loop *L) {
// checks for the main plan.
LVP.addMinimumIterationCheck(BestMainPlan, EPI.EpilogueVF, EPI.EpilogueUF,
ElementCount::getFixed(0));
- LVP.attachRuntimeChecks(BestMainPlan, Checks, HasBranchWeights);
+ // Epilogue vectorization has not been converted to VPlan memory checks
+ // yet; it shares the checks between the main and epilogue plans via the
+ // pre-built IR block.
+ LVP.attachRuntimeChecks(BestMainPlan, Checks, HasBranchWeights,
+ /*UseVPlanMemChecks=*/false);
RUN_VPLAN_PASS(
VPlanTransforms::addIterationCountCheckBlock, BestMainPlan,
EPI.MainLoopVF, EPI.MainLoopUF, BestMainPlan.requiresScalarEpilogue(),
@@ -8226,7 +8285,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
BestPlan);
LVP.addMinimumIterationCheck(BestPlan, VF.Width, IC,
VF.MinProfitableTripCount);
- LVP.attachRuntimeChecks(BestPlan, Checks, HasBranchWeights);
+ LVP.attachRuntimeChecks(BestPlan, Checks, HasBranchWeights,
+ /*UseVPlanMemChecks=*/true);
if (!IsInnerLoop)
LLVM_DEBUG(dbgs() << "Vectorizing outer loop in \"" << F->getName()
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 18498084a601a..982ac67c17bfd 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -24,6 +24,7 @@
#include "llvm/ADT/SmallVectorExtras.h"
#include "llvm/Analysis/BranchProbabilityInfo.h"
#include "llvm/Analysis/Loads.h"
+#include "llvm/Analysis/LoopAccessAnalysis.h"
#include "llvm/Analysis/LoopInfo.h"
#include "llvm/Analysis/LoopIterator.h"
#include "llvm/Analysis/OptimizationRemarkEmitter.h"
@@ -1493,13 +1494,55 @@ void VPlanTransforms::attachCheckBlock(VPlan &Plan, Value *Cond,
attachVPCheckBlock(Plan, CondVPV, CheckBlockVPBB, AddBranchWeights);
}
+void VPlanTransforms::addMemoryRuntimeChecks(
+ VPlan &Plan, ArrayRef<RuntimePointerCheck> Checks, ScalarEvolution &SE,
+ DebugLoc DL, bool AddBranchWeights) {
+ assert(!Checks.empty() && "no checks to replace the pre-built block with");
+
+ auto *MemCheckVPBB = Plan.createVPBasicBlock("vector.memcheck");
+ VPBuilder Builder(MemCheckVPBB);
+ insertCheckBlockBeforeVectorLoop(Plan, MemCheckVPBB);
+ VPSCEVExpander Expander(Builder, SE, DL);
+
+ // Expand each group's bounds once and up front.
+ SmallDenseMap<const RuntimeCheckingPtrGroup *,
+ std::pair<VPValue *, VPValue *>>
+ GroupToBounds;
+ for (const auto &[A, B] : Checks)
+ for (const RuntimeCheckingPtrGroup *CG : {A, B}) {
+ if (GroupToBounds.contains(CG))
+ continue;
+ VPValue *Start = Expander.expand(CG->Low);
+ VPValue *End = Expander.expand(CG->High);
+ if (CG->NeedsFreeze) {
+ Start = Builder.createScalarFreeze(Start, DL);
+ End = Builder.createScalarFreeze(End, DL);
+ }
+ GroupToBounds.try_emplace(CG, Start, End);
+ }
+
+ VPValue *Cond = nullptr;
+ for (const auto &[A, B] : Checks) {
+ auto [AStart, AEnd] = GroupToBounds.at(A);
+ auto [BStart, BEnd] = GroupToBounds.at(B);
+ VPValue *Bound0 =
+ Builder.createICmp(CmpInst::ICMP_ULT, AStart, BEnd, DL, "bound0");
+ VPValue *Bound1 =
+ Builder.createICmp(CmpInst::ICMP_ULT, BStart, AEnd, DL, "bound1");
+ VPValue *IsConflict =
+ Builder.createAnd(Bound0, Bound1, DL, "found.conflict");
+ Cond = Cond ? Builder.createOr(Cond, IsConflict, DL, "conflict.rdx")
+ : IsConflict;
+ }
+ addBypassBranch(Plan, MemCheckVPBB, Cond, AddBranchWeights);
+}
+
void VPlanTransforms::addMinimumIterationCheck(
VPlan &Plan, ElementCount VF, unsigned UF,
ElementCount MinProfitableTripCount, bool RequiresScalarEpilogue,
bool TailFolded, Loop *OrigLoop, const uint32_t *MinItersBypassWeights,
DebugLoc DL, PredicatedScalarEvolution &PSE, VPBasicBlock *CheckBlock) {
// Generate code to check if the loop's trip count is less than VF * UF, or
- // equal to it in case a scalar epilogue is required; this implies that the
// vector trip count is zero. This check also covers the case where adding one
// to the backedge-taken count overflowed leading to an incorrect trip count
// of zero. In this case we will also jump to the scalar loop.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 6e5a2184270a5..7cf2143bc03ed 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -17,6 +17,7 @@
#include "VPlanVerifier.h"
#include "llvm/ADT/STLFunctionalExtras.h"
#include "llvm/ADT/ScopeExit.h"
+#include "llvm/Analysis/LoopAccessAnalysis.h"
#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/Support/CommandLine.h"
#include "llvm/Support/Compiler.h"
@@ -236,6 +237,13 @@ struct VPlanTransforms {
static void attachCheckBlock(VPlan &Plan, Value *Cond, BasicBlock *CheckBlock,
bool AddBranchWeights);
+ /// Generate recipes for the memory runtime checks \p Checks in a new block
+ /// added to \p Plan.
+ static void addMemoryRuntimeChecks(VPlan &Plan,
+ ArrayRef<RuntimePointerCheck> Checks,
+ ScalarEvolution &SE, DebugLoc DL,
+ bool AddBranchWeights);
+
/// Replaces the VPInstructions in \p Plan with corresponding
/// widen recipes. Returns false if any VPInstructions could not be converted
/// to a wide recipe if needed. Uses \p PSE to detect contiguous memory
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 9a07697f66762..edc2f212e2509 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -954,6 +954,20 @@ VPValue *VPSCEVExpander::expand(const SCEV *S) {
}
case scMulExpr: {
auto *MulE = cast<SCEVMulExpr>(S);
+
+ // mul(PowerOf2C, udiv(X, PowerOf2C)) == (X >> C) << C -> X & (-1 << C),
+ // matching SCEVExpander::visitMulExpr.
+ const SCEVConstant *C1, *C2;
+ const SCEV *Val;
+ if (match(S, m_scev_Mul(m_SCEVConstant(C1),
+ m_scev_UDiv(m_SCEV(Val), m_SCEVConstant(C2)))) &&
+ C1 == C2 && C1->getAPInt().isPowerOf2()) {
+ VPValue *LHS = expand(Val);
+ APInt Mask = APInt::getBitsSetFrom(MulE->getType()->getScalarSizeInBits(),
+ C1->getAPInt().logBase2());
+ return Builder.createAnd(LHS, Builder.getPlan().getConstantInt(Mask), DL);
+ }
+
VPIRFlags::WrapFlagsTy WrapFlags(MulE->hasNoUnsignedWrap(),
MulE->hasNoSignedWrap());
SmallVector<VPValue *, 2> Ops;
@@ -1077,9 +1091,18 @@ VPValue *VPSCEVExpander::expand(const SCEV *S) {
Ops.push_back(OpV);
}
VPValue *Result = Ops.front();
- for (VPValue *Op : drop_begin(Ops))
- Result = Builder.createScalarIntrinsic(IntrinsicID, {Result, Op},
- ResultTy, DL);
+ for (VPValue *Op : drop_begin(Ops)) {
+ if (ResultTy->isPointerTy()) {
+ // The min/max intrinsics don't support pointer operands, so expand
+ // pointer-typed min/max as cmp + select, matching SCEVExpander.
+ VPValue *Cmp = Builder.createICmp(
+ MinMaxIntrinsic::getPredicate(IntrinsicID), Result, Op, DL);
+ Result = Builder.createSelect(Cmp, Result, Op, DL);
+ } else {
+ Result = Builder.createScalarIntrinsic(IntrinsicID, {Result, Op},
+ ResultTy, DL);
+ }
+ }
return Result;
}
case scAddRecExpr: {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/predicated-costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/predicated-costs.ll
index 285e978a5b9a8..7cdfe9be2ce06 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/predicated-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/predicated-costs.ll
@@ -45,8 +45,7 @@ define void @test_predicated_load_cast_hint(ptr %dst.1, ptr %dst.2, ptr %src, i8
; CHECK-NEXT: [[SCEVGEP5:%.*]] = getelementptr i8, ptr [[SRC]], i64 1
; CHECK-NEXT: [[TMP18:%.*]] = shl i64 [[OFF]], 3
; CHECK-NEXT: [[SCEVGEP6:%.*]] = getelementptr i8, ptr [[DST_1]], i64 [[TMP18]]
-; CHECK-NEXT: [[SMAX7:%.*]] = call i32 @llvm.smax.i32(i32 [[N_SUB]], i32 4)
-; CHECK-NEXT: [[TMP19:%.*]] = zext nneg i32 [[SMAX7]] to i64
+; CHECK-NEXT: [[TMP19:%.*]] = zext nneg i32 [[SMAX16]] to i64
; CHECK-NEXT: [[TMP20:%.*]] = add nsw i64 [[TMP19]], -1
; CHECK-NEXT: [[TMP21:%.*]] = lshr i64 [[TMP20]], 2
; CHECK-NEXT: [[TMP22:%.*]] = shl nuw nsw i64 [[TMP21]], 9
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll
index 32eeca5c60b88..f7d912dfedf50 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll
@@ -107,8 +107,7 @@ define i32 @or_reduction_with_freeze(ptr %dst, ptr %src) {
; CHECK-NEXT: [[IDENT_CHECK:%.*]] = icmp ne i64 [[TMP6]], 0
; CHECK-NEXT: br i1 [[IDENT_CHECK]], label %[[SCALAR_PH]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT: [[TMP7:%.*]] = sub i64 [[DST1]], [[SRC7]]
-; CHECK-NEXT: [[TMP9:%.*]] = and i64 [[TMP7]], -8
+; CHECK-NEXT: [[TMP9:%.*]] = and i64 [[TMP0]], -8
; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[TMP9]], 8
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP10]]
; CHECK-NEXT: [[SCEVGEP5:%.*]] = getelementptr i8, ptr [[DST]], i64 8
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/blocks-with-dead-instructions.ll b/llvm/test/Transforms/LoopVectorize/RISCV/blocks-with-dead-instructions.ll
index 9fa4da804956b..08ae8312afd61 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/blocks-with-dead-instructions.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/blocks-with-dead-instructions.ll
@@ -459,9 +459,7 @@ define void @dead_load_in_block(ptr %dst, ptr %src, i8 %N, i64 %x) #0 {
; CHECK-NEXT: [[TMP2:%.*]] = add nuw nsw i64 [[TMP1]], 1
; CHECK-NEXT: br label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT: [[TMP7:%.*]] = add nuw nsw i64 [[N_EXT]], 2
-; CHECK-NEXT: [[TMP8:%.*]] = udiv i64 [[TMP7]], 3
-; CHECK-NEXT: [[TMP5:%.*]] = mul nuw nsw i64 [[TMP8]], 12
+; CHECK-NEXT: [[TMP5:%.*]] = mul nuw nsw i64 [[TMP1]], 12
; CHECK-NEXT: [[TMP11:%.*]] = add nuw nsw i64 [[TMP5]], 4
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 [[TMP11]]
; CHECK-NEXT: [[TMP12:%.*]] = shl i64 [[X]], 2
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll b/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll
index 5d90cafc565c8..b0431edb04662 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll
@@ -130,7 +130,7 @@ exit:
; Test case for https://github.com/llvm/llvm-project/issues/106780.
define i32 @cost_of_exit_branch_and_cond_insts(ptr %a, ptr %b, i1 %c, i16 %x) vscale_range(2, 1024) {
; CHECK-LABEL: define i32 @cost_of_exit_branch_and_cond_insts(
-; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i1 [[C:%.*]], i16 [[X:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i1 [[C:%.*]], i16 [[X:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[TMP0:%.*]] = zext i16 [[X]] to i32
; CHECK-NEXT: [[UMAX3:%.*]] = call i32 @llvm.umax.i32(i32 [[TMP0]], i32 111)
@@ -141,11 +141,7 @@ define i32 @cost_of_exit_branch_and_cond_insts(ptr %a, ptr %b, i1 %c, i16 %x) vs
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 1
-; CHECK-NEXT: [[TMP3:%.*]] = zext i16 [[X]] to i32
-; CHECK-NEXT: [[UMAX1:%.*]] = call i32 @llvm.umax.i32(i32 [[TMP3]], i32 111)
-; CHECK-NEXT: [[TMP4:%.*]] = sub i32 770, [[UMAX1]]
-; CHECK-NEXT: [[SMAX:%.*]] = call i32 @llvm.smax.i32(i32 [[TMP4]], i32 0)
-; CHECK-NEXT: [[TMP5:%.*]] = zext nneg i32 [[SMAX]] to i64
+; CHECK-NEXT: [[TMP5:%.*]] = zext nneg i32 [[SMAX4]] to i64
; CHECK-NEXT: [[TMP6:%.*]] = shl nuw nsw i64 [[TMP5]], 2
; CHECK-NEXT: [[TMP7:%.*]] = add nuw nsw i64 [[TMP6]], 4
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP7]]
@@ -269,7 +265,7 @@ return:
; Test case for https://github.com/llvm/llvm-project/issues/107473.
define void @test_phi_in_latch_redundant(ptr %dst, i32 %a) vscale_range(2, 1024) {
; CHECK-LABEL: define void @test_phi_in_latch_redundant(
-; CHECK-SAME: ptr [[DST:%.*]], i32 [[A:%.*]]) #[[ATTR1]] {
+; CHECK-SAME: ptr [[DST:%.*]], i32 [[A:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/induction-costs.ll b/llvm/test/Transforms/LoopVectorize/RISCV/induction-costs.ll
index aa5734371f0d5..344f68b50aa82 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/induction-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/induction-costs.ll
@@ -21,17 +21,13 @@ define void @skip_free_iv_truncate(i16 %x, ptr %A) #0 {
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[TMP31:%.*]] = shl nsw i64 [[X_I64]], 1
; CHECK-NEXT: [[SCEVGEP9:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP31]]
-; CHECK-NEXT: [[SMAX10:%.*]] = call i64 @llvm.smax.i64(i64 [[X_I64]], i64 99)
-; CHECK-NEXT: [[TMP33:%.*]] = add i64 [[SMAX10]], 2
-; CHECK-NEXT: [[TMP34:%.*]] = sub i64 [[TMP33]], [[X_I64]]
-; CHECK-NEXT: [[TMP35:%.*]] = udiv i64 [[TMP34]], 3
-; CHECK-NEXT: [[TMP37:%.*]] = mul i64 [[TMP35]], 6
+; CHECK-NEXT: [[TMP37:%.*]] = mul i64 [[TMP3]], 6
; CHECK-NEXT: [[TMP38:%.*]] = add i64 [[TMP37]], [[TMP31]]
; CHECK-NEXT: [[TMP39:%.*]] = add i64 [[TMP38]], 2
; CHECK-NEXT: [[SCEVGEP12:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP39]]
; CHECK-NEXT: [[TMP40:%.*]] = shl nsw i64 [[X_I64]], 3
; CHECK-NEXT: [[SCEVGEP13:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP40]]
-; CHECK-NEXT: [[TMP41:%.*]] = mul i64 [[TMP35]], 24
+; CHECK-NEXT: [[TMP41:%.*]] = mul i64 [[TMP3]], 24
; CHECK-NEXT: [[TMP42:%.*]] = add i64 [[TMP41]], [[TMP40]]
; CHECK-NEXT: [[TMP43:%.*]] = add i64 [[TMP42]], 8
; CHECK-NEXT: [[SCEVGEP14:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP43]]
@@ -47,15 +43,13 @@ define void @skip_free_iv_truncate(i16 %x, ptr %A) #0 {
; CHECK-NEXT: [[CONFLICT_RDX:%.*]] = or i1 [[FOUND_CONFLICT]], [[FOUND_CONFLICT19]]
; CHECK-NEXT: br i1 [[CONFLICT_RDX]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP24:%.*]] = shl nsw i64 [[X_I64]], 1
-; CHECK-NEXT: [[SCEVGEP10:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP24]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[AVL:%.*]] = phi i64 [ [[TMP4]], %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP27:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 8, i1 true)
; CHECK-NEXT: [[TMP22:%.*]] = mul i64 [[INDEX]], 6
-; CHECK-NEXT: [[TMP23:%.*]] = getelementptr i8, ptr [[SCEVGEP10]], i64 [[TMP22]]
+; CHECK-NEXT: [[TMP23:%.*]] = getelementptr i8, ptr [[SCEVGEP9]], i64 [[TMP22]]
; CHECK-NEXT: call void @llvm.experimental.vp.strided.store.nxv8i16.p0.i64(<vscale x 8 x i16> zeroinitializer, ptr align 2 [[TMP23]], i64 6, <vscale x 8 x i1> splat (i1 true), i32 [[TMP27]]), !alias.scope [[META0:![0-9]+]], !noalias [[META3:![0-9]+]]
; CHECK-NEXT: [[TMP28:%.*]] = zext i32 [[TMP27]] to i64
; CHECK-NEXT: [[CURRENT_ITERATION_NEXT]] = add nuw i64 [[TMP28]], [[INDEX]]
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/truncate-to-minimal-bitwidth-cost.ll b/llvm/test/Transforms/LoopVectorize/RISCV/truncate-to-minimal-bitwidth-cost.ll
index e7acfe9a4f4fc..3848f7f89726c 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/truncate-to-minimal-bitwidth-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/truncate-to-minimal-bitwidth-cost.ll
@@ -264,12 +264,9 @@ define void @test_minbws_for_trunc(i32 %n, ptr noalias %p1, ptr noalias %p2) vsc
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[P2]], [[SCEVGEP]]
; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
; CHECK-NEXT: [[BOUND06:%.*]] = icmp ult ptr [[P2]], [[SCEVGEP5]]
-; CHECK-NEXT: [[BOUND17:%.*]] = icmp ult ptr [[P2]], [[SCEVGEP]]
-; CHECK-NEXT: [[FOUND_CONFLICT8:%.*]] = and i1 [[BOUND06]], [[BOUND17]]
+; CHECK-NEXT: [[FOUND_CONFLICT8:%.*]] = and i1 [[BOUND06]], [[BOUND1]]
; CHECK-NEXT: [[CONFLICT_RDX:%.*]] = or i1 [[FOUND_CONFLICT]], [[FOUND_CONFLICT8]]
-; CHECK-NEXT: [[BOUND09:%.*]] = icmp ult ptr [[P2]], [[SCEVGEP5]]
-; CHECK-NEXT: [[BOUND110:%.*]] = icmp ult ptr [[P2]], [[SCEVGEP4]]
-; CHECK-NEXT: [[FOUND_CONFLICT11:%.*]] = and i1 [[BOUND09]], [[BOUND110]]
+; CHECK-NEXT: [[FOUND_CONFLICT11:%.*]] = and i1 [[BOUND06]], [[BOUND0]]
; CHECK-NEXT: [[CONFLICT_RDX12:%.*]] = or i1 [[CONFLICT_RDX]], [[FOUND_CONFLICT11]]
; CHECK-NEXT: br i1 [[CONFLICT_RDX12]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/type-info-cache-evl-crash.ll b/llvm/test/Transforms/LoopVectorize/RISCV/type-info-cache-evl-crash.ll
index a9c793406032e..063722538d1ef 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/type-info-cache-evl-crash.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/type-info-cache-evl-crash.ll
@@ -12,8 +12,7 @@ define void @type_info_cache_clobber(ptr %dstv, ptr %src, i64 %wide.trip.count)
; CHECK-NEXT: br label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DSTV]], i64 1
-; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[WIDE_TRIP_COUNT]], 1
-; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP5]]
+; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP0]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[DSTV]], [[SCEVGEP1]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[SRC]], [[SCEVGEP]]
; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll b/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
index 344ccf7d03685..f52a8edeacef9 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
@@ -9,31 +9,30 @@ define void @three_groups_shared_bounds(ptr %a, ptr %b, ptr %c, i64 %n) {
; CHECK-NEXT: EMIT-SCALAR vp<[[VP2:%[0-9]+]]> = call i64 @llvm.umax(ir<%n>, ir<1>)
; CHECK-NEXT: EMIT vp<%min.iters.check> = icmp ult vp<[[VP2]]>, ir<4>
; CHECK-NEXT: EMIT branch-on-cond vp<%min.iters.check>
-; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, ir-bb<vector.memcheck>
+; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, vector.memcheck
; CHECK-EMPTY:
-; CHECK-NEXT: ir-bb<vector.memcheck>:
-; CHECK-NEXT: IR %umax = call i64 @llvm.umax.i64(i64 %n, i64 1)
-; CHECK-NEXT: IR %0 = shl i64 %umax, 2
-; CHECK-NEXT: IR %scevgep = getelementptr i8, ptr %c, i64 %0
-; CHECK-NEXT: IR %scevgep1 = getelementptr i8, ptr %a, i64 %0
-; CHECK-NEXT: IR %scevgep2 = getelementptr i8, ptr %b, i64 %0
-; CHECK-NEXT: IR %bound0 = icmp ult ptr %c, %scevgep1
-; CHECK-NEXT: IR %bound1 = icmp ult ptr %a, %scevgep
-; CHECK-NEXT: IR %found.conflict = and i1 %bound0, %bound1
-; CHECK-NEXT: IR %bound03 = icmp ult ptr %c, %scevgep2
-; CHECK-NEXT: IR %bound14 = icmp ult ptr %b, %scevgep
-; CHECK-NEXT: IR %found.conflict5 = and i1 %bound03, %bound14
-; CHECK-NEXT: IR %conflict.rdx = or i1 %found.conflict, %found.conflict5
-; CHECK-NEXT: IR %bound06 = icmp ult ptr %a, %scevgep2
-; CHECK-NEXT: IR %bound17 = icmp ult ptr %b, %scevgep1
-; CHECK-NEXT: IR %found.conflict8 = and i1 %bound06, %bound17
-; CHECK-NEXT: IR %conflict.rdx9 = or i1 %conflict.rdx, %found.conflict8
-; CHECK-NEXT: EMIT branch-on-cond ir<%conflict.rdx9>
+; CHECK-NEXT: vector.memcheck:
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = shl vp<[[VP2]]>, ir<2>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = ptradd ir<%c>, vp<[[VP4]]>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = ptradd ir<%a>, vp<[[VP4]]>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = ptradd ir<%b>, vp<[[VP4]]>
+; CHECK-NEXT: EMIT vp<%bound0> = icmp ult ir<%c>, vp<[[VP6]]>
+; CHECK-NEXT: EMIT vp<%bound1> = icmp ult ir<%a>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<%found.conflict> = and vp<%bound0>, vp<%bound1>
+; CHECK-NEXT: EMIT vp<%bound0>.1 = icmp ult ir<%c>, vp<[[VP7]]>
+; CHECK-NEXT: EMIT vp<%bound1>.1 = icmp ult ir<%b>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<%found.conflict>.1 = and vp<%bound0>.1, vp<%bound1>.1
+; CHECK-NEXT: EMIT vp<%conflict.rdx> = or vp<%found.conflict>, vp<%found.conflict>.1
+; CHECK-NEXT: EMIT vp<%bound0>.2 = icmp ult ir<%a>, vp<[[VP7]]>
+; CHECK-NEXT: EMIT vp<%bound1>.2 = icmp ult ir<%b>, vp<[[VP6]]>
+; CHECK-NEXT: EMIT vp<%found.conflict>.2 = and vp<%bound0>.2, vp<%bound1>.2
+; CHECK-NEXT: EMIT vp<%conflict.rdx>.1 = or vp<%conflict.rdx>, vp<%found.conflict>.2
+; CHECK-NEXT: EMIT branch-on-cond vp<%conflict.rdx>.1
; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, vector.ph
; CHECK-EMPTY:
; CHECK-NEXT: vector.ph:
-; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = and vp<[[VP2]]>, ir<3>
-; CHECK-NEXT: EMIT vp<%n.vec> = sub vp<[[VP2]]>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = and vp<[[VP2]]>, ir<3>
+; CHECK-NEXT: EMIT vp<%n.vec> = sub vp<[[VP2]]>, vp<[[VP9]]>
; CHECK-NEXT: Successor(s): vector.body
; CHECK-EMPTY:
; CHECK-NEXT: vector.body:
@@ -68,33 +67,33 @@ define void @ptr_minmax_bounds(ptr %a, ptr %b, i64 %n, i64 %s, i64 %t) {
; CHECK-NEXT: IR %step = mul i64 %s, %t
; CHECK-NEXT: EMIT vp<%min.iters.check> = icmp ult ir<%n>, ir<4>
; CHECK-NEXT: EMIT branch-on-cond vp<%min.iters.check>
-; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, ir-bb<vector.memcheck>
+; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, vector.memcheck
; CHECK-EMPTY:
-; CHECK-NEXT: ir-bb<vector.memcheck>:
-; CHECK-NEXT: IR %0 = shl i64 %n, 2
-; CHECK-NEXT: IR %scevgep = getelementptr i8, ptr %b, i64 %0
-; CHECK-NEXT: IR %1 = mul i64 %t, %s
-; CHECK-NEXT: IR %2 = add i64 %n, -1
-; CHECK-NEXT: IR %3 = mul i64 %1, %2
-; CHECK-NEXT: IR %4 = shl i64 %3, 2
-; CHECK-NEXT: IR %scevgep1 = getelementptr i8, ptr %a, i64 %4
-; CHECK-NEXT: IR %5 = icmp ult ptr %a, %scevgep1
-; CHECK-NEXT: IR %umin = select i1 %5, ptr %a, ptr %scevgep1
-; CHECK-NEXT: IR %6 = icmp ugt ptr %a, %scevgep1
-; CHECK-NEXT: IR %umax = select i1 %6, ptr %a, ptr %scevgep1
-; CHECK-NEXT: IR %scevgep2 = getelementptr i8, ptr %umax, i64 4
-; CHECK-NEXT: IR %bound0 = icmp ult ptr %b, %scevgep2
-; CHECK-NEXT: IR %bound1 = icmp ult ptr %umin, %scevgep
-; CHECK-NEXT: IR %found.conflict = and i1 %bound0, %bound1
-; CHECK-NEXT: EMIT branch-on-cond ir<%found.conflict>
+; CHECK-NEXT: vector.memcheck:
+; CHECK-NEXT: EMIT vp<[[VP3:%[0-9]+]]> = shl ir<%n>, ir<2>
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = ptradd ir<%b>, vp<[[VP3]]>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = add ir<%n>, ir<-1>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = mul ir<%t>, ir<%s>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = mul vp<[[VP6]]>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = shl vp<[[VP7]]>, ir<2>
+; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = ptradd ir<%a>, vp<[[VP8]]>
+; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = icmp ult ir<%a>, vp<[[VP9]]>
+; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = select vp<[[VP10]]>, ir<%a>, vp<[[VP9]]>
+; CHECK-NEXT: EMIT vp<[[VP12:%[0-9]+]]> = icmp ugt ir<%a>, vp<[[VP9]]>
+; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = select vp<[[VP12]]>, ir<%a>, vp<[[VP9]]>
+; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = ptradd vp<[[VP13]]>, ir<4>
+; CHECK-NEXT: EMIT vp<%bound0> = icmp ult ir<%b>, vp<[[VP14]]>
+; CHECK-NEXT: EMIT vp<%bound1> = icmp ult vp<[[VP11]]>, vp<[[VP4]]>
+; CHECK-NEXT: EMIT vp<%found.conflict> = and vp<%bound0>, vp<%bound1>
+; CHECK-NEXT: EMIT branch-on-cond vp<%found.conflict>
; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, vector.ph
; CHECK-EMPTY:
; CHECK-NEXT: vector.ph:
-; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = and ir<%n>, ir<3>
-; CHECK-NEXT: EMIT vp<%n.vec> = sub ir<%n>, vp<[[VP4]]>
-; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = broadcast ir<%step>
-; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = step-vector i64
-; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = broadcast ir<4>
+; CHECK-NEXT: EMIT vp<[[VP16:%[0-9]+]]> = and ir<%n>, ir<3>
+; CHECK-NEXT: EMIT vp<%n.vec> = sub ir<%n>, vp<[[VP16]]>
+; CHECK-NEXT: EMIT vp<[[VP17:%[0-9]+]]> = broadcast ir<%step>
+; CHECK-NEXT: EMIT vp<[[VP18:%[0-9]+]]> = step-vector i64
+; CHECK-NEXT: EMIT vp<[[VP19:%[0-9]+]]> = broadcast ir<4>
; CHECK-NEXT: Successor(s): vector.body
; CHECK-EMPTY:
; CHECK-NEXT: vector.body:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll b/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
index fe53027045aa2..42c8fe3011f7d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
@@ -473,10 +473,7 @@ define void @multi_exit(ptr %dst, ptr %src.1, ptr %src.2, i64 %A, i64 %B) #0 {
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 1
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[SRC_2]], i64 8
-; CHECK-NEXT: [[TMP15:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[B]], i64 1)
-; CHECK-NEXT: [[TMP16:%.*]] = freeze i64 [[TMP15]]
-; CHECK-NEXT: [[UMIN4:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP16]], i64 [[A]])
-; CHECK-NEXT: [[TMP17:%.*]] = shl i64 [[UMIN4]], 3
+; CHECK-NEXT: [[TMP17:%.*]] = shl i64 [[UMIN10]], 3
; CHECK-NEXT: [[TMP18:%.*]] = add i64 [[TMP17]], 8
; CHECK-NEXT: [[SCEVGEP5:%.*]] = getelementptr i8, ptr [[SRC_1]], i64 [[TMP18]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[DST]], [[SCEVGEP2]]
diff --git a/llvm/test/Transforms/LoopVectorize/X86/interleave-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/interleave-cost.ll
index aa5a0f5be24be..a4d7aa15dc266 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/interleave-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/interleave-cost.ll
@@ -250,9 +250,8 @@ define void @geps_feeding_interleave_groups_with_reuse2(ptr %A, ptr %B, i64 %N)
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP36]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[A]], [[SCEVGEP35]]
; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
-; CHECK-NEXT: [[BOUND038:%.*]] = icmp ult ptr [[B]], [[SCEVGEP36]]
; CHECK-NEXT: [[BOUND139:%.*]] = icmp ult ptr [[A]], [[SCEVGEP37]]
-; CHECK-NEXT: [[FOUND_CONFLICT40:%.*]] = and i1 [[BOUND038]], [[BOUND139]]
+; CHECK-NEXT: [[FOUND_CONFLICT40:%.*]] = and i1 [[BOUND0]], [[BOUND139]]
; CHECK-NEXT: [[CONFLICT_RDX:%.*]] = or i1 [[FOUND_CONFLICT]], [[FOUND_CONFLICT40]]
; CHECK-NEXT: br i1 [[CONFLICT_RDX]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
diff --git a/llvm/test/Transforms/LoopVectorize/consecutive-ptr-uniforms.ll b/llvm/test/Transforms/LoopVectorize/consecutive-ptr-uniforms.ll
index 50b72c49bf6bf..a82a0dfcc6185 100644
--- a/llvm/test/Transforms/LoopVectorize/consecutive-ptr-uniforms.ll
+++ b/llvm/test/Transforms/LoopVectorize/consecutive-ptr-uniforms.ll
@@ -1276,10 +1276,9 @@ define i32 @pointer_iv_mixed(ptr %a, ptr %b, i64 %n) {
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 3
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 3
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
-; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[A]], [[SCEVGEP]]
@@ -1342,10 +1341,9 @@ define i32 @pointer_iv_mixed(ptr %a, ptr %b, i64 %n) {
; INTER-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; INTER-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; INTER: [[VECTOR_MEMCHECK]]:
-; INTER-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; INTER-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 3
+; INTER-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 3
; INTER-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
-; INTER-NEXT: [[TMP1:%.*]] = shl i64 [[SMAX]], 2
+; INTER-NEXT: [[TMP1:%.*]] = shl i64 [[SMAX2]], 2
; INTER-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
; INTER-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
; INTER-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[A]], [[SCEVGEP]]
@@ -1436,9 +1434,8 @@ define void @pointer_operand_geps_with_different_indexed_types(ptr %A, ptr %B, i
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[SMAX]]
-; CHECK-NEXT: [[TMP6:%.*]] = shl i64 [[SMAX]], 3
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[SMAX2]]
+; CHECK-NEXT: [[TMP6:%.*]] = shl i64 [[SMAX2]], 3
; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[TMP6]], -4
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP0]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[TMP1]]
@@ -1512,9 +1509,8 @@ define void @pointer_operand_geps_with_different_indexed_types(ptr %A, ptr %B, i
; INTER-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ule i64 [[SMAX2]], 4
; INTER-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; INTER: [[VECTOR_MEMCHECK]]:
-; INTER-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; INTER-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[SMAX]]
-; INTER-NEXT: [[TMP8:%.*]] = shl i64 [[SMAX]], 3
+; INTER-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[SMAX2]]
+; INTER-NEXT: [[TMP8:%.*]] = shl i64 [[SMAX2]], 3
; INTER-NEXT: [[TMP0:%.*]] = add i64 [[TMP8]], -4
; INTER-NEXT: [[TMP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP0]]
; INTER-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[TMP1]]
diff --git a/llvm/test/Transforms/LoopVectorize/if-conversion.ll b/llvm/test/Transforms/LoopVectorize/if-conversion.ll
index 948746df3f580..b972d41dbbabb 100644
--- a/llvm/test/Transforms/LoopVectorize/if-conversion.ll
+++ b/llvm/test/Transforms/LoopVectorize/if-conversion.ll
@@ -34,10 +34,7 @@ define void @function0(ptr nocapture %a, ptr nocapture %b, i32 %start, i32 %end)
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[TMP5:%.*]] = shl nsw i64 [[TMP0]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP5]]
-; CHECK-NEXT: [[TMP6:%.*]] = add i32 [[END]], -1
-; CHECK-NEXT: [[TMP7:%.*]] = sub i32 [[TMP6]], [[START]]
-; CHECK-NEXT: [[TMP8:%.*]] = zext i32 [[TMP7]] to i64
-; CHECK-NEXT: [[TMP9:%.*]] = shl nuw nsw i64 [[TMP8]], 2
+; CHECK-NEXT: [[TMP9:%.*]] = shl nuw nsw i64 [[TMP3]], 2
; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[TMP5]], [[TMP9]]
; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[TMP10]], 4
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP11]]
diff --git a/llvm/test/Transforms/LoopVectorize/induction.ll b/llvm/test/Transforms/LoopVectorize/induction.ll
index 1683f784daf75..7d29db44a43b6 100644
--- a/llvm/test/Transforms/LoopVectorize/induction.ll
+++ b/llvm/test/Transforms/LoopVectorize/induction.ll
@@ -1551,12 +1551,10 @@ define void @scalarize_induction_variable_04(ptr %a, ptr %p, i32 %n) {
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[P:%.*]], i64 4
-; CHECK-NEXT: [[TMP3:%.*]] = add i32 [[N]], -1
-; CHECK-NEXT: [[TMP4:%.*]] = zext i32 [[TMP3]] to i64
-; CHECK-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP4]], 3
+; CHECK-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP1]], 3
; CHECK-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 8
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP6]]
-; CHECK-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP4]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP1]], 4
; CHECK-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 4
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 [[TMP8]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP2]]
@@ -1615,12 +1613,10 @@ define void @scalarize_induction_variable_04(ptr %a, ptr %p, i32 %n) {
; IND-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; IND: vector.memcheck:
; IND-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[P:%.*]], i64 4
-; IND-NEXT: [[TMP3:%.*]] = add i32 [[N]], -1
-; IND-NEXT: [[TMP4:%.*]] = zext i32 [[TMP3]] to i64
-; IND-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP4]], 3
+; IND-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP1]], 3
; IND-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 8
; IND-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP6]]
-; IND-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP4]], 4
+; IND-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP1]], 4
; IND-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 4
; IND-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 [[TMP8]]
; IND-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP2]]
@@ -1679,12 +1675,10 @@ define void @scalarize_induction_variable_04(ptr %a, ptr %p, i32 %n) {
; UNROLL-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; UNROLL: vector.memcheck:
; UNROLL-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[P:%.*]], i64 4
-; UNROLL-NEXT: [[TMP3:%.*]] = add i32 [[N]], -1
-; UNROLL-NEXT: [[TMP4:%.*]] = zext i32 [[TMP3]] to i64
-; UNROLL-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP4]], 3
+; UNROLL-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP1]], 3
; UNROLL-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 8
; UNROLL-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP6]]
-; UNROLL-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP4]], 4
+; UNROLL-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP1]], 4
; UNROLL-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 4
; UNROLL-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 [[TMP8]]
; UNROLL-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP2]]
@@ -1757,12 +1751,10 @@ define void @scalarize_induction_variable_04(ptr %a, ptr %p, i32 %n) {
; UNROLL-NO-IC-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; UNROLL-NO-IC: vector.memcheck:
; UNROLL-NO-IC-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[P:%.*]], i64 4
-; UNROLL-NO-IC-NEXT: [[TMP3:%.*]] = add i32 [[N]], -1
-; UNROLL-NO-IC-NEXT: [[TMP4:%.*]] = zext i32 [[TMP3]] to i64
-; UNROLL-NO-IC-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP4]], 3
+; UNROLL-NO-IC-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP1]], 3
; UNROLL-NO-IC-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 8
; UNROLL-NO-IC-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP6]]
-; UNROLL-NO-IC-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP4]], 4
+; UNROLL-NO-IC-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP1]], 4
; UNROLL-NO-IC-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 4
; UNROLL-NO-IC-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 [[TMP8]]
; UNROLL-NO-IC-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP2]]
@@ -1835,12 +1827,10 @@ define void @scalarize_induction_variable_04(ptr %a, ptr %p, i32 %n) {
; INTERLEAVE-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; INTERLEAVE: vector.memcheck:
; INTERLEAVE-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[P:%.*]], i64 4
-; INTERLEAVE-NEXT: [[TMP3:%.*]] = add i32 [[N]], -1
-; INTERLEAVE-NEXT: [[TMP4:%.*]] = zext i32 [[TMP3]] to i64
-; INTERLEAVE-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP4]], 3
+; INTERLEAVE-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP1]], 3
; INTERLEAVE-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 8
; INTERLEAVE-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP6]]
-; INTERLEAVE-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP4]], 4
+; INTERLEAVE-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP1]], 4
; INTERLEAVE-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 4
; INTERLEAVE-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 [[TMP8]]
; INTERLEAVE-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP2]]
diff --git a/llvm/test/Transforms/LoopVectorize/interleaved-accesses-metadata.ll b/llvm/test/Transforms/LoopVectorize/interleaved-accesses-metadata.ll
index df243aea6bbde..f09ceadc12629 100644
--- a/llvm/test/Transforms/LoopVectorize/interleaved-accesses-metadata.ll
+++ b/llvm/test/Transforms/LoopVectorize/interleaved-accesses-metadata.ll
@@ -76,8 +76,8 @@ define void @ir_tbaa_different(ptr %base, ptr %end, ptr %src) {
; CHECK-LABEL: define void @ir_tbaa_different(
; CHECK-SAME: ptr [[BASE:%.*]], ptr [[END:%.*]], ptr [[SRC:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
-; CHECK-NEXT: [[BASE2:%.*]] = ptrtoaddr ptr [[BASE]] to i64
; CHECK-NEXT: [[END2:%.*]] = ptrtoaddr ptr [[END]] to i64
+; CHECK-NEXT: [[BASE2:%.*]] = ptrtoaddr ptr [[BASE]] to i64
; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[END2]], -8
; CHECK-NEXT: [[TMP1:%.*]] = sub i64 [[TMP0]], [[BASE2]]
; CHECK-NEXT: [[TMP2:%.*]] = lshr i64 [[TMP1]], 3
@@ -85,9 +85,7 @@ define void @ir_tbaa_different(ptr %base, ptr %end, ptr %src) {
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP3]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT: [[TMP17:%.*]] = add i64 [[END2]], -8
-; CHECK-NEXT: [[TMP12:%.*]] = sub i64 [[TMP17]], [[BASE2]]
-; CHECK-NEXT: [[TMP14:%.*]] = and i64 [[TMP12]], -8
+; CHECK-NEXT: [[TMP14:%.*]] = and i64 [[TMP1]], -8
; CHECK-NEXT: [[TMP15:%.*]] = add i64 [[TMP14]], 8
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP15]]
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[SRC]], i64 4
diff --git a/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization-2.ll b/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization-2.ll
index 9c31a47e748fc..4401a0e00392f 100644
--- a/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization-2.ll
@@ -25,8 +25,7 @@ define void @inv_val_store_to_inv_address_conditional_diff_values_ic(ptr %a, i64
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
@@ -128,8 +127,7 @@ define void @inv_val_store_to_inv_address_conditional_inv(ptr %a, i64 %n, ptr %b
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
@@ -218,8 +216,7 @@ define i32 @variant_val_store_to_inv_address(ptr %a, i64 %n, ptr %b, i32 %k) {
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[A]], [[SCEVGEP1]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[B]], [[SCEVGEP]]
diff --git a/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll b/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll
index 3929994915bf4..638b3780c8adc 100644
--- a/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll
+++ b/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll
@@ -25,8 +25,7 @@ define i32 @inv_val_store_to_inv_address_with_reduction(ptr %a, i64 %n, ptr %b)
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[A]], [[SCEVGEP1]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[B]], [[SCEVGEP]]
@@ -101,8 +100,7 @@ define void @inv_val_store_to_inv_address(ptr %a, i64 %n, ptr %b) {
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[A]], [[SCEVGEP1]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[B]], [[SCEVGEP]]
@@ -177,8 +175,7 @@ define void @inv_val_store_to_inv_address_conditional(ptr %a, i64 %n, ptr %b, i3
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
@@ -355,8 +352,8 @@ define void @multiple_uniform_stores(ptr nocapture %var1, ptr nocapture readonly
; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[TMP2]], 4
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[VAR1:%.*]], i64 [[TMP3]]
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[VAR2:%.*]], i64 4
-; CHECK-NEXT: [[TMP5:%.*]] = add i32 [[ITR]], -1
; CHECK-NEXT: [[TMP10:%.*]] = add i32 [[ITR]], -1
+; CHECK-NEXT: [[TMP5:%.*]] = add i32 [[ITR]], -1
; CHECK-NEXT: br label [[FOR_COND1_PREHEADER:%.*]]
; CHECK: for.cond1.preheader:
; CHECK-NEXT: [[INDVARS_IV23:%.*]] = phi i64 [ [[INDVARS_IV_NEXT24:%.*]], [[FOR_INC8:%.*]] ], [ 0, [[FOR_COND1_PREHEADER_PREHEADER]] ]
@@ -367,7 +364,7 @@ define void @multiple_uniform_stores(ptr nocapture %var1, ptr nocapture readonly
; CHECK-NEXT: [[ARRAYIDX5:%.*]] = getelementptr inbounds i32, ptr [[VAR1]], i64 [[INDVARS_IV23]]
; CHECK-NEXT: [[TMP4:%.*]] = zext i32 [[J_022]] to i64
; CHECK-NEXT: [[ARRAYIDX5_PROMOTED:%.*]] = load i32, ptr [[ARRAYIDX5]], align 4
-; CHECK-NEXT: [[TMP6:%.*]] = sub i32 [[TMP10]], [[J_022]]
+; CHECK-NEXT: [[TMP6:%.*]] = sub i32 [[TMP5]], [[J_022]]
; CHECK-NEXT: [[TMP7:%.*]] = zext i32 [[TMP6]] to i64
; CHECK-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 1
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP8]], 4
@@ -375,7 +372,7 @@ define void @multiple_uniform_stores(ptr nocapture %var1, ptr nocapture readonly
; CHECK: vector.memcheck:
; CHECK-NEXT: [[TMP9:%.*]] = shl nuw nsw i64 [[TMP4]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[VAR2]], i64 [[TMP9]]
-; CHECK-NEXT: [[TMP11:%.*]] = sub i32 [[TMP5]], [[J_022]]
+; CHECK-NEXT: [[TMP11:%.*]] = sub i32 [[TMP10]], [[J_022]]
; CHECK-NEXT: [[TMP12:%.*]] = zext i32 [[TMP11]] to i64
; CHECK-NEXT: [[TMP13:%.*]] = shl nuw nsw i64 [[TMP12]], 2
; CHECK-NEXT: [[TMP14:%.*]] = add i64 [[TMP9]], [[TMP13]]
diff --git a/llvm/test/Transforms/LoopVectorize/metadata.ll b/llvm/test/Transforms/LoopVectorize/metadata.ll
index 4cc50326cdb2d..76b790d427935 100644
--- a/llvm/test/Transforms/LoopVectorize/metadata.ll
+++ b/llvm/test/Transforms/LoopVectorize/metadata.ll
@@ -501,8 +501,8 @@ define void @noalias_metadata(ptr align 8 %dst, ptr align 8 %src) {
; CHECK-LABEL: define void @noalias_metadata(
; CHECK-SAME: ptr align 8 [[DST:%.*]], ptr align 8 [[SRC:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
-; CHECK-NEXT: [[SRC2:%.*]] = ptrtoaddr ptr [[SRC]] to i64
; CHECK-NEXT: [[DST1:%.*]] = ptrtoaddr ptr [[DST]] to i64
+; CHECK-NEXT: [[SRC2:%.*]] = ptrtoaddr ptr [[SRC]] to i64
; CHECK-NEXT: [[TMP0:%.*]] = sub i64 [[DST1]], [[SRC2]]
; CHECK-NEXT: [[TMP1:%.*]] = lshr i64 [[TMP0]], 3
; CHECK-NEXT: [[TMP2:%.*]] = add nuw nsw i64 [[TMP1]], 1
@@ -510,8 +510,8 @@ define void @noalias_metadata(ptr align 8 %dst, ptr align 8 %src) {
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 8
-; CHECK-NEXT: [[TMP3:%.*]] = add i64 [[DST1]], 8
-; CHECK-NEXT: [[TMP4:%.*]] = sub i64 [[TMP3]], [[SRC2]]
+; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[DST1]], 8
+; CHECK-NEXT: [[TMP4:%.*]] = sub i64 [[TMP11]], [[SRC2]]
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP4]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[DST]], [[SCEVGEP3]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[SRC]], [[SCEVGEP]]
@@ -552,8 +552,8 @@ define void @noalias_metadata(ptr align 8 %dst, ptr align 8 %src) {
; INTERLEAVE-LABEL: define void @noalias_metadata(
; INTERLEAVE-SAME: ptr align 8 [[DST:%.*]], ptr align 8 [[SRC:%.*]]) {
; INTERLEAVE-NEXT: [[ENTRY:.*]]:
-; INTERLEAVE-NEXT: [[SRC2:%.*]] = ptrtoaddr ptr [[SRC]] to i64
; INTERLEAVE-NEXT: [[DST1:%.*]] = ptrtoaddr ptr [[DST]] to i64
+; INTERLEAVE-NEXT: [[SRC2:%.*]] = ptrtoaddr ptr [[SRC]] to i64
; INTERLEAVE-NEXT: [[TMP0:%.*]] = sub i64 [[DST1]], [[SRC2]]
; INTERLEAVE-NEXT: [[TMP1:%.*]] = lshr i64 [[TMP0]], 3
; INTERLEAVE-NEXT: [[TMP2:%.*]] = add nuw nsw i64 [[TMP1]], 1
@@ -561,8 +561,8 @@ define void @noalias_metadata(ptr align 8 %dst, ptr align 8 %src) {
; INTERLEAVE-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; INTERLEAVE: [[VECTOR_MEMCHECK]]:
; INTERLEAVE-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 8
-; INTERLEAVE-NEXT: [[TMP3:%.*]] = add i64 [[DST1]], 8
-; INTERLEAVE-NEXT: [[TMP4:%.*]] = sub i64 [[TMP3]], [[SRC2]]
+; INTERLEAVE-NEXT: [[TMP12:%.*]] = add i64 [[DST1]], 8
+; INTERLEAVE-NEXT: [[TMP4:%.*]] = sub i64 [[TMP12]], [[SRC2]]
; INTERLEAVE-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP4]]
; INTERLEAVE-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[DST]], [[SCEVGEP3]]
; INTERLEAVE-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[SRC]], [[SCEVGEP]]
diff --git a/llvm/test/Transforms/LoopVectorize/opaque-ptr.ll b/llvm/test/Transforms/LoopVectorize/opaque-ptr.ll
index 37dd2afb7da4c..ebcf8407da35f 100644
--- a/llvm/test/Transforms/LoopVectorize/opaque-ptr.ll
+++ b/llvm/test/Transforms/LoopVectorize/opaque-ptr.ll
@@ -20,9 +20,7 @@ define void @test_ptr_iv_no_inbounds(ptr %p1.start, ptr %p2.start, ptr %p1.end)
; CHECK-NEXT: [[IDENT_CHECK:%.*]] = icmp ne i64 [[TMP7]], 0
; CHECK-NEXT: br i1 [[IDENT_CHECK]], label [[SCALAR_PH]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP15:%.*]] = add i64 [[TMP8]], -4
-; CHECK-NEXT: [[TMP9:%.*]] = sub i64 [[TMP15]], [[P1_END1]]
-; CHECK-NEXT: [[TMP11:%.*]] = and i64 [[TMP9]], -4
+; CHECK-NEXT: [[TMP11:%.*]] = and i64 [[TMP1]], -4
; CHECK-NEXT: [[TMP12:%.*]] = add i64 [[TMP11]], 4
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[P1_END]], i64 [[TMP12]]
; CHECK-NEXT: [[SCEVGEP5:%.*]] = getelementptr i8, ptr [[P2_START:%.*]], i64 [[TMP12]]
@@ -94,20 +92,18 @@ define void @test_ptr_iv_with_inbounds(ptr %p1.start, ptr %p2.start, ptr %p1.end
; CHECK-NEXT: entry:
; CHECK-NEXT: [[P1_END1:%.*]] = ptrtoaddr ptr [[P1_END:%.*]] to i64
; CHECK-NEXT: [[TMP4:%.*]] = ptrtoaddr ptr [[P1_START:%.*]] to i64
-; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[TMP4]], -4
-; CHECK-NEXT: [[TMP1:%.*]] = sub i64 [[TMP0]], [[P1_END1]]
+; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[P1_END1]], -4
+; CHECK-NEXT: [[TMP1:%.*]] = sub i64 [[TMP5]], [[TMP4]]
; CHECK-NEXT: [[TMP2:%.*]] = lshr i64 [[TMP1]], 2
; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[TMP2]], 1
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP3]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[TMP4]], -4
-; CHECK-NEXT: [[TMP5:%.*]] = sub i64 [[TMP11]], [[P1_END1]]
-; CHECK-NEXT: [[TMP7:%.*]] = and i64 [[TMP5]], -4
+; CHECK-NEXT: [[TMP7:%.*]] = and i64 [[TMP1]], -4
; CHECK-NEXT: [[TMP8:%.*]] = add i64 [[TMP7]], 4
-; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[P1_END]], i64 [[TMP8]]
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[P1_START]], i64 [[TMP8]]
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[P2_START:%.*]], i64 [[TMP8]]
-; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[P1_END]], [[SCEVGEP3]]
+; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[P1_START]], [[SCEVGEP3]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[P2_START]], [[SCEVGEP]]
; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
; CHECK-NEXT: br i1 [[FOUND_CONFLICT]], label [[SCALAR_PH]], label [[VECTOR_PH:%.*]]
@@ -115,13 +111,13 @@ define void @test_ptr_iv_with_inbounds(ptr %p1.start, ptr %p2.start, ptr %p1.end
; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i64 [[TMP3]], 1
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP3]], [[N_MOD_VF]]
; CHECK-NEXT: [[TMP9:%.*]] = shl i64 [[N_VEC]], 2
-; CHECK-NEXT: [[TMP12:%.*]] = getelementptr i8, ptr [[P1_END]], i64 [[TMP9]]
+; CHECK-NEXT: [[TMP14:%.*]] = getelementptr i8, ptr [[P1_START]], i64 [[TMP9]]
; CHECK-NEXT: [[IND_END6:%.*]] = getelementptr i8, ptr [[P2_START]], i64 [[TMP9]]
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[OFFSET_IDX:%.*]] = shl i64 [[INDEX]], 2
-; CHECK-NEXT: [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[P1_END]], i64 [[OFFSET_IDX]]
+; CHECK-NEXT: [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[P1_START]], i64 [[OFFSET_IDX]]
; CHECK-NEXT: [[NEXT_GEP6:%.*]] = getelementptr i8, ptr [[P2_START]], i64 [[OFFSET_IDX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x float>, ptr [[NEXT_GEP]], align 4, !alias.scope [[META9:![0-9]+]], !noalias [[META12:![0-9]+]]
; CHECK-NEXT: [[WIDE_LOAD10:%.*]] = load <2 x float>, ptr [[NEXT_GEP6]], align 4, !alias.scope [[META12]]
@@ -134,7 +130,7 @@ define void @test_ptr_iv_with_inbounds(ptr %p1.start, ptr %p2.start, ptr %p1.end
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP3]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label [[EXIT:%.*]], label [[SCALAR_PH]]
; CHECK: scalar.ph:
-; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi ptr [ [[TMP12]], [[MIDDLE_BLOCK]] ], [ [[P1_END]], [[ENTRY:%.*]] ], [ [[P1_END]], [[VECTOR_MEMCHECK]] ]
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi ptr [ [[TMP14]], [[MIDDLE_BLOCK]] ], [ [[P1_START]], [[ENTRY:%.*]] ], [ [[P1_START]], [[VECTOR_MEMCHECK]] ]
; CHECK-NEXT: [[BC_RESUME_VAL7:%.*]] = phi ptr [ [[IND_END6]], [[MIDDLE_BLOCK]] ], [ [[P2_START]], [[ENTRY]] ], [ [[P2_START]], [[VECTOR_MEMCHECK]] ]
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
@@ -146,7 +142,7 @@ define void @test_ptr_iv_with_inbounds(ptr %p1.start, ptr %p2.start, ptr %p1.end
; CHECK-NEXT: store float [[SUM]], ptr [[P1]], align 4
; CHECK-NEXT: [[P1_NEXT]] = getelementptr inbounds float, ptr [[P1]], i64 1
; CHECK-NEXT: [[P2_NEXT]] = getelementptr inbounds float, ptr [[P2]], i64 1
-; CHECK-NEXT: [[C:%.*]] = icmp ne ptr [[P1_NEXT]], [[P1_START]]
+; CHECK-NEXT: [[C:%.*]] = icmp ne ptr [[P1_NEXT]], [[P1_END]]
; CHECK-NEXT: br i1 [[C]], label [[LOOP]], label [[EXIT]], !llvm.loop [[LOOP15:![0-9]+]]
; CHECK: exit:
; CHECK-NEXT: ret void
diff --git a/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll b/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll
index 432571456a888..afd4fc57c4642 100644
--- a/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll
+++ b/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll
@@ -11,8 +11,7 @@ define void @test1_select_invariant(ptr %src.1, ptr %src.2, ptr %dst, i1 %c, i8
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP3:%.*]] = add i8 [[N]], -1
-; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP3]] to i64
+; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP0]] to i64
; CHECK-NEXT: [[TMP5:%.*]] = add nuw nsw i64 [[TMP4]], 1
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST:%.*]], i64 [[TMP5]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[PTR_SEL]], i64 1
@@ -166,8 +165,7 @@ define void @test_loop_dependent_select2(ptr %src.1, ptr %src.2, ptr %dst, i8 %n
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP3:%.*]] = add i8 [[N]], -1
-; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP3]] to i64
+; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP0]] to i64
; CHECK-NEXT: [[TMP5:%.*]] = add nuw nsw i64 [[TMP4]], 1
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST:%.*]], i64 [[TMP5]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[SRC_1:%.*]], i64 1
@@ -197,12 +195,12 @@ define void @test_loop_dependent_select2(ptr %src.1, ptr %src.2, ptr %dst, i8 %n
; CHECK-NEXT: [[TMP10:%.*]] = icmp ult i8 [[TMP8]], [[X]]
; CHECK-NEXT: [[TMP11:%.*]] = select i1 [[TMP9]], ptr [[SRC_1]], ptr [[SRC_2]]
; CHECK-NEXT: [[TMP12:%.*]] = select i1 [[TMP10]], ptr [[SRC_1]], ptr [[SRC_2]]
-; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META11:![0-9]+]]
-; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META11]]
+; CHECK-NEXT: [[TMP20:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META11:![0-9]+]]
+; CHECK-NEXT: [[TMP21:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META11]]
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP7]]
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP8]]
-; CHECK-NEXT: store i8 [[TMP14]], ptr [[TMP15]], align 2, !alias.scope [[META14:![0-9]+]], !noalias [[META16:![0-9]+]]
-; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP16]], align 2, !alias.scope [[META14]], !noalias [[META16]]
+; CHECK-NEXT: store i8 [[TMP20]], ptr [[TMP15]], align 2, !alias.scope [[META14:![0-9]+]], !noalias [[META16:![0-9]+]]
+; CHECK-NEXT: store i8 [[TMP21]], ptr [[TMP16]], align 2, !alias.scope [[META14]], !noalias [[META16]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 2
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP17]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP18:![0-9]+]]
@@ -252,8 +250,7 @@ define void @test_loop_dependent_select_first_ptr_noundef(ptr noundef %src.1, pt
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP3:%.*]] = add i8 [[N]], -1
-; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP3]] to i64
+; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP0]] to i64
; CHECK-NEXT: [[TMP5:%.*]] = add nuw nsw i64 [[TMP4]], 1
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST:%.*]], i64 [[TMP5]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[SRC_1:%.*]], i64 1
@@ -281,12 +278,12 @@ define void @test_loop_dependent_select_first_ptr_noundef(ptr noundef %src.1, pt
; CHECK-NEXT: [[TMP10:%.*]] = icmp ult i8 [[TMP8]], [[X]]
; CHECK-NEXT: [[TMP11:%.*]] = select i1 [[TMP9]], ptr [[SRC_1]], ptr [[SRC_2]]
; CHECK-NEXT: [[TMP12:%.*]] = select i1 [[TMP10]], ptr [[SRC_1]], ptr [[SRC_2]]
-; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META20:![0-9]+]]
-; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META20]]
+; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META20:![0-9]+]]
+; CHECK-NEXT: [[TMP19:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META20]]
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP7]]
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP8]]
-; CHECK-NEXT: store i8 [[TMP14]], ptr [[TMP15]], align 2, !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
-; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP16]], align 2, !alias.scope [[META23]], !noalias [[META25]]
+; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP15]], align 2, !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
+; CHECK-NEXT: store i8 [[TMP19]], ptr [[TMP16]], align 2, !alias.scope [[META23]], !noalias [[META25]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 2
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP17]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP27:![0-9]+]]
@@ -336,8 +333,7 @@ define void @test_loop_dependent_select_second_ptr_noundef(ptr %src.1, ptr nound
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP3:%.*]] = add i8 [[N]], -1
-; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP3]] to i64
+; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP0]] to i64
; CHECK-NEXT: [[TMP5:%.*]] = add nuw nsw i64 [[TMP4]], 1
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST:%.*]], i64 [[TMP5]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[SRC_1:%.*]], i64 1
@@ -365,12 +361,12 @@ define void @test_loop_dependent_select_second_ptr_noundef(ptr %src.1, ptr nound
; CHECK-NEXT: [[TMP10:%.*]] = icmp ult i8 [[TMP8]], [[X]]
; CHECK-NEXT: [[TMP11:%.*]] = select i1 [[TMP9]], ptr [[SRC_1]], ptr [[SRC_2]]
; CHECK-NEXT: [[TMP12:%.*]] = select i1 [[TMP10]], ptr [[SRC_1]], ptr [[SRC_2]]
-; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META29:![0-9]+]]
-; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META29]]
+; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META29:![0-9]+]]
+; CHECK-NEXT: [[TMP19:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META29]]
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP7]]
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP8]]
-; CHECK-NEXT: store i8 [[TMP14]], ptr [[TMP15]], align 2, !alias.scope [[META32:![0-9]+]], !noalias [[META34:![0-9]+]]
-; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP16]], align 2, !alias.scope [[META32]], !noalias [[META34]]
+; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP15]], align 2, !alias.scope [[META32:![0-9]+]], !noalias [[META34:![0-9]+]]
+; CHECK-NEXT: store i8 [[TMP19]], ptr [[TMP16]], align 2, !alias.scope [[META32]], !noalias [[META34]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 2
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP17]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
diff --git a/llvm/test/Transforms/LoopVectorize/pr59319-loop-access-info-invalidation.ll b/llvm/test/Transforms/LoopVectorize/pr59319-loop-access-info-invalidation.ll
index 0caa6b2e1d621..6a4c3bc2c1fde 100644
--- a/llvm/test/Transforms/LoopVectorize/pr59319-loop-access-info-invalidation.ll
+++ b/llvm/test/Transforms/LoopVectorize/pr59319-loop-access-info-invalidation.ll
@@ -69,7 +69,7 @@ define void @reduced(ptr %0, ptr %1, i64 %iv, ptr %2, i64 %iv76, i64 %iv93) {
; CHECK-NEXT: [[ARRAYIDX_I_I62:%.*]] = getelementptr i32, ptr [[TMP0]], i64 [[IDXPROM_I_I61]]
; CHECK-NEXT: [[MIN_ITERS_CHECK20:%.*]] = icmp ult i64 [[TMP3]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK20]], label [[SCALAR_PH19:%.*]], label [[VECTOR_MEMCHECK13:%.*]]
-; CHECK: vector.memcheck13:
+; CHECK: vector.memcheck20:
; CHECK-NEXT: [[SCEVGEP14:%.*]] = getelementptr i8, ptr [[TMP1]], i64 4
; CHECK-NEXT: [[TMP10:%.*]] = shl nuw nsw i64 [[IDXPROM_I_I61]], 2
; CHECK-NEXT: [[TMP11:%.*]] = add nuw nsw i64 [[TMP10]], 4
@@ -78,20 +78,20 @@ define void @reduced(ptr %0, ptr %1, i64 %iv, ptr %2, i64 %iv76, i64 %iv93) {
; CHECK-NEXT: [[BOUND117:%.*]] = icmp ult ptr [[ARRAYIDX_I_I62]], [[SCEVGEP14]]
; CHECK-NEXT: [[FOUND_CONFLICT18:%.*]] = and i1 [[BOUND016]], [[BOUND117]]
; CHECK-NEXT: br i1 [[FOUND_CONFLICT18]], label [[SCALAR_PH19]], label [[VECTOR_PH21:%.*]]
-; CHECK: vector.ph21:
+; CHECK: vector.ph24:
; CHECK-NEXT: [[TMP12:%.*]] = and i64 [[TMP3]], 3
; CHECK-NEXT: [[N_VEC22:%.*]] = sub i64 [[TMP3]], [[TMP12]]
; CHECK-NEXT: br label [[VECTOR_BODY23:%.*]]
-; CHECK: vector.body23:
+; CHECK: vector.body26:
; CHECK-NEXT: [[INDEX24:%.*]] = phi i64 [ 0, [[VECTOR_PH21]] ], [ [[INDEX_NEXT25:%.*]], [[VECTOR_BODY23]] ]
; CHECK-NEXT: [[INDEX_NEXT25]] = add nuw i64 [[INDEX24]], 4
; CHECK-NEXT: [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT25]], [[N_VEC22]]
; CHECK-NEXT: br i1 [[TMP13]], label [[MIDDLE_BLOCK26:%.*]], label [[VECTOR_BODY23]], !llvm.loop [[LOOP10:![0-9]+]]
-; CHECK: middle.block26:
+; CHECK: middle.block29:
; CHECK-NEXT: store i32 0, ptr [[TMP1]], align 4, !alias.scope [[META11:![0-9]+]], !noalias [[META14:![0-9]+]]
; CHECK-NEXT: [[CMP_N27:%.*]] = icmp eq i64 [[TMP3]], [[N_VEC22]]
; CHECK-NEXT: br i1 [[CMP_N27]], label [[LOOP_CLEANUP:%.*]], label [[SCALAR_PH19]]
-; CHECK: scalar.ph19:
+; CHECK: scalar.ph18:
; CHECK-NEXT: [[BC_RESUME_VAL28:%.*]] = phi i64 [ [[N_VEC22]], [[MIDDLE_BLOCK26]] ], [ 0, [[LOOP_3_LR_PH]] ], [ 0, [[VECTOR_MEMCHECK13]] ]
; CHECK-NEXT: br label [[LOOP_3:%.*]]
; CHECK: loop.2:
diff --git a/llvm/test/Transforms/LoopVectorize/runtime-check.ll b/llvm/test/Transforms/LoopVectorize/runtime-check.ll
index 21e8a3ccdfa2f..beb8aa2c99811 100644
--- a/llvm/test/Transforms/LoopVectorize/runtime-check.ll
+++ b/llvm/test/Transforms/LoopVectorize/runtime-check.ll
@@ -310,10 +310,9 @@ define void @different_load_store_pairs(ptr %src.1, ptr %src.2, ptr %dst.1, ptr
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[UMAX]], 2
+; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[TMP0]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST_1:%.*]], i64 [[TMP1]]
-; CHECK-NEXT: [[TMP2:%.*]] = shl i64 [[UMAX]], 3
+; CHECK-NEXT: [[TMP2:%.*]] = shl i64 [[TMP0]], 3
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[DST_2:%.*]], i64 [[TMP2]]
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[SRC_1:%.*]], i64 [[TMP1]]
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[SRC_2:%.*]], i64 [[TMP2]]
@@ -346,11 +345,11 @@ define void @different_load_store_pairs(ptr %src.1, ptr %src.2, ptr %dst.1, ptr
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr i32, ptr [[SRC_1]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META22:![0-9]+]]
; CHECK-NEXT: [[TMP5:%.*]] = getelementptr i64, ptr [[SRC_2]], i64 [[INDEX]]
-; CHECK-NEXT: [[WIDE_LOAD19:%.*]] = load <4 x i64>, ptr [[TMP5]], align 8, !alias.scope [[META25:![0-9]+]]
+; CHECK-NEXT: [[WIDE_LOAD34:%.*]] = load <4 x i64>, ptr [[TMP5]], align 8, !alias.scope [[META25:![0-9]+]]
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr nusw i32, ptr [[DST_1]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x i32> [[WIDE_LOAD]], ptr [[TMP6]], align 4, !alias.scope [[META27:![0-9]+]], !noalias [[META29:![0-9]+]]
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr nusw i64, ptr [[DST_2]], i64 [[INDEX]]
-; CHECK-NEXT: store <4 x i64> [[WIDE_LOAD19]], ptr [[TMP7]], align 8, !alias.scope [[META31:![0-9]+]], !noalias [[META32:![0-9]+]]
+; CHECK-NEXT: store <4 x i64> [[WIDE_LOAD34]], ptr [[TMP7]], align 8, !alias.scope [[META31:![0-9]+]], !noalias [[META32:![0-9]+]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
@@ -494,8 +493,7 @@ define void @test_scev_check_mul_add_expansion(ptr %out, ptr %in, i32 %len, i32
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[OUT:%.*]], i64 12
-; CHECK-NEXT: [[SMAX:%.*]] = call i32 @llvm.smax.i32(i32 [[LEN]], i32 7)
-; CHECK-NEXT: [[TMP2:%.*]] = zext nneg i32 [[SMAX]] to i64
+; CHECK-NEXT: [[TMP2:%.*]] = zext nneg i32 [[TMP0]] to i64
; CHECK-NEXT: [[TMP3:%.*]] = shl nuw nsw i64 [[TMP2]], 1
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[OUT]], i64 [[TMP3]]
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[IN:%.*]], i64 4
>From 317abefdb0d344bce51f4ec84851552c3b43c7ad Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 15 Sep 2026 10:44:58 +0100
Subject: [PATCH 2/8] !fixup address comments, thanks
---
.../Transforms/Vectorize/LoopVectorize.cpp | 21 +++++++++----------
.../Transforms/Vectorize/VPlanTransforms.cpp | 8 +++++++
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 14 -------------
3 files changed, 18 insertions(+), 25 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index c6062de3042b8..e72d62f1b1d8e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7016,14 +7016,6 @@ void LoopVectorizationPlanner::addReductionResultComputation(
RUN_VPLAN_PASS(VPlanTransforms::clearReductionWrapFlags, *Plan);
}
-/// Return true if \p CG's bounds can be expanded in the check block.
-/// VPSCEVExpander only expands AddRecs of loops enclosing the plan's scope.
-static bool boundsAreVPlanExpandable(const RuntimeCheckingPtrGroup &CG,
- ScalarEvolution &SE) {
- return !SE.containsAddRecurrence(CG.Low) &&
- !SE.containsAddRecurrence(CG.High);
-}
-
/// Return true if \p RtPtrChecking's memory checks for \p OrigLoop can be
/// modelled as VPlan recipes.
static bool
@@ -7037,10 +7029,17 @@ canModelMemChecksInVPlan(const RuntimePointerChecking &RtPtrChecking,
if (OrigLoop.getParentLoop())
return false;
+ // Return true if \p CG's bounds can be expanded in the check block.
+ // VPSCEVExpander only expands AddRecs of loops enclosing the plan's scope.
+ auto BoundsAreVPlanExpandable = [&SE](const RuntimeCheckingPtrGroup &CG) {
+ return !SE.containsAddRecurrence(CG.Low) &&
+ !SE.containsAddRecurrence(CG.High);
+ };
+
ArrayRef<RuntimePointerCheck> Checks = RtPtrChecking.getChecks();
- return !Checks.empty() && all_of(Checks, [&SE](const RuntimePointerCheck &C) {
- return boundsAreVPlanExpandable(*C.first, SE) &&
- boundsAreVPlanExpandable(*C.second, SE);
+ return !Checks.empty() && all_of(Checks, [&](const RuntimePointerCheck &C) {
+ return BoundsAreVPlanExpandable(*C.first) &&
+ BoundsAreVPlanExpandable(*C.second);
});
}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index a2f2a2997086e..c49976d35f0ec 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -1541,6 +1541,14 @@ static VPSingleDefRecipe *combineRecipe(VPlan &Plan, VPSingleDefRecipe *Def) {
{X, Plan.getConstantInt(APC->getBitWidth(), APC->exactLogBase2())},
*cast<VPRecipeWithIRFlags>(Def), Def->getDebugLoc());
+ // (X >> C) << C -> X & (-1 << C).
+ if (CanCreateNewRecipe &&
+ match(Def, m_Shl(m_LShr(m_VPValue(X), m_VPValue(Y, m_APInt(APC))),
+ m_Deferred(Y))))
+ return Builder.createAnd(
+ X, Plan.getConstantInt(APInt::getAllOnes(APC->getBitWidth()) << *APC),
+ Def->getDebugLoc());
+
if (match(Def, m_Not(m_VPValue(X)))) {
// Try to fold Not into compares by adjusting the predicate in-place.
CmpPredicate Pred;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index edc2f212e2509..7d6a261518998 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -954,20 +954,6 @@ VPValue *VPSCEVExpander::expand(const SCEV *S) {
}
case scMulExpr: {
auto *MulE = cast<SCEVMulExpr>(S);
-
- // mul(PowerOf2C, udiv(X, PowerOf2C)) == (X >> C) << C -> X & (-1 << C),
- // matching SCEVExpander::visitMulExpr.
- const SCEVConstant *C1, *C2;
- const SCEV *Val;
- if (match(S, m_scev_Mul(m_SCEVConstant(C1),
- m_scev_UDiv(m_SCEV(Val), m_SCEVConstant(C2)))) &&
- C1 == C2 && C1->getAPInt().isPowerOf2()) {
- VPValue *LHS = expand(Val);
- APInt Mask = APInt::getBitsSetFrom(MulE->getType()->getScalarSizeInBits(),
- C1->getAPInt().logBase2());
- return Builder.createAnd(LHS, Builder.getPlan().getConstantInt(Mask), DL);
- }
-
VPIRFlags::WrapFlagsTy WrapFlags(MulE->hasNoUnsignedWrap(),
MulE->hasNoSignedWrap());
SmallVector<VPValue *, 2> Ops;
>From 59a5235edfdd4a61dbdbaf072a7e9485932d9949 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 22 Sep 2026 10:18:00 +0100
Subject: [PATCH 3/8] !fixup adjust to createScalarFreeze -> createFreeze
rename
---
llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 982ac67c17bfd..11d063948d568 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1515,8 +1515,8 @@ void VPlanTransforms::addMemoryRuntimeChecks(
VPValue *Start = Expander.expand(CG->Low);
VPValue *End = Expander.expand(CG->High);
if (CG->NeedsFreeze) {
- Start = Builder.createScalarFreeze(Start, DL);
- End = Builder.createScalarFreeze(End, DL);
+ Start = Builder.createFreeze(Start, DL);
+ End = Builder.createFreeze(End, DL);
}
GroupToBounds.try_emplace(CG, Start, End);
}
>From e698ce9ab8181960e7c996b88e1665bde2269c22 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 22 Sep 2026 20:58:34 +0100
Subject: [PATCH 4/8] !fixup update tests
---
.../predicated-inductions-vs-first-order-recurrences.ll | 5 ++---
1 file changed, 2 insertions(+), 3 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/predicated-inductions-vs-first-order-recurrences.ll b/llvm/test/Transforms/LoopVectorize/predicated-inductions-vs-first-order-recurrences.ll
index 0f6ba95f9aee3..4777d6ef825b2 100644
--- a/llvm/test/Transforms/LoopVectorize/predicated-inductions-vs-first-order-recurrences.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicated-inductions-vs-first-order-recurrences.ll
@@ -907,11 +907,10 @@ define void @for_and_ind_indupdate_feeds_gep_index(ptr %dst, ptr %dst2, i64 %n)
; CHECK-NEXT: br i1 [[TMP7]], label %[[SCALAR_PH]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 3
-; CHECK-NEXT: [[SMAX1:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP8:%.*]] = mul i64 [[SMAX1]], 3
+; CHECK-NEXT: [[TMP8:%.*]] = mul i64 [[TMP0]], 3
; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[TMP8]], 1
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[DST]], i64 [[TMP9]]
-; CHECK-NEXT: [[TMP10:%.*]] = mul i64 [[SMAX1]], 24
+; CHECK-NEXT: [[TMP10:%.*]] = mul i64 [[TMP0]], 24
; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[TMP10]], -16
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[DST2]], i64 [[TMP11]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP3]]
>From 4b5451ead95edecaecc89dffafa29249b615a2cd Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 22 Sep 2026 22:11:25 +0100
Subject: [PATCH 5/8] !fixup preserve branch weights for expanded selects
---
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 7 ++--
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 16 +++++++++-
.../LoopVectorize/scev-check-unknown-prof.ll | 32 +++++++++----------
3 files changed, 36 insertions(+), 19 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 38c94512f0546..efbec89c90117 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -808,8 +808,11 @@ Value *VPInstruction::generate(VPTransformState &State) {
OnlyFirstLaneUsed || vputils::isSingleScalar(getOperand(0)));
Value *Op1 = State.get(getOperand(1), OnlyFirstLaneUsed);
Value *Op2 = State.get(getOperand(2), OnlyFirstLaneUsed);
- return Builder.CreateSelectFMF(Cond, Op1, Op2, getFastMathFlagsOrNone(),
- Name);
+ Value *Sel =
+ Builder.CreateSelectFMF(Cond, Op1, Op2, getFastMathFlagsOrNone(), Name);
+ if (auto *I = dyn_cast<Instruction>(Sel))
+ applyMetadata(*I);
+ return Sel;
}
case VPInstruction::ActiveLaneMask:
case VPInstruction::WideActiveLaneMask: {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 7d6a261518998..ec301abd0b58c 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -23,6 +23,7 @@
#include "llvm/Analysis/ScalarEvolutionExpressions.h"
#include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
#include "llvm/IR/Dominators.h"
+#include "llvm/IR/MDBuilder.h"
#include "llvm/IR/ProfDataUtils.h"
#include "llvm/Transforms/Utils/ScalarEvolutionExpander.h"
@@ -1083,7 +1084,20 @@ VPValue *VPSCEVExpander::expand(const SCEV *S) {
// pointer-typed min/max as cmp + select, matching SCEVExpander.
VPValue *Cmp = Builder.createICmp(
MinMaxIntrinsic::getPredicate(IntrinsicID), Result, Op, DL);
- Result = Builder.createSelect(Cmp, Result, Op, DL);
+ VPInstruction *Sel = Builder.createSelect(Cmp, Result, Op, DL);
+ Function *F =
+ Builder.getPlan().getScalarHeader()->getIRBasicBlock()->getParent();
+ std::optional<uint64_t> EC = F->getEntryCount();
+ if (EC && *EC > 0) {
+ MDBuilder MDB(SE.getContext());
+ Sel->setMetadata(
+ LLVMContext::MD_prof,
+ MDNode::get(SE.getContext(),
+ {MDB.createString(
+ MDProfLabels::UnknownBranchWeightsMarker),
+ MDB.createString("scev-expander")}));
+ }
+ Result = Sel;
} else {
Result = Builder.createScalarIntrinsic(IntrinsicID, {Result, Op},
ResultTy, DL);
diff --git a/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll b/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll
index f2d2f13a39a6b..05ef106661e4c 100644
--- a/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll
+++ b/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll
@@ -16,12 +16,12 @@ define void @wrap_check(i32 %n, i32 %step) !prof !0 {
; CHECK: br i1 [[TMP16:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK: [[VECTOR_BODY:.*]]:
-; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP1:![0-9]+]]
+; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP2:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK: br i1 [[CMP_N:%.*]], label %[[EXIT_LOOPEXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK: [[LOOP:.*]]:
-; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[EXIT_LOOPEXIT]]:
; CHECK: [[EXIT]]:
;
@@ -61,17 +61,17 @@ define void @runtime_step_memcheck(ptr %in, ptr %out, i64 %n, i64 %step) !prof !
; CHECK: [[TMP9:%.*]] = select i1 [[TMP3]], i1 [[TMP8:%.*]], i1 [[TMP7:%.*]], !prof [[PROF1]]
; CHECK: br i1 [[TMP16:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK: [[UMIN:%.*]] = select i1 [[TMP23:%.*]], ptr [[IN]], ptr [[SCEVGEP1:%.*]], !prof [[PROF1]]
-; CHECK: [[UMAX:%.*]] = select i1 [[TMP24:%.*]], ptr [[IN]], ptr [[SCEVGEP1]], !prof [[PROF1]]
+; CHECK: [[TMP26:%.*]] = select i1 [[TMP25:%.*]], ptr [[IN]], ptr [[TMP24:%.*]], !prof [[PROF1]]
+; CHECK: [[TMP28:%.*]] = select i1 [[TMP27:%.*]], ptr [[IN]], ptr [[TMP24]], !prof [[PROF1]]
; CHECK: br i1 [[FOUND_CONFLICT:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK: [[VECTOR_BODY:.*]]:
-; CHECK: br i1 [[TMP47:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK: br i1 [[TMP52:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK: br i1 [[CMP_N:%.*]], label %[[EXIT_LOOPEXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK: [[LOOP:.*]]:
-; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP12:![0-9]+]]
; CHECK: [[EXIT_LOOPEXIT]]:
; CHECK: [[EXIT]]:
;
@@ -113,12 +113,12 @@ define void @wrap_check_not_profiled(i32 %n, i32 %step) {
; CHECK: br i1 [[TMP16:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK: [[VECTOR_BODY:.*]]:
-; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK: br i1 [[CMP_N:%.*]], label %[[EXIT_LOOPEXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK: [[LOOP:.*]]:
-; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP14:![0-9]+]]
; CHECK: [[EXIT_LOOPEXIT]]:
; CHECK: [[EXIT]]:
;
@@ -150,12 +150,12 @@ exit:
;.
; CHECK: [[PROF0]] = !{!"function_entry_count", i64 1000}
; CHECK: [[PROF1]] = !{!"unknown", !"scev-expander"}
-; CHECK: [[LOOP1]] = distinct !{[[LOOP1]], [[META2:![0-9]+]], [[META3:![0-9]+]]}
-; CHECK: [[META2]] = !{!"llvm.loop.isvectorized", i32 1}
-; CHECK: [[META3]] = !{!"llvm.loop.unroll.runtime.disable"}
-; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META2]]}
-; CHECK: [[LOOP10]] = distinct !{[[LOOP10]], [[META2]], [[META3]]}
-; CHECK: [[LOOP11]] = distinct !{[[LOOP11]], [[META2]]}
-; CHECK: [[LOOP12]] = distinct !{[[LOOP12]], [[META2]], [[META3]]}
-; CHECK: [[LOOP13]] = distinct !{[[LOOP13]], [[META2]]}
+; CHECK: [[LOOP2]] = distinct !{[[LOOP2]], [[META3:![0-9]+]], [[META4:![0-9]+]]}
+; CHECK: [[META3]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META4]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META3]]}
+; CHECK: [[LOOP11]] = distinct !{[[LOOP11]], [[META3]], [[META4]]}
+; CHECK: [[LOOP12]] = distinct !{[[LOOP12]], [[META3]]}
+; CHECK: [[LOOP13]] = distinct !{[[LOOP13]], [[META3]], [[META4]]}
+; CHECK: [[LOOP14]] = distinct !{[[LOOP14]], [[META3]]}
;.
>From 0303bb4a0d66edbdb4dd39d92094c33447d6b22d Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 23 Sep 2026 10:41:14 +0100
Subject: [PATCH 6/8] !fixup restore file header comment after merge
---
llvm/lib/Transforms/Vectorize/VPlanTransforms.h | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 3c22dcdd3cdcc..b511d9465a0c2 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -1,4 +1,4 @@
-/=- VPlanTransforms.h - Utility VPlan to VPlan transforms --------------===//
+//===- VPlanTransforms.h - Utility VPlan to VPlan transforms --------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
>From 2086f9406f5c4453a9368249114dd80ccdf25692 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 23 Sep 2026 10:41:15 +0100
Subject: [PATCH 7/8] !fixup split off pointer min/max support to follow-up
---
.../Vectorize/LoopVectorizationPlanner.h | 3 +-
.../Transforms/Vectorize/LoopVectorize.cpp | 80 ++++++++-----------
.../Vectorize/VPlanConstruction.cpp | 14 ++--
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 7 +-
.../Transforms/Vectorize/VPlanTransforms.h | 1 -
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 29 +------
.../LoopVectorize/VPlan/memory-checks.ll | 46 +++++------
.../LoopVectorize/scev-check-unknown-prof.ll | 32 ++++----
8 files changed, 87 insertions(+), 125 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 3961d53926cfe..c7f64156ca844 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -1011,7 +1011,8 @@ class LoopVectorizationPlanner {
/// Attach the runtime checks of \p RTChecks to \p Plan. Generates the memory
/// checks as recipes if \p UseVPlanMemChecks is true and they are supported.
void attachRuntimeChecks(VPlan &Plan, GeneratedRTChecks &RTChecks,
- bool HasBranchWeights, bool UseVPlanMemChecks) const;
+ bool HasBranchWeights,
+ bool UseVPlanMemChecks = true) const;
/// Update loop metadata and profile info for both the scalar remainder loop
/// and \p VectorLoop, if it exists. Keeps all loop hints from the original
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 4c930fb97308d..6ee54630a7d5c 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1817,12 +1817,38 @@ class GeneratedRTChecks {
/// Return true if any runtime checks have been added
bool hasChecks() const { return getSCEVChecks().first || HasMemChecks; }
- /// Drop the pre-built memory check block in favour of VPlan recipes.
- /// TODO: Remove once the checks can be costed in VPlan, before VF selection.
- void dropMemRuntimeChecks() {
+ /// Try to generate the memory runtime checks for \p RtPtrChecking as recipes
+ /// in \p Plan, dropping the pre-built block. Returns false if unsupported.
+ /// TODO: Remove the pre-built block once the checks can be costed in VPlan,
+ /// before VF selection.
+ bool
+ tryToAddMemRuntimeChecksToVPlan(VPlan &Plan,
+ const RuntimePointerChecking &RtPtrChecking,
+ DebugLoc DL, bool AddBranchWeights) {
assert(MemCheckBlock && pred_empty(MemCheckBlock) &&
"cannot drop memory checks that are missing or already connected");
+ // Diff checks are not modelled in VPlan yet, and the VPlan expander cannot
+ // hoist bounds out of an enclosing loop.
+ if (RtPtrChecking.getDiffChecks() || OuterLoop)
+ return false;
+
+ // VPSCEVExpander expands AddRecs in the plan's entry, not the check block,
+ // and does not support pointer-typed min/max yet.
+ ScalarEvolution &SE = *PSE.getSE();
+ auto IsPtrMinMax = [](const SCEV *S) {
+ return isa<SCEVMinMaxExpr>(S) && S->getType()->isPointerTy();
+ };
+ for (const RuntimeCheckingPtrGroup &CG : RtPtrChecking.CheckingGroups)
+ if (SE.containsAddRecurrence(CG.Low) ||
+ SE.containsAddRecurrence(CG.High) ||
+ SCEVExprContains(CG.Low, IsPtrMinMax) ||
+ SCEVExprContains(CG.High, IsPtrMinMax))
+ return false;
+
eraseMemCheckBlock();
+ RUN_VPLAN_PASS(VPlanTransforms::addMemoryRuntimeChecks, Plan,
+ RtPtrChecking.getChecks(), SE, DL, AddBranchWeights);
+ return true;
}
private:
@@ -6989,33 +7015,6 @@ void LoopVectorizationPlanner::addReductionResultComputation(
RUN_VPLAN_PASS(VPlanTransforms::clearReductionWrapFlags, *Plan);
}
-/// Return true if \p RtPtrChecking's memory checks for \p OrigLoop can be
-/// modelled as VPlan recipes.
-static bool
-canModelMemChecksInVPlan(const RuntimePointerChecking &RtPtrChecking,
- const Loop &OrigLoop, ScalarEvolution &SE) {
- // Diff checks are not modelled in VPlan yet.
- if (RtPtrChecking.getDiffChecks())
- return false;
-
- // The VPlan expander cannot hoist bounds out of an enclosing loop.
- if (OrigLoop.getParentLoop())
- return false;
-
- // Return true if \p CG's bounds can be expanded in the check block.
- // VPSCEVExpander only expands AddRecs of loops enclosing the plan's scope.
- auto BoundsAreVPlanExpandable = [&SE](const RuntimeCheckingPtrGroup &CG) {
- return !SE.containsAddRecurrence(CG.Low) &&
- !SE.containsAddRecurrence(CG.High);
- };
-
- ArrayRef<RuntimePointerCheck> Checks = RtPtrChecking.getChecks();
- return !Checks.empty() && all_of(Checks, [&](const RuntimePointerCheck &C) {
- return BoundsAreVPlanExpandable(*C.first) &&
- BoundsAreVPlanExpandable(*C.second);
- });
-}
-
void LoopVectorizationPlanner::attachRuntimeChecks(
VPlan &Plan, GeneratedRTChecks &RTChecks, bool HasBranchWeights,
bool UseVPlanMemChecks) const {
@@ -7049,17 +7048,10 @@ void LoopVectorizationPlanner::attachRuntimeChecks(
"(e.g., adding 'restrict').";
});
}
- const RuntimePointerChecking &RtPtrChecking =
- *Legal->getLAI()->getRuntimePointerChecking();
- ScalarEvolution &SE = *PSE.getSE();
- if (UseVPlanMemChecks &&
- canModelMemChecksInVPlan(RtPtrChecking, *OrigLoop, SE)) {
- RTChecks.dropMemRuntimeChecks();
- RUN_VPLAN_PASS(VPlanTransforms::addMemoryRuntimeChecks, Plan,
- RtPtrChecking.getChecks(), SE, OrigLoop->getStartLoc(),
- HasBranchWeights);
+ if (UseVPlanMemChecks && RTChecks.tryToAddMemRuntimeChecksToVPlan(
+ Plan, *Legal->getRuntimePointerChecking(),
+ OrigLoop->getStartLoc(), HasBranchWeights))
return;
- }
RUN_VPLAN_PASS(VPlanTransforms::attachCheckBlock, Plan, MemCheckCond,
MemCheckBlock, HasBranchWeights);
}
@@ -8158,9 +8150,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
// checks for the main plan.
LVP.addMinimumIterationCheck(BestMainPlan, EPI.EpilogueVF, EPI.EpilogueUF,
ElementCount::getFixed(0));
- // Epilogue vectorization has not been converted to VPlan memory checks
- // yet; it shares the checks between the main and epilogue plans via the
- // pre-built IR block.
+ // Epilogue vectorization shares the pre-built memory checks between the
+ // main and epilogue plans, so it cannot use VPlan memory checks yet.
LVP.attachRuntimeChecks(BestMainPlan, Checks, HasBranchWeights,
/*UseVPlanMemChecks=*/false);
RUN_VPLAN_PASS(
@@ -8200,8 +8191,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
BestPlan);
LVP.addMinimumIterationCheck(BestPlan, VF.Width, IC,
VF.MinProfitableTripCount);
- LVP.attachRuntimeChecks(BestPlan, Checks, HasBranchWeights,
- /*UseVPlanMemChecks=*/true);
+ LVP.attachRuntimeChecks(BestPlan, Checks, HasBranchWeights);
if (!IsInnerLoop)
LLVM_DEBUG(dbgs() << "Vectorizing outer loop in \"" << F->getName()
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 4ded471eb4c72..a69d5b710f387 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -24,7 +24,6 @@
#include "llvm/ADT/SmallVectorExtras.h"
#include "llvm/Analysis/BranchProbabilityInfo.h"
#include "llvm/Analysis/Loads.h"
-#include "llvm/Analysis/LoopAccessAnalysis.h"
#include "llvm/Analysis/LoopInfo.h"
#include "llvm/Analysis/LoopIterator.h"
#include "llvm/Analysis/OptimizationRemarkEmitter.h"
@@ -1549,7 +1548,6 @@ void VPlanTransforms::addMemoryRuntimeChecks(
auto *MemCheckVPBB = Plan.createVPBasicBlock("vector.memcheck");
VPBuilder Builder(MemCheckVPBB);
- insertCheckBlockBeforeVectorLoop(Plan, MemCheckVPBB);
VPSCEVExpander Expander(Builder, SE, DL);
// Expand each group's bounds once and up front.
@@ -1569,20 +1567,19 @@ void VPlanTransforms::addMemoryRuntimeChecks(
GroupToBounds.try_emplace(CG, Start, End);
}
- VPValue *Cond = nullptr;
+ VPValue *Cond = Plan.getFalse();
for (const auto &[A, B] : Checks) {
- auto [AStart, AEnd] = GroupToBounds.at(A);
- auto [BStart, BEnd] = GroupToBounds.at(B);
+ auto [AStart, AEnd] = GroupToBounds[A];
+ auto [BStart, BEnd] = GroupToBounds[B];
VPValue *Bound0 =
Builder.createICmp(CmpInst::ICMP_ULT, AStart, BEnd, DL, "bound0");
VPValue *Bound1 =
Builder.createICmp(CmpInst::ICMP_ULT, BStart, AEnd, DL, "bound1");
VPValue *IsConflict =
Builder.createAnd(Bound0, Bound1, DL, "found.conflict");
- Cond = Cond ? Builder.createOr(Cond, IsConflict, DL, "conflict.rdx")
- : IsConflict;
+ Cond = Builder.createOr(Cond, IsConflict, DL, "conflict.rdx");
}
- addBypassBranch(Plan, MemCheckVPBB, Cond, AddBranchWeights);
+ attachVPCheckBlock(Plan, Cond, MemCheckVPBB, AddBranchWeights);
}
void VPlanTransforms::addMinimumIterationCheck(
@@ -1591,6 +1588,7 @@ void VPlanTransforms::addMinimumIterationCheck(
bool TailFolded, Loop *OrigLoop, const uint32_t *MinItersBypassWeights,
DebugLoc DL, PredicatedScalarEvolution &PSE, VPBasicBlock *CheckBlock) {
// Generate code to check if the loop's trip count is less than VF * UF, or
+ // equal to it in case a scalar epilogue is required; this implies that the
// vector trip count is zero. This check also covers the case where adding one
// to the backedge-taken count overflowed leading to an incorrect trip count
// of zero. In this case we will also jump to the scalar loop.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index efbec89c90117..38c94512f0546 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -808,11 +808,8 @@ Value *VPInstruction::generate(VPTransformState &State) {
OnlyFirstLaneUsed || vputils::isSingleScalar(getOperand(0)));
Value *Op1 = State.get(getOperand(1), OnlyFirstLaneUsed);
Value *Op2 = State.get(getOperand(2), OnlyFirstLaneUsed);
- Value *Sel =
- Builder.CreateSelectFMF(Cond, Op1, Op2, getFastMathFlagsOrNone(), Name);
- if (auto *I = dyn_cast<Instruction>(Sel))
- applyMetadata(*I);
- return Sel;
+ return Builder.CreateSelectFMF(Cond, Op1, Op2, getFastMathFlagsOrNone(),
+ Name);
}
case VPInstruction::ActiveLaneMask:
case VPInstruction::WideActiveLaneMask: {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index b511d9465a0c2..09c44749915b0 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -17,7 +17,6 @@
#include "VPlanVerifier.h"
#include "llvm/ADT/STLFunctionalExtras.h"
#include "llvm/ADT/ScopeExit.h"
-#include "llvm/Analysis/LoopAccessAnalysis.h"
#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/Support/CommandLine.h"
#include "llvm/Support/Compiler.h"
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index ec301abd0b58c..9a07697f66762 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -23,7 +23,6 @@
#include "llvm/Analysis/ScalarEvolutionExpressions.h"
#include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
#include "llvm/IR/Dominators.h"
-#include "llvm/IR/MDBuilder.h"
#include "llvm/IR/ProfDataUtils.h"
#include "llvm/Transforms/Utils/ScalarEvolutionExpander.h"
@@ -1078,31 +1077,9 @@ VPValue *VPSCEVExpander::expand(const SCEV *S) {
Ops.push_back(OpV);
}
VPValue *Result = Ops.front();
- for (VPValue *Op : drop_begin(Ops)) {
- if (ResultTy->isPointerTy()) {
- // The min/max intrinsics don't support pointer operands, so expand
- // pointer-typed min/max as cmp + select, matching SCEVExpander.
- VPValue *Cmp = Builder.createICmp(
- MinMaxIntrinsic::getPredicate(IntrinsicID), Result, Op, DL);
- VPInstruction *Sel = Builder.createSelect(Cmp, Result, Op, DL);
- Function *F =
- Builder.getPlan().getScalarHeader()->getIRBasicBlock()->getParent();
- std::optional<uint64_t> EC = F->getEntryCount();
- if (EC && *EC > 0) {
- MDBuilder MDB(SE.getContext());
- Sel->setMetadata(
- LLVMContext::MD_prof,
- MDNode::get(SE.getContext(),
- {MDB.createString(
- MDProfLabels::UnknownBranchWeightsMarker),
- MDB.createString("scev-expander")}));
- }
- Result = Sel;
- } else {
- Result = Builder.createScalarIntrinsic(IntrinsicID, {Result, Op},
- ResultTy, DL);
- }
- }
+ for (VPValue *Op : drop_begin(Ops))
+ Result = Builder.createScalarIntrinsic(IntrinsicID, {Result, Op},
+ ResultTy, DL);
return Result;
}
case scAddRecExpr: {
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll b/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
index f52a8edeacef9..2389a5434dfe1 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
@@ -67,33 +67,33 @@ define void @ptr_minmax_bounds(ptr %a, ptr %b, i64 %n, i64 %s, i64 %t) {
; CHECK-NEXT: IR %step = mul i64 %s, %t
; CHECK-NEXT: EMIT vp<%min.iters.check> = icmp ult ir<%n>, ir<4>
; CHECK-NEXT: EMIT branch-on-cond vp<%min.iters.check>
-; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, vector.memcheck
+; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, ir-bb<vector.memcheck>
; CHECK-EMPTY:
-; CHECK-NEXT: vector.memcheck:
-; CHECK-NEXT: EMIT vp<[[VP3:%[0-9]+]]> = shl ir<%n>, ir<2>
-; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = ptradd ir<%b>, vp<[[VP3]]>
-; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = add ir<%n>, ir<-1>
-; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = mul ir<%t>, ir<%s>
-; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = mul vp<[[VP6]]>, vp<[[VP5]]>
-; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = shl vp<[[VP7]]>, ir<2>
-; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = ptradd ir<%a>, vp<[[VP8]]>
-; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = icmp ult ir<%a>, vp<[[VP9]]>
-; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = select vp<[[VP10]]>, ir<%a>, vp<[[VP9]]>
-; CHECK-NEXT: EMIT vp<[[VP12:%[0-9]+]]> = icmp ugt ir<%a>, vp<[[VP9]]>
-; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = select vp<[[VP12]]>, ir<%a>, vp<[[VP9]]>
-; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = ptradd vp<[[VP13]]>, ir<4>
-; CHECK-NEXT: EMIT vp<%bound0> = icmp ult ir<%b>, vp<[[VP14]]>
-; CHECK-NEXT: EMIT vp<%bound1> = icmp ult vp<[[VP11]]>, vp<[[VP4]]>
-; CHECK-NEXT: EMIT vp<%found.conflict> = and vp<%bound0>, vp<%bound1>
-; CHECK-NEXT: EMIT branch-on-cond vp<%found.conflict>
+; CHECK-NEXT: ir-bb<vector.memcheck>:
+; CHECK-NEXT: IR %0 = shl i64 %n, 2
+; CHECK-NEXT: IR %scevgep = getelementptr i8, ptr %b, i64 %0
+; CHECK-NEXT: IR %1 = mul i64 %t, %s
+; CHECK-NEXT: IR %2 = add i64 %n, -1
+; CHECK-NEXT: IR %3 = mul i64 %1, %2
+; CHECK-NEXT: IR %4 = shl i64 %3, 2
+; CHECK-NEXT: IR %scevgep1 = getelementptr i8, ptr %a, i64 %4
+; CHECK-NEXT: IR %5 = icmp ult ptr %a, %scevgep1
+; CHECK-NEXT: IR %umin = select i1 %5, ptr %a, ptr %scevgep1
+; CHECK-NEXT: IR %6 = icmp ugt ptr %a, %scevgep1
+; CHECK-NEXT: IR %umax = select i1 %6, ptr %a, ptr %scevgep1
+; CHECK-NEXT: IR %scevgep2 = getelementptr i8, ptr %umax, i64 4
+; CHECK-NEXT: IR %bound0 = icmp ult ptr %b, %scevgep2
+; CHECK-NEXT: IR %bound1 = icmp ult ptr %umin, %scevgep
+; CHECK-NEXT: IR %found.conflict = and i1 %bound0, %bound1
+; CHECK-NEXT: EMIT branch-on-cond ir<%found.conflict>
; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, vector.ph
; CHECK-EMPTY:
; CHECK-NEXT: vector.ph:
-; CHECK-NEXT: EMIT vp<[[VP16:%[0-9]+]]> = and ir<%n>, ir<3>
-; CHECK-NEXT: EMIT vp<%n.vec> = sub ir<%n>, vp<[[VP16]]>
-; CHECK-NEXT: EMIT vp<[[VP17:%[0-9]+]]> = broadcast ir<%step>
-; CHECK-NEXT: EMIT vp<[[VP18:%[0-9]+]]> = step-vector i64
-; CHECK-NEXT: EMIT vp<[[VP19:%[0-9]+]]> = broadcast ir<4>
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = and ir<%n>, ir<3>
+; CHECK-NEXT: EMIT vp<%n.vec> = sub ir<%n>, vp<[[VP4]]>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = broadcast ir<%step>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = step-vector i64
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = broadcast ir<4>
; CHECK-NEXT: Successor(s): vector.body
; CHECK-EMPTY:
; CHECK-NEXT: vector.body:
diff --git a/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll b/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll
index 05ef106661e4c..f2d2f13a39a6b 100644
--- a/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll
+++ b/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll
@@ -16,12 +16,12 @@ define void @wrap_check(i32 %n, i32 %step) !prof !0 {
; CHECK: br i1 [[TMP16:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK: [[VECTOR_BODY:.*]]:
-; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP2:![0-9]+]]
+; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP1:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK: br i1 [[CMP_N:%.*]], label %[[EXIT_LOOPEXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK: [[LOOP:.*]]:
-; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[EXIT_LOOPEXIT]]:
; CHECK: [[EXIT]]:
;
@@ -61,17 +61,17 @@ define void @runtime_step_memcheck(ptr %in, ptr %out, i64 %n, i64 %step) !prof !
; CHECK: [[TMP9:%.*]] = select i1 [[TMP3]], i1 [[TMP8:%.*]], i1 [[TMP7:%.*]], !prof [[PROF1]]
; CHECK: br i1 [[TMP16:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK: [[TMP26:%.*]] = select i1 [[TMP25:%.*]], ptr [[IN]], ptr [[TMP24:%.*]], !prof [[PROF1]]
-; CHECK: [[TMP28:%.*]] = select i1 [[TMP27:%.*]], ptr [[IN]], ptr [[TMP24]], !prof [[PROF1]]
+; CHECK: [[UMIN:%.*]] = select i1 [[TMP23:%.*]], ptr [[IN]], ptr [[SCEVGEP1:%.*]], !prof [[PROF1]]
+; CHECK: [[UMAX:%.*]] = select i1 [[TMP24:%.*]], ptr [[IN]], ptr [[SCEVGEP1]], !prof [[PROF1]]
; CHECK: br i1 [[FOUND_CONFLICT:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK: [[VECTOR_BODY:.*]]:
-; CHECK: br i1 [[TMP52:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK: br i1 [[TMP47:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK: br i1 [[CMP_N:%.*]], label %[[EXIT_LOOPEXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK: [[LOOP:.*]]:
-; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP11:![0-9]+]]
; CHECK: [[EXIT_LOOPEXIT]]:
; CHECK: [[EXIT]]:
;
@@ -113,12 +113,12 @@ define void @wrap_check_not_profiled(i32 %n, i32 %step) {
; CHECK: br i1 [[TMP16:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK: [[VECTOR_BODY:.*]]:
-; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK: br i1 [[CMP_N:%.*]], label %[[EXIT_LOOPEXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK: [[LOOP:.*]]:
-; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP13:![0-9]+]]
; CHECK: [[EXIT_LOOPEXIT]]:
; CHECK: [[EXIT]]:
;
@@ -150,12 +150,12 @@ exit:
;.
; CHECK: [[PROF0]] = !{!"function_entry_count", i64 1000}
; CHECK: [[PROF1]] = !{!"unknown", !"scev-expander"}
-; CHECK: [[LOOP2]] = distinct !{[[LOOP2]], [[META3:![0-9]+]], [[META4:![0-9]+]]}
-; CHECK: [[META3]] = !{!"llvm.loop.isvectorized", i32 1}
-; CHECK: [[META4]] = !{!"llvm.loop.unroll.runtime.disable"}
-; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META3]]}
-; CHECK: [[LOOP11]] = distinct !{[[LOOP11]], [[META3]], [[META4]]}
-; CHECK: [[LOOP12]] = distinct !{[[LOOP12]], [[META3]]}
-; CHECK: [[LOOP13]] = distinct !{[[LOOP13]], [[META3]], [[META4]]}
-; CHECK: [[LOOP14]] = distinct !{[[LOOP14]], [[META3]]}
+; CHECK: [[LOOP1]] = distinct !{[[LOOP1]], [[META2:![0-9]+]], [[META3:![0-9]+]]}
+; CHECK: [[META2]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META3]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META2]]}
+; CHECK: [[LOOP10]] = distinct !{[[LOOP10]], [[META2]], [[META3]]}
+; CHECK: [[LOOP11]] = distinct !{[[LOOP11]], [[META2]]}
+; CHECK: [[LOOP12]] = distinct !{[[LOOP12]], [[META2]], [[META3]]}
+; CHECK: [[LOOP13]] = distinct !{[[LOOP13]], [[META2]]}
;.
>From 0474bbc83543376f6d0604821b0ceca2c188bb0d Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Mon, 28 Sep 2026 16:26:24 +0100
Subject: [PATCH 8/8] !fixup address comments, thanks
---
.../Vectorize/LoopVectorizationPlanner.h | 6 +-
.../Transforms/Vectorize/LoopVectorize.cpp | 91 +++++++------------
.../Vectorize/VPlanConstruction.cpp | 19 ++--
.../Transforms/Vectorize/VPlanTransforms.h | 15 ++-
.../LoopVectorize/AArch64/induction-costs.ll | 2 +-
.../invariant-store-vectorization.ll | 6 +-
.../test/Transforms/LoopVectorize/metadata.ll | 8 +-
.../pointer-select-runtime-checks.ll | 24 ++---
.../Transforms/LoopVectorize/runtime-check.ll | 4 +-
9 files changed, 74 insertions(+), 101 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 4b49770967fec..3c07a6e159656 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -1022,11 +1022,9 @@ class LoopVectorizationPlanner {
void addMinimumIterationCheck(VPlan &Plan, ElementCount VF, unsigned UF,
ElementCount MinProfitableTripCount) const;
- /// Attach the runtime checks of \p RTChecks to \p Plan. Generates the memory
- /// checks as recipes if \p UseVPlanMemChecks is true and they are supported.
+ /// Attach the runtime checks of \p RTChecks to \p Plan.
void attachRuntimeChecks(VPlan &Plan, GeneratedRTChecks &RTChecks,
- bool HasBranchWeights,
- bool UseVPlanMemChecks = true) const;
+ bool HasBranchWeights) const;
/// Update loop metadata and profile info for both the scalar remainder loop
/// and \p VectorLoop, if it exists. Keeps all loop hints from the original
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ba1d1da6dd3bf..7b0e80390379f 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1523,8 +1523,8 @@ namespace {
///
/// The runtime checks are created up-front in temporary blocks to allow better
/// estimating the cost and un-linked from the existing IR. After deciding to
-/// vectorize, the checks are moved back. If deciding not to vectorize, the
-/// temporary blocks are completely removed.
+/// vectorize, the checks are attached to VPlan as IR or recipes. If deciding
+/// not to vectorize, the temporary blocks are completely removed.
class GeneratedRTChecks {
/// Basic block which contains the generated SCEV checks, if any.
BasicBlock *SCEVCheckBlock = nullptr;
@@ -1540,9 +1540,9 @@ class GeneratedRTChecks {
/// If it is nullptr no memory runtime checks have been generated.
Value *MemRuntimeCheckCond = nullptr;
- /// Set in create() if memory checks were generated and did not fold away.
- /// Unlike MemRuntimeCheckCond, stays set when the pre-built block is dropped.
- bool HasMemChecks = false;
+ /// Whether checks were generated, retained after their IR is replaced or
+ /// removed during VPlan execution.
+ bool HasChecks = false;
DominatorTree *DT;
LoopInfo *LI;
@@ -1576,9 +1576,8 @@ class GeneratedRTChecks {
/// Generate runtime checks in SCEVCheckBlock and MemCheckBlock, so we can
/// accurately estimate the cost of the runtime checks. The blocks are
- /// un-linked from the IR and are added back during vector code generation. If
- /// there is no vector code generation, the check blocks are removed
- /// completely.
+ /// un-linked from the IR and attached to VPlan as IR or recipes if
+ /// profitable. Otherwise, the check blocks are removed completely.
void create(Loop *L, const LoopAccessInfo &LAI,
const SCEVPredicate &UnionPred, ElementCount VF, unsigned IC,
OptimizationRemarkEmitter &ORE) {
@@ -1645,10 +1644,10 @@ class GeneratedRTChecks {
assert(MemRuntimeCheckCond &&
"no RT checks generated although RtPtrChecking "
"claimed checks are required");
- HasMemChecks = getMemRuntimeChecks().first != nullptr;
}
SCEVExp.eraseDeadInstructions(SCEVCheckCond);
+ HasChecks = getSCEVChecks().first || getMemRuntimeChecks().first;
if (!MemCheckBlock && !SCEVCheckBlock)
return;
@@ -1805,43 +1804,8 @@ class GeneratedRTChecks {
}
/// Return true if any runtime checks have been added
- bool hasChecks() const { return getSCEVChecks().first || HasMemChecks; }
-
- /// Try to generate the memory runtime checks for \p RtPtrChecking as recipes
- /// in \p Plan, dropping the pre-built block. Returns false if unsupported.
- /// TODO: Remove the pre-built block once the checks can be costed in VPlan,
- /// before VF selection.
- bool
- tryToAddMemRuntimeChecksToVPlan(VPlan &Plan,
- const RuntimePointerChecking &RtPtrChecking,
- DebugLoc DL, bool AddBranchWeights) {
- assert(MemCheckBlock && pred_empty(MemCheckBlock) &&
- "cannot drop memory checks that are missing or already connected");
- // Diff checks are not modelled in VPlan yet, and the VPlan expander cannot
- // hoist bounds out of an enclosing loop.
- if (RtPtrChecking.getDiffChecks() || OuterLoop)
- return false;
-
- // VPSCEVExpander expands AddRecs in the plan's entry, not the check block,
- // and does not support pointer-typed min/max yet.
- ScalarEvolution &SE = *PSE.getSE();
- auto IsPtrMinMax = [](const SCEV *S) {
- return isa<SCEVMinMaxExpr>(S) && S->getType()->isPointerTy();
- };
- for (const RuntimeCheckingPtrGroup &CG : RtPtrChecking.CheckingGroups)
- if (SE.containsAddRecurrence(CG.Low) ||
- SE.containsAddRecurrence(CG.High) ||
- SCEVExprContains(CG.Low, IsPtrMinMax) ||
- SCEVExprContains(CG.High, IsPtrMinMax))
- return false;
-
- eraseMemCheckBlock();
- RUN_VPLAN_PASS(VPlanTransforms::addMemoryRuntimeChecks, Plan,
- RtPtrChecking.getChecks(), SE, DL, AddBranchWeights);
- return true;
- }
+ bool hasChecks() const { return HasChecks; }
-private:
/// Erase the memory check block, its instructions and their SCEV expansions.
void eraseMemCheckBlock() {
SCEVExpanderCleaner MemCheckCleaner(MemCheckExp);
@@ -7037,8 +7001,7 @@ void LoopVectorizationPlanner::addReductionResultComputation(
}
void LoopVectorizationPlanner::attachRuntimeChecks(
- VPlan &Plan, GeneratedRTChecks &RTChecks, bool HasBranchWeights,
- bool UseVPlanMemChecks) const {
+ VPlan &Plan, GeneratedRTChecks &RTChecks, bool HasBranchWeights) const {
const auto &[SCEVCheckCond, SCEVCheckBlock] = RTChecks.getSCEVChecks();
if (SCEVCheckBlock && SCEVCheckBlock->hasNPredecessors(0)) {
assert((!Config.OptForSize ||
@@ -7069,12 +7032,29 @@ void LoopVectorizationPlanner::attachRuntimeChecks(
"(e.g., adding 'restrict').";
});
}
- if (UseVPlanMemChecks && RTChecks.tryToAddMemRuntimeChecksToVPlan(
- Plan, *Legal->getRuntimePointerChecking(),
- OrigLoop->getStartLoc(), HasBranchWeights))
- return;
- RUN_VPLAN_PASS(VPlanTransforms::attachCheckBlock, Plan, MemCheckCond,
- MemCheckBlock, HasBranchWeights);
+ // VPSCEVExpander expands AddRecs in the plan's entry, not the check block,
+ // and does not support pointer-typed min/max yet.
+ auto IsUnsupported = [](const SCEV *S) {
+ return isa<SCEVAddRecExpr>(S) ||
+ (isa<SCEVMinMaxExpr>(S) && S->getType()->isPointerTy());
+ };
+ // Diff checks are not modelled in VPlan yet, and the VPlan expander cannot
+ // hoist bounds out of an enclosing loop.
+ const auto &RtPtrChecking = *Legal->getRuntimePointerChecking();
+ if (RtPtrChecking.getDiffChecks() || OrigLoop->getParentLoop() ||
+ any_of(RtPtrChecking.CheckingGroups,
+ [&](const RuntimeCheckingPtrGroup &CG) {
+ return SCEVExprContains(CG.Low, IsUnsupported) ||
+ SCEVExprContains(CG.High, IsUnsupported);
+ }))
+ return RUN_VPLAN_PASS(VPlanTransforms::attachCheckBlock, Plan,
+ MemCheckCond, MemCheckBlock, HasBranchWeights);
+
+ // Erase the temporary IR before recipe expansion can reuse its values.
+ RTChecks.eraseMemCheckBlock();
+ RUN_VPLAN_PASS(VPlanTransforms::attachMemoryChecks, Plan,
+ RtPtrChecking.getChecks(), *PSE.getSE(),
+ OrigLoop->getStartLoc(), HasBranchWeights);
}
}
@@ -8173,10 +8153,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
// checks for the main plan.
LVP.addMinimumIterationCheck(BestMainPlan, EPI.EpilogueVF, /*UF=*/1,
ElementCount::getFixed(0));
- // Epilogue vectorization shares the pre-built memory checks between the
- // main and epilogue plans, so it cannot use VPlan memory checks yet.
- LVP.attachRuntimeChecks(BestMainPlan, Checks, HasBranchWeights,
- /*UseVPlanMemChecks=*/false);
+ LVP.attachRuntimeChecks(BestMainPlan, Checks, HasBranchWeights);
RUN_VPLAN_PASS(
VPlanTransforms::addIterationCountCheckBlock, BestMainPlan,
EPI.MainLoopVF, EPI.MainLoopUF, BestMainPlan.requiresScalarEpilogue(),
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 5da720dbde5d0..d4700879d43a1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1538,30 +1538,31 @@ void VPlanTransforms::attachCheckBlock(VPlan &Plan, Value *Cond,
attachVPCheckBlock(Plan, CondVPV, CheckBlockVPBB, AddBranchWeights);
}
-void VPlanTransforms::addMemoryRuntimeChecks(
- VPlan &Plan, ArrayRef<RuntimePointerCheck> Checks, ScalarEvolution &SE,
- DebugLoc DL, bool AddBranchWeights) {
- assert(!Checks.empty() && "no checks to replace the pre-built block with");
+void VPlanTransforms::attachMemoryChecks(VPlan &Plan,
+ ArrayRef<RuntimePointerCheck> Checks,
+ ScalarEvolution &SE, DebugLoc DL,
+ bool AddBranchWeights) {
+ assert(!Checks.empty() && "no checks to generate");
auto *MemCheckVPBB = Plan.createVPBasicBlock("vector.memcheck");
VPBuilder Builder(MemCheckVPBB);
VPSCEVExpander Expander(Builder, SE, DL);
- // Expand each group's bounds once and up front.
+ // Expand each group's bounds once so all checks reuse the same frozen values.
SmallDenseMap<const RuntimeCheckingPtrGroup *,
std::pair<VPValue *, VPValue *>>
GroupToBounds;
for (const auto &[A, B] : Checks)
for (const RuntimeCheckingPtrGroup *CG : {A, B}) {
- if (GroupToBounds.contains(CG))
+ auto &[Start, End] = GroupToBounds[CG];
+ if (Start)
continue;
- VPValue *Start = Expander.expand(CG->Low);
- VPValue *End = Expander.expand(CG->High);
+ Start = Expander.expand(CG->Low);
+ End = Expander.expand(CG->High);
if (CG->NeedsFreeze) {
Start = Builder.createFreeze(Start, DL);
End = Builder.createFreeze(End, DL);
}
- GroupToBounds.try_emplace(CG, Start, End);
}
VPValue *Cond = Plan.getFalse();
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 36d41eb242f94..0469324023f52 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -226,21 +226,18 @@ struct VPlanTransforms {
/// BranchOnCond with BranchOnCount, using \p DL for the canonical IV.
LLVM_ABI_FOR_TEST static void createLoopRegions(VPlan &Plan, DebugLoc DL);
- /// Wrap runtime check block \p CheckBlock in a VPIRBB and \p Cond in a
- /// VPValue and connect the block to \p Plan, using the VPValue as branch
- /// condition.
+ /// Connect \p CheckBlock to \p Plan, branching on \p Cond.
static void attachVPCheckBlock(VPlan &Plan, VPValue *Cond,
VPBasicBlock *CheckBlock,
bool AddBranchWeights);
static void attachCheckBlock(VPlan &Plan, Value *Cond, BasicBlock *CheckBlock,
bool AddBranchWeights);
- /// Generate recipes for the memory runtime checks \p Checks in a new block
- /// added to \p Plan.
- static void addMemoryRuntimeChecks(VPlan &Plan,
- ArrayRef<RuntimePointerCheck> Checks,
- ScalarEvolution &SE, DebugLoc DL,
- bool AddBranchWeights);
+ /// Generate \p Checks as recipes and attach the check block to \p Plan.
+ static void attachMemoryChecks(VPlan &Plan,
+ ArrayRef<RuntimePointerCheck> Checks,
+ ScalarEvolution &SE, DebugLoc DL,
+ bool AddBranchWeights);
/// Model the blocks the executed \p MainPlan generated for the main vector
/// loop in \p EpiPlan during epilogue vectorization, wrapping each in a
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs.ll
index 0817c7622641e..38e50c8614eb3 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs.ll
@@ -558,9 +558,9 @@ define void at sext_sub_nsw_for_address(ptr %base, i64 %n, ptr %src) #0 {
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP10:%.*]] = getelementptr i8, ptr [[SRC]], i64 -8
; CHECK-NEXT: [[TMP21:%.*]] = shl i64 [[N]], 4
-; CHECK-NEXT: [[TMP22:%.*]] = add i64 [[TMP21]], 8
; CHECK-NEXT: [[SMIN11:%.*]] = call i64 @llvm.smin.i64(i64 [[N]], i64 0)
; CHECK-NEXT: [[TMP23:%.*]] = shl i64 [[SMIN11]], 4
+; CHECK-NEXT: [[TMP22:%.*]] = add i64 [[TMP21]], 8
; CHECK-NEXT: [[TMP24:%.*]] = sub i64 [[TMP22]], [[TMP23]]
; CHECK-NEXT: [[SCEVGEP12:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP24]]
; CHECK-NEXT: [[TMP25:%.*]] = sub i64 [[TMP23]], [[TMP21]]
diff --git a/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll b/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll
index 638b3780c8adc..1e65b4ff53966 100644
--- a/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll
+++ b/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll
@@ -352,8 +352,8 @@ define void @multiple_uniform_stores(ptr nocapture %var1, ptr nocapture readonly
; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[TMP2]], 4
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[VAR1:%.*]], i64 [[TMP3]]
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[VAR2:%.*]], i64 4
-; CHECK-NEXT: [[TMP10:%.*]] = add i32 [[ITR]], -1
; CHECK-NEXT: [[TMP5:%.*]] = add i32 [[ITR]], -1
+; CHECK-NEXT: [[TMP10:%.*]] = add i32 [[ITR]], -1
; CHECK-NEXT: br label [[FOR_COND1_PREHEADER:%.*]]
; CHECK: for.cond1.preheader:
; CHECK-NEXT: [[INDVARS_IV23:%.*]] = phi i64 [ [[INDVARS_IV_NEXT24:%.*]], [[FOR_INC8:%.*]] ], [ 0, [[FOR_COND1_PREHEADER_PREHEADER]] ]
@@ -364,7 +364,7 @@ define void @multiple_uniform_stores(ptr nocapture %var1, ptr nocapture readonly
; CHECK-NEXT: [[ARRAYIDX5:%.*]] = getelementptr inbounds i32, ptr [[VAR1]], i64 [[INDVARS_IV23]]
; CHECK-NEXT: [[TMP4:%.*]] = zext i32 [[J_022]] to i64
; CHECK-NEXT: [[ARRAYIDX5_PROMOTED:%.*]] = load i32, ptr [[ARRAYIDX5]], align 4
-; CHECK-NEXT: [[TMP6:%.*]] = sub i32 [[TMP5]], [[J_022]]
+; CHECK-NEXT: [[TMP6:%.*]] = sub i32 [[TMP10]], [[J_022]]
; CHECK-NEXT: [[TMP7:%.*]] = zext i32 [[TMP6]] to i64
; CHECK-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 1
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP8]], 4
@@ -372,7 +372,7 @@ define void @multiple_uniform_stores(ptr nocapture %var1, ptr nocapture readonly
; CHECK: vector.memcheck:
; CHECK-NEXT: [[TMP9:%.*]] = shl nuw nsw i64 [[TMP4]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[VAR2]], i64 [[TMP9]]
-; CHECK-NEXT: [[TMP11:%.*]] = sub i32 [[TMP10]], [[J_022]]
+; CHECK-NEXT: [[TMP11:%.*]] = sub i32 [[TMP5]], [[J_022]]
; CHECK-NEXT: [[TMP12:%.*]] = zext i32 [[TMP11]] to i64
; CHECK-NEXT: [[TMP13:%.*]] = shl nuw nsw i64 [[TMP12]], 2
; CHECK-NEXT: [[TMP14:%.*]] = add i64 [[TMP9]], [[TMP13]]
diff --git a/llvm/test/Transforms/LoopVectorize/metadata.ll b/llvm/test/Transforms/LoopVectorize/metadata.ll
index 5421944f9f376..6fa9c66514381 100644
--- a/llvm/test/Transforms/LoopVectorize/metadata.ll
+++ b/llvm/test/Transforms/LoopVectorize/metadata.ll
@@ -510,8 +510,8 @@ define void @noalias_metadata(ptr align 8 %dst, ptr align 8 %src) {
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 8
-; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[DST1]], 8
-; CHECK-NEXT: [[TMP4:%.*]] = sub i64 [[TMP11]], [[SRC2]]
+; CHECK-NEXT: [[TMP3:%.*]] = add i64 [[DST1]], 8
+; CHECK-NEXT: [[TMP4:%.*]] = sub i64 [[TMP3]], [[SRC2]]
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP4]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[DST]], [[SCEVGEP3]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[SRC]], [[SCEVGEP]]
@@ -561,8 +561,8 @@ define void @noalias_metadata(ptr align 8 %dst, ptr align 8 %src) {
; INTERLEAVE-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; INTERLEAVE: [[VECTOR_MEMCHECK]]:
; INTERLEAVE-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 8
-; INTERLEAVE-NEXT: [[TMP12:%.*]] = add i64 [[DST1]], 8
-; INTERLEAVE-NEXT: [[TMP4:%.*]] = sub i64 [[TMP12]], [[SRC2]]
+; INTERLEAVE-NEXT: [[TMP3:%.*]] = add i64 [[DST1]], 8
+; INTERLEAVE-NEXT: [[TMP4:%.*]] = sub i64 [[TMP3]], [[SRC2]]
; INTERLEAVE-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP4]]
; INTERLEAVE-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[DST]], [[SCEVGEP3]]
; INTERLEAVE-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[SRC]], [[SCEVGEP]]
diff --git a/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll b/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll
index afd4fc57c4642..f3efb2c7ab0fe 100644
--- a/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll
+++ b/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll
@@ -195,12 +195,12 @@ define void @test_loop_dependent_select2(ptr %src.1, ptr %src.2, ptr %dst, i8 %n
; CHECK-NEXT: [[TMP10:%.*]] = icmp ult i8 [[TMP8]], [[X]]
; CHECK-NEXT: [[TMP11:%.*]] = select i1 [[TMP9]], ptr [[SRC_1]], ptr [[SRC_2]]
; CHECK-NEXT: [[TMP12:%.*]] = select i1 [[TMP10]], ptr [[SRC_1]], ptr [[SRC_2]]
-; CHECK-NEXT: [[TMP20:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META11:![0-9]+]]
-; CHECK-NEXT: [[TMP21:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META11]]
+; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META11:![0-9]+]]
+; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META11]]
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP7]]
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP8]]
-; CHECK-NEXT: store i8 [[TMP20]], ptr [[TMP15]], align 2, !alias.scope [[META14:![0-9]+]], !noalias [[META16:![0-9]+]]
-; CHECK-NEXT: store i8 [[TMP21]], ptr [[TMP16]], align 2, !alias.scope [[META14]], !noalias [[META16]]
+; CHECK-NEXT: store i8 [[TMP14]], ptr [[TMP15]], align 2, !alias.scope [[META14:![0-9]+]], !noalias [[META16:![0-9]+]]
+; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP16]], align 2, !alias.scope [[META14]], !noalias [[META16]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 2
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP17]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP18:![0-9]+]]
@@ -278,12 +278,12 @@ define void @test_loop_dependent_select_first_ptr_noundef(ptr noundef %src.1, pt
; CHECK-NEXT: [[TMP10:%.*]] = icmp ult i8 [[TMP8]], [[X]]
; CHECK-NEXT: [[TMP11:%.*]] = select i1 [[TMP9]], ptr [[SRC_1]], ptr [[SRC_2]]
; CHECK-NEXT: [[TMP12:%.*]] = select i1 [[TMP10]], ptr [[SRC_1]], ptr [[SRC_2]]
-; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META20:![0-9]+]]
-; CHECK-NEXT: [[TMP19:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META20]]
+; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META20:![0-9]+]]
+; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META20]]
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP7]]
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP8]]
-; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP15]], align 2, !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
-; CHECK-NEXT: store i8 [[TMP19]], ptr [[TMP16]], align 2, !alias.scope [[META23]], !noalias [[META25]]
+; CHECK-NEXT: store i8 [[TMP14]], ptr [[TMP15]], align 2, !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
+; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP16]], align 2, !alias.scope [[META23]], !noalias [[META25]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 2
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP17]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP27:![0-9]+]]
@@ -361,12 +361,12 @@ define void @test_loop_dependent_select_second_ptr_noundef(ptr %src.1, ptr nound
; CHECK-NEXT: [[TMP10:%.*]] = icmp ult i8 [[TMP8]], [[X]]
; CHECK-NEXT: [[TMP11:%.*]] = select i1 [[TMP9]], ptr [[SRC_1]], ptr [[SRC_2]]
; CHECK-NEXT: [[TMP12:%.*]] = select i1 [[TMP10]], ptr [[SRC_1]], ptr [[SRC_2]]
-; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META29:![0-9]+]]
-; CHECK-NEXT: [[TMP19:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META29]]
+; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META29:![0-9]+]]
+; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META29]]
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP7]]
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP8]]
-; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP15]], align 2, !alias.scope [[META32:![0-9]+]], !noalias [[META34:![0-9]+]]
-; CHECK-NEXT: store i8 [[TMP19]], ptr [[TMP16]], align 2, !alias.scope [[META32]], !noalias [[META34]]
+; CHECK-NEXT: store i8 [[TMP14]], ptr [[TMP15]], align 2, !alias.scope [[META32:![0-9]+]], !noalias [[META34:![0-9]+]]
+; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP16]], align 2, !alias.scope [[META32]], !noalias [[META34]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 2
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP17]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
diff --git a/llvm/test/Transforms/LoopVectorize/runtime-check.ll b/llvm/test/Transforms/LoopVectorize/runtime-check.ll
index beb8aa2c99811..1f6e86f9e502a 100644
--- a/llvm/test/Transforms/LoopVectorize/runtime-check.ll
+++ b/llvm/test/Transforms/LoopVectorize/runtime-check.ll
@@ -345,11 +345,11 @@ define void @different_load_store_pairs(ptr %src.1, ptr %src.2, ptr %dst.1, ptr
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr i32, ptr [[SRC_1]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META22:![0-9]+]]
; CHECK-NEXT: [[TMP5:%.*]] = getelementptr i64, ptr [[SRC_2]], i64 [[INDEX]]
-; CHECK-NEXT: [[WIDE_LOAD34:%.*]] = load <4 x i64>, ptr [[TMP5]], align 8, !alias.scope [[META25:![0-9]+]]
+; CHECK-NEXT: [[WIDE_LOAD19:%.*]] = load <4 x i64>, ptr [[TMP5]], align 8, !alias.scope [[META25:![0-9]+]]
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr nusw i32, ptr [[DST_1]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x i32> [[WIDE_LOAD]], ptr [[TMP6]], align 4, !alias.scope [[META27:![0-9]+]], !noalias [[META29:![0-9]+]]
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr nusw i64, ptr [[DST_2]], i64 [[INDEX]]
-; CHECK-NEXT: store <4 x i64> [[WIDE_LOAD34]], ptr [[TMP7]], align 8, !alias.scope [[META31:![0-9]+]], !noalias [[META32:![0-9]+]]
+; CHECK-NEXT: store <4 x i64> [[WIDE_LOAD19]], ptr [[TMP7]], align 8, !alias.scope [[META31:![0-9]+]], !noalias [[META32:![0-9]+]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
More information about the llvm-commits
mailing list