[llvm] [VPlan] Model first memory runtime checks as VPlan recipes. (PR #221483)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 22 14:12:06 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/221483
>From 53c5f377e7d07d08b65104a048a0f938f6f736cb Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 3 Sep 2026 14:40:45 +0100
Subject: [PATCH 1/5] [VPlan] Model first memory runtime checks as VPlan
recipes.
Add VPlanTransforms::addMemoryRuntimeChecks to generate memory runtime
checks directly as VPlan recipes, re-using VPSCEVExpander. This enables
more CSE/simplification opportunities in VPlan, and in the long term
allows us to remove the current slightly hacky runtime check generation
code, which builds checks up front in IR, and removes them again if we
do not vectorize. This whole dance requires quite a bit of code to make
the removal step transparent if the loop was not vectorized.
The initial patch excludes the following checks from VPlan expansion (to
be added in follow-ups), to keep the initial logic as simple as
possible:
* checks in nested loops (VPlan expansion cannot hoist out of the outer
loop)
* diff checks
* bounds requiring expanding AddRecs (currently AddRecs must be
expanded in the Plan's entry)
The IR block built by GeneratedRTChecks::create() is still needed to cost the
checks and to detect that they folded to a constant. Once all kinds of
runtime checks can be created directly in VPlan, we can compute their
cost via the VPlan-based cost model, like other skeleton blocks.
---
.../Vectorize/LoopVectorizationPlanner.h | 5 +-
.../Transforms/Vectorize/LoopVectorize.cpp | 106 ++++++++++++++----
.../Vectorize/VPlanConstruction.cpp | 45 +++++++-
.../Transforms/Vectorize/VPlanTransforms.h | 8 ++
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 29 ++++-
.../LoopVectorize/AArch64/predicated-costs.ll | 3 +-
.../LoopVectorize/AArch64/reduction-cost.ll | 3 +-
.../RISCV/blocks-with-dead-instructions.ll | 4 +-
.../LoopVectorize/RISCV/dead-ops-cost.ll | 10 +-
.../LoopVectorize/RISCV/induction-costs.ll | 12 +-
.../truncate-to-minimal-bitwidth-cost.ll | 7 +-
.../RISCV/type-info-cache-evl-crash.ll | 3 +-
.../LoopVectorize/VPlan/memory-checks.ll | 87 +++++++-------
.../LoopVectorize/X86/cost-model.ll | 5 +-
.../LoopVectorize/X86/interleave-cost.ll | 3 +-
.../LoopVectorize/consecutive-ptr-uniforms.ll | 20 ++--
.../Transforms/LoopVectorize/if-conversion.ll | 5 +-
.../Transforms/LoopVectorize/induction.ll | 30 ++---
.../interleaved-accesses-metadata.ll | 6 +-
.../invariant-store-vectorization-2.ll | 9 +-
.../invariant-store-vectorization.ll | 15 +--
.../test/Transforms/LoopVectorize/metadata.ll | 12 +-
.../Transforms/LoopVectorize/opaque-ptr.ll | 24 ++--
.../pointer-select-runtime-checks.ll | 36 +++---
.../pr59319-loop-access-info-invalidation.ll | 10 +-
.../Transforms/LoopVectorize/runtime-check.ll | 12 +-
26 files changed, 293 insertions(+), 216 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index e14c335701f628..a848f4c4484f3b 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -1009,9 +1009,10 @@ class LoopVectorizationPlanner {
void addMinimumIterationCheck(VPlan &Plan, ElementCount VF, unsigned UF,
ElementCount MinProfitableTripCount) const;
- /// Attach the runtime checks of \p RTChecks to \p Plan.
+ /// Attach the runtime checks of \p RTChecks to \p Plan. Generates the memory
+ /// checks as recipes if \p UseVPlanMemChecks is true and they are supported.
void attachRuntimeChecks(VPlan &Plan, GeneratedRTChecks &RTChecks,
- bool HasBranchWeights) const;
+ bool HasBranchWeights, bool UseVPlanMemChecks) const;
/// Update loop metadata and profile info for both the scalar remainder loop
/// and \p VectorLoop, if it exists. Keeps all loop hints from the original
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 6e27d1e209daf1..c6062de3042b85 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1546,6 +1546,10 @@ class GeneratedRTChecks {
/// If it is nullptr no memory runtime checks have been generated.
Value *MemRuntimeCheckCond = nullptr;
+ /// Set in create() if memory checks were generated and did not fold away.
+ /// Unlike MemRuntimeCheckCond, stays set when the pre-built block is dropped.
+ bool HasMemChecks = false;
+
DominatorTree *DT;
LoopInfo *LI;
TargetTransformInfo *TTI;
@@ -1647,6 +1651,7 @@ class GeneratedRTChecks {
assert(MemRuntimeCheckCond &&
"no RT checks generated although RtPtrChecking "
"claimed checks are required");
+ HasMemChecks = getMemRuntimeChecks().first != nullptr;
}
SCEVExp.eraseDeadInstructions(SCEVCheckCond);
@@ -1773,32 +1778,17 @@ class GeneratedRTChecks {
/// unused.
~GeneratedRTChecks() {
SCEVExpanderCleaner SCEVCleaner(SCEVExp);
- SCEVExpanderCleaner MemCheckCleaner(MemCheckExp);
bool SCEVChecksUsed = !SCEVCheckBlock || !pred_empty(SCEVCheckBlock);
- bool MemChecksUsed = !MemCheckBlock || !pred_empty(MemCheckBlock);
if (SCEVChecksUsed)
SCEVCleaner.markResultUsed();
- if (MemChecksUsed) {
- MemCheckCleaner.markResultUsed();
- } else {
- auto &SE = *MemCheckExp.getSE();
- // Memory runtime check generation creates compares that use expanded
- // values. Remove them before running the SCEVExpanderCleaners.
- for (auto &I : make_early_inc_range(reverse(*MemCheckBlock))) {
- if (MemCheckExp.isInsertedInstruction(&I))
- continue;
- SE.forgetValue(&I);
- I.eraseFromParent();
- }
- }
- MemCheckCleaner.cleanup();
+ if (MemCheckBlock && pred_empty(MemCheckBlock))
+ eraseMemCheckBlock();
+
SCEVCleaner.cleanup();
if (!SCEVChecksUsed)
SCEVCheckBlock->eraseFromParent();
- if (!MemChecksUsed)
- MemCheckBlock->eraseFromParent();
}
/// Retrieves the SCEVCheckCond and SCEVCheckBlock that were generated as IR
@@ -1821,8 +1811,33 @@ class GeneratedRTChecks {
}
/// Return true if any runtime checks have been added
- bool hasChecks() const {
- return getSCEVChecks().first || getMemRuntimeChecks().first;
+ bool hasChecks() const { return getSCEVChecks().first || HasMemChecks; }
+
+ /// Drop the pre-built memory check block in favour of VPlan recipes.
+ /// TODO: Remove once the checks can be costed in VPlan, before VF selection.
+ void dropMemRuntimeChecks() {
+ assert(MemCheckBlock && pred_empty(MemCheckBlock) &&
+ "cannot drop memory checks that are missing or already connected");
+ eraseMemCheckBlock();
+ }
+
+private:
+ /// Erase the memory check block, its instructions and their SCEV expansions.
+ void eraseMemCheckBlock() {
+ SCEVExpanderCleaner MemCheckCleaner(MemCheckExp);
+ auto &SE = *MemCheckExp.getSE();
+ // Memory runtime check generation creates compares that use expanded
+ // values. Remove them before running the SCEVExpanderCleaner.
+ for (auto &I : make_early_inc_range(reverse(*MemCheckBlock))) {
+ if (MemCheckExp.isInsertedInstruction(&I))
+ continue;
+ SE.forgetValue(&I);
+ I.eraseFromParent();
+ }
+ MemCheckCleaner.cleanup();
+ MemCheckBlock->eraseFromParent();
+ MemCheckBlock = nullptr;
+ MemRuntimeCheckCond = nullptr;
}
};
} // namespace
@@ -7001,8 +7016,37 @@ void LoopVectorizationPlanner::addReductionResultComputation(
RUN_VPLAN_PASS(VPlanTransforms::clearReductionWrapFlags, *Plan);
}
+/// Return true if \p CG's bounds can be expanded in the check block.
+/// VPSCEVExpander only expands AddRecs of loops enclosing the plan's scope.
+static bool boundsAreVPlanExpandable(const RuntimeCheckingPtrGroup &CG,
+ ScalarEvolution &SE) {
+ return !SE.containsAddRecurrence(CG.Low) &&
+ !SE.containsAddRecurrence(CG.High);
+}
+
+/// Return true if \p RtPtrChecking's memory checks for \p OrigLoop can be
+/// modelled as VPlan recipes.
+static bool
+canModelMemChecksInVPlan(const RuntimePointerChecking &RtPtrChecking,
+ const Loop &OrigLoop, ScalarEvolution &SE) {
+ // Diff checks are not modelled in VPlan yet.
+ if (RtPtrChecking.getDiffChecks())
+ return false;
+
+ // The VPlan expander cannot hoist bounds out of an enclosing loop.
+ if (OrigLoop.getParentLoop())
+ return false;
+
+ ArrayRef<RuntimePointerCheck> Checks = RtPtrChecking.getChecks();
+ return !Checks.empty() && all_of(Checks, [&SE](const RuntimePointerCheck &C) {
+ return boundsAreVPlanExpandable(*C.first, SE) &&
+ boundsAreVPlanExpandable(*C.second, SE);
+ });
+}
+
void LoopVectorizationPlanner::attachRuntimeChecks(
- VPlan &Plan, GeneratedRTChecks &RTChecks, bool HasBranchWeights) const {
+ VPlan &Plan, GeneratedRTChecks &RTChecks, bool HasBranchWeights,
+ bool UseVPlanMemChecks) const {
const auto &[SCEVCheckCond, SCEVCheckBlock] = RTChecks.getSCEVChecks();
if (SCEVCheckBlock && SCEVCheckBlock->hasNPredecessors(0)) {
assert((!Config.OptForSize ||
@@ -7033,6 +7077,17 @@ void LoopVectorizationPlanner::attachRuntimeChecks(
"(e.g., adding 'restrict').";
});
}
+ const RuntimePointerChecking &RtPtrChecking =
+ *Legal->getLAI()->getRuntimePointerChecking();
+ ScalarEvolution &SE = *PSE.getSE();
+ if (UseVPlanMemChecks &&
+ canModelMemChecksInVPlan(RtPtrChecking, *OrigLoop, SE)) {
+ RTChecks.dropMemRuntimeChecks();
+ RUN_VPLAN_PASS(VPlanTransforms::addMemoryRuntimeChecks, Plan,
+ RtPtrChecking.getChecks(), SE, OrigLoop->getStartLoc(),
+ HasBranchWeights);
+ return;
+ }
RUN_VPLAN_PASS(VPlanTransforms::attachCheckBlock, Plan, MemCheckCond,
MemCheckBlock, HasBranchWeights);
}
@@ -8175,7 +8230,11 @@ bool LoopVectorizePass::processLoop(Loop *L) {
// checks for the main plan.
LVP.addMinimumIterationCheck(BestMainPlan, EPI.EpilogueVF, EPI.EpilogueUF,
ElementCount::getFixed(0));
- LVP.attachRuntimeChecks(BestMainPlan, Checks, HasBranchWeights);
+ // Epilogue vectorization has not been converted to VPlan memory checks
+ // yet; it shares the checks between the main and epilogue plans via the
+ // pre-built IR block.
+ LVP.attachRuntimeChecks(BestMainPlan, Checks, HasBranchWeights,
+ /*UseVPlanMemChecks=*/false);
RUN_VPLAN_PASS(
VPlanTransforms::addIterationCountCheckBlock, BestMainPlan,
EPI.MainLoopVF, EPI.MainLoopUF, BestMainPlan.requiresScalarEpilogue(),
@@ -8226,7 +8285,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
BestPlan);
LVP.addMinimumIterationCheck(BestPlan, VF.Width, IC,
VF.MinProfitableTripCount);
- LVP.attachRuntimeChecks(BestPlan, Checks, HasBranchWeights);
+ LVP.attachRuntimeChecks(BestPlan, Checks, HasBranchWeights,
+ /*UseVPlanMemChecks=*/true);
if (!IsInnerLoop)
LLVM_DEBUG(dbgs() << "Vectorizing outer loop in \"" << F->getName()
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 18498084a601a0..982ac67c17bfd6 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -24,6 +24,7 @@
#include "llvm/ADT/SmallVectorExtras.h"
#include "llvm/Analysis/BranchProbabilityInfo.h"
#include "llvm/Analysis/Loads.h"
+#include "llvm/Analysis/LoopAccessAnalysis.h"
#include "llvm/Analysis/LoopInfo.h"
#include "llvm/Analysis/LoopIterator.h"
#include "llvm/Analysis/OptimizationRemarkEmitter.h"
@@ -1493,13 +1494,55 @@ void VPlanTransforms::attachCheckBlock(VPlan &Plan, Value *Cond,
attachVPCheckBlock(Plan, CondVPV, CheckBlockVPBB, AddBranchWeights);
}
+void VPlanTransforms::addMemoryRuntimeChecks(
+ VPlan &Plan, ArrayRef<RuntimePointerCheck> Checks, ScalarEvolution &SE,
+ DebugLoc DL, bool AddBranchWeights) {
+ assert(!Checks.empty() && "no checks to replace the pre-built block with");
+
+ auto *MemCheckVPBB = Plan.createVPBasicBlock("vector.memcheck");
+ VPBuilder Builder(MemCheckVPBB);
+ insertCheckBlockBeforeVectorLoop(Plan, MemCheckVPBB);
+ VPSCEVExpander Expander(Builder, SE, DL);
+
+ // Expand each group's bounds once and up front.
+ SmallDenseMap<const RuntimeCheckingPtrGroup *,
+ std::pair<VPValue *, VPValue *>>
+ GroupToBounds;
+ for (const auto &[A, B] : Checks)
+ for (const RuntimeCheckingPtrGroup *CG : {A, B}) {
+ if (GroupToBounds.contains(CG))
+ continue;
+ VPValue *Start = Expander.expand(CG->Low);
+ VPValue *End = Expander.expand(CG->High);
+ if (CG->NeedsFreeze) {
+ Start = Builder.createScalarFreeze(Start, DL);
+ End = Builder.createScalarFreeze(End, DL);
+ }
+ GroupToBounds.try_emplace(CG, Start, End);
+ }
+
+ VPValue *Cond = nullptr;
+ for (const auto &[A, B] : Checks) {
+ auto [AStart, AEnd] = GroupToBounds.at(A);
+ auto [BStart, BEnd] = GroupToBounds.at(B);
+ VPValue *Bound0 =
+ Builder.createICmp(CmpInst::ICMP_ULT, AStart, BEnd, DL, "bound0");
+ VPValue *Bound1 =
+ Builder.createICmp(CmpInst::ICMP_ULT, BStart, AEnd, DL, "bound1");
+ VPValue *IsConflict =
+ Builder.createAnd(Bound0, Bound1, DL, "found.conflict");
+ Cond = Cond ? Builder.createOr(Cond, IsConflict, DL, "conflict.rdx")
+ : IsConflict;
+ }
+ addBypassBranch(Plan, MemCheckVPBB, Cond, AddBranchWeights);
+}
+
void VPlanTransforms::addMinimumIterationCheck(
VPlan &Plan, ElementCount VF, unsigned UF,
ElementCount MinProfitableTripCount, bool RequiresScalarEpilogue,
bool TailFolded, Loop *OrigLoop, const uint32_t *MinItersBypassWeights,
DebugLoc DL, PredicatedScalarEvolution &PSE, VPBasicBlock *CheckBlock) {
// Generate code to check if the loop's trip count is less than VF * UF, or
- // equal to it in case a scalar epilogue is required; this implies that the
// vector trip count is zero. This check also covers the case where adding one
// to the backedge-taken count overflowed leading to an incorrect trip count
// of zero. In this case we will also jump to the scalar loop.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 6e5a2184270a5b..7cf2143bc03eda 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -17,6 +17,7 @@
#include "VPlanVerifier.h"
#include "llvm/ADT/STLFunctionalExtras.h"
#include "llvm/ADT/ScopeExit.h"
+#include "llvm/Analysis/LoopAccessAnalysis.h"
#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/Support/CommandLine.h"
#include "llvm/Support/Compiler.h"
@@ -236,6 +237,13 @@ struct VPlanTransforms {
static void attachCheckBlock(VPlan &Plan, Value *Cond, BasicBlock *CheckBlock,
bool AddBranchWeights);
+ /// Generate recipes for the memory runtime checks \p Checks in a new block
+ /// added to \p Plan.
+ static void addMemoryRuntimeChecks(VPlan &Plan,
+ ArrayRef<RuntimePointerCheck> Checks,
+ ScalarEvolution &SE, DebugLoc DL,
+ bool AddBranchWeights);
+
/// Replaces the VPInstructions in \p Plan with corresponding
/// widen recipes. Returns false if any VPInstructions could not be converted
/// to a wide recipe if needed. Uses \p PSE to detect contiguous memory
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 9a07697f66762f..edc2f212e25092 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -954,6 +954,20 @@ VPValue *VPSCEVExpander::expand(const SCEV *S) {
}
case scMulExpr: {
auto *MulE = cast<SCEVMulExpr>(S);
+
+ // mul(PowerOf2C, udiv(X, PowerOf2C)) == (X >> C) << C -> X & (-1 << C),
+ // matching SCEVExpander::visitMulExpr.
+ const SCEVConstant *C1, *C2;
+ const SCEV *Val;
+ if (match(S, m_scev_Mul(m_SCEVConstant(C1),
+ m_scev_UDiv(m_SCEV(Val), m_SCEVConstant(C2)))) &&
+ C1 == C2 && C1->getAPInt().isPowerOf2()) {
+ VPValue *LHS = expand(Val);
+ APInt Mask = APInt::getBitsSetFrom(MulE->getType()->getScalarSizeInBits(),
+ C1->getAPInt().logBase2());
+ return Builder.createAnd(LHS, Builder.getPlan().getConstantInt(Mask), DL);
+ }
+
VPIRFlags::WrapFlagsTy WrapFlags(MulE->hasNoUnsignedWrap(),
MulE->hasNoSignedWrap());
SmallVector<VPValue *, 2> Ops;
@@ -1077,9 +1091,18 @@ VPValue *VPSCEVExpander::expand(const SCEV *S) {
Ops.push_back(OpV);
}
VPValue *Result = Ops.front();
- for (VPValue *Op : drop_begin(Ops))
- Result = Builder.createScalarIntrinsic(IntrinsicID, {Result, Op},
- ResultTy, DL);
+ for (VPValue *Op : drop_begin(Ops)) {
+ if (ResultTy->isPointerTy()) {
+ // The min/max intrinsics don't support pointer operands, so expand
+ // pointer-typed min/max as cmp + select, matching SCEVExpander.
+ VPValue *Cmp = Builder.createICmp(
+ MinMaxIntrinsic::getPredicate(IntrinsicID), Result, Op, DL);
+ Result = Builder.createSelect(Cmp, Result, Op, DL);
+ } else {
+ Result = Builder.createScalarIntrinsic(IntrinsicID, {Result, Op},
+ ResultTy, DL);
+ }
+ }
return Result;
}
case scAddRecExpr: {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/predicated-costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/predicated-costs.ll
index 285e978a5b9a82..7cdfe9be2ce067 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/predicated-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/predicated-costs.ll
@@ -45,8 +45,7 @@ define void @test_predicated_load_cast_hint(ptr %dst.1, ptr %dst.2, ptr %src, i8
; CHECK-NEXT: [[SCEVGEP5:%.*]] = getelementptr i8, ptr [[SRC]], i64 1
; CHECK-NEXT: [[TMP18:%.*]] = shl i64 [[OFF]], 3
; CHECK-NEXT: [[SCEVGEP6:%.*]] = getelementptr i8, ptr [[DST_1]], i64 [[TMP18]]
-; CHECK-NEXT: [[SMAX7:%.*]] = call i32 @llvm.smax.i32(i32 [[N_SUB]], i32 4)
-; CHECK-NEXT: [[TMP19:%.*]] = zext nneg i32 [[SMAX7]] to i64
+; CHECK-NEXT: [[TMP19:%.*]] = zext nneg i32 [[SMAX16]] to i64
; CHECK-NEXT: [[TMP20:%.*]] = add nsw i64 [[TMP19]], -1
; CHECK-NEXT: [[TMP21:%.*]] = lshr i64 [[TMP20]], 2
; CHECK-NEXT: [[TMP22:%.*]] = shl nuw nsw i64 [[TMP21]], 9
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll
index 32eeca5c60b889..f7d912dfedf50e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/reduction-cost.ll
@@ -107,8 +107,7 @@ define i32 @or_reduction_with_freeze(ptr %dst, ptr %src) {
; CHECK-NEXT: [[IDENT_CHECK:%.*]] = icmp ne i64 [[TMP6]], 0
; CHECK-NEXT: br i1 [[IDENT_CHECK]], label %[[SCALAR_PH]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT: [[TMP7:%.*]] = sub i64 [[DST1]], [[SRC7]]
-; CHECK-NEXT: [[TMP9:%.*]] = and i64 [[TMP7]], -8
+; CHECK-NEXT: [[TMP9:%.*]] = and i64 [[TMP0]], -8
; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[TMP9]], 8
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP10]]
; CHECK-NEXT: [[SCEVGEP5:%.*]] = getelementptr i8, ptr [[DST]], i64 8
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/blocks-with-dead-instructions.ll b/llvm/test/Transforms/LoopVectorize/RISCV/blocks-with-dead-instructions.ll
index 9fa4da804956bf..08ae8312afd61b 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/blocks-with-dead-instructions.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/blocks-with-dead-instructions.ll
@@ -459,9 +459,7 @@ define void @dead_load_in_block(ptr %dst, ptr %src, i8 %N, i64 %x) #0 {
; CHECK-NEXT: [[TMP2:%.*]] = add nuw nsw i64 [[TMP1]], 1
; CHECK-NEXT: br label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT: [[TMP7:%.*]] = add nuw nsw i64 [[N_EXT]], 2
-; CHECK-NEXT: [[TMP8:%.*]] = udiv i64 [[TMP7]], 3
-; CHECK-NEXT: [[TMP5:%.*]] = mul nuw nsw i64 [[TMP8]], 12
+; CHECK-NEXT: [[TMP5:%.*]] = mul nuw nsw i64 [[TMP1]], 12
; CHECK-NEXT: [[TMP11:%.*]] = add nuw nsw i64 [[TMP5]], 4
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 [[TMP11]]
; CHECK-NEXT: [[TMP12:%.*]] = shl i64 [[X]], 2
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll b/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll
index 5d90cafc565c87..b0431edb046627 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/dead-ops-cost.ll
@@ -130,7 +130,7 @@ exit:
; Test case for https://github.com/llvm/llvm-project/issues/106780.
define i32 @cost_of_exit_branch_and_cond_insts(ptr %a, ptr %b, i1 %c, i16 %x) vscale_range(2, 1024) {
; CHECK-LABEL: define i32 @cost_of_exit_branch_and_cond_insts(
-; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i1 [[C:%.*]], i16 [[X:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i1 [[C:%.*]], i16 [[X:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[TMP0:%.*]] = zext i16 [[X]] to i32
; CHECK-NEXT: [[UMAX3:%.*]] = call i32 @llvm.umax.i32(i32 [[TMP0]], i32 111)
@@ -141,11 +141,7 @@ define i32 @cost_of_exit_branch_and_cond_insts(ptr %a, ptr %b, i1 %c, i16 %x) vs
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 1
-; CHECK-NEXT: [[TMP3:%.*]] = zext i16 [[X]] to i32
-; CHECK-NEXT: [[UMAX1:%.*]] = call i32 @llvm.umax.i32(i32 [[TMP3]], i32 111)
-; CHECK-NEXT: [[TMP4:%.*]] = sub i32 770, [[UMAX1]]
-; CHECK-NEXT: [[SMAX:%.*]] = call i32 @llvm.smax.i32(i32 [[TMP4]], i32 0)
-; CHECK-NEXT: [[TMP5:%.*]] = zext nneg i32 [[SMAX]] to i64
+; CHECK-NEXT: [[TMP5:%.*]] = zext nneg i32 [[SMAX4]] to i64
; CHECK-NEXT: [[TMP6:%.*]] = shl nuw nsw i64 [[TMP5]], 2
; CHECK-NEXT: [[TMP7:%.*]] = add nuw nsw i64 [[TMP6]], 4
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP7]]
@@ -269,7 +265,7 @@ return:
; Test case for https://github.com/llvm/llvm-project/issues/107473.
define void @test_phi_in_latch_redundant(ptr %dst, i32 %a) vscale_range(2, 1024) {
; CHECK-LABEL: define void @test_phi_in_latch_redundant(
-; CHECK-SAME: ptr [[DST:%.*]], i32 [[A:%.*]]) #[[ATTR1]] {
+; CHECK-SAME: ptr [[DST:%.*]], i32 [[A:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/induction-costs.ll b/llvm/test/Transforms/LoopVectorize/RISCV/induction-costs.ll
index aa5734371f0d5d..344f68b50aa827 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/induction-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/induction-costs.ll
@@ -21,17 +21,13 @@ define void @skip_free_iv_truncate(i16 %x, ptr %A) #0 {
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[TMP31:%.*]] = shl nsw i64 [[X_I64]], 1
; CHECK-NEXT: [[SCEVGEP9:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP31]]
-; CHECK-NEXT: [[SMAX10:%.*]] = call i64 @llvm.smax.i64(i64 [[X_I64]], i64 99)
-; CHECK-NEXT: [[TMP33:%.*]] = add i64 [[SMAX10]], 2
-; CHECK-NEXT: [[TMP34:%.*]] = sub i64 [[TMP33]], [[X_I64]]
-; CHECK-NEXT: [[TMP35:%.*]] = udiv i64 [[TMP34]], 3
-; CHECK-NEXT: [[TMP37:%.*]] = mul i64 [[TMP35]], 6
+; CHECK-NEXT: [[TMP37:%.*]] = mul i64 [[TMP3]], 6
; CHECK-NEXT: [[TMP38:%.*]] = add i64 [[TMP37]], [[TMP31]]
; CHECK-NEXT: [[TMP39:%.*]] = add i64 [[TMP38]], 2
; CHECK-NEXT: [[SCEVGEP12:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP39]]
; CHECK-NEXT: [[TMP40:%.*]] = shl nsw i64 [[X_I64]], 3
; CHECK-NEXT: [[SCEVGEP13:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP40]]
-; CHECK-NEXT: [[TMP41:%.*]] = mul i64 [[TMP35]], 24
+; CHECK-NEXT: [[TMP41:%.*]] = mul i64 [[TMP3]], 24
; CHECK-NEXT: [[TMP42:%.*]] = add i64 [[TMP41]], [[TMP40]]
; CHECK-NEXT: [[TMP43:%.*]] = add i64 [[TMP42]], 8
; CHECK-NEXT: [[SCEVGEP14:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP43]]
@@ -47,15 +43,13 @@ define void @skip_free_iv_truncate(i16 %x, ptr %A) #0 {
; CHECK-NEXT: [[CONFLICT_RDX:%.*]] = or i1 [[FOUND_CONFLICT]], [[FOUND_CONFLICT19]]
; CHECK-NEXT: br i1 [[CONFLICT_RDX]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP24:%.*]] = shl nsw i64 [[X_I64]], 1
-; CHECK-NEXT: [[SCEVGEP10:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP24]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[AVL:%.*]] = phi i64 [ [[TMP4]], %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP27:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 8, i1 true)
; CHECK-NEXT: [[TMP22:%.*]] = mul i64 [[INDEX]], 6
-; CHECK-NEXT: [[TMP23:%.*]] = getelementptr i8, ptr [[SCEVGEP10]], i64 [[TMP22]]
+; CHECK-NEXT: [[TMP23:%.*]] = getelementptr i8, ptr [[SCEVGEP9]], i64 [[TMP22]]
; CHECK-NEXT: call void @llvm.experimental.vp.strided.store.nxv8i16.p0.i64(<vscale x 8 x i16> zeroinitializer, ptr align 2 [[TMP23]], i64 6, <vscale x 8 x i1> splat (i1 true), i32 [[TMP27]]), !alias.scope [[META0:![0-9]+]], !noalias [[META3:![0-9]+]]
; CHECK-NEXT: [[TMP28:%.*]] = zext i32 [[TMP27]] to i64
; CHECK-NEXT: [[CURRENT_ITERATION_NEXT]] = add nuw i64 [[TMP28]], [[INDEX]]
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/truncate-to-minimal-bitwidth-cost.ll b/llvm/test/Transforms/LoopVectorize/RISCV/truncate-to-minimal-bitwidth-cost.ll
index e7acfe9a4f4fc0..3848f7f89726c3 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/truncate-to-minimal-bitwidth-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/truncate-to-minimal-bitwidth-cost.ll
@@ -264,12 +264,9 @@ define void @test_minbws_for_trunc(i32 %n, ptr noalias %p1, ptr noalias %p2) vsc
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[P2]], [[SCEVGEP]]
; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
; CHECK-NEXT: [[BOUND06:%.*]] = icmp ult ptr [[P2]], [[SCEVGEP5]]
-; CHECK-NEXT: [[BOUND17:%.*]] = icmp ult ptr [[P2]], [[SCEVGEP]]
-; CHECK-NEXT: [[FOUND_CONFLICT8:%.*]] = and i1 [[BOUND06]], [[BOUND17]]
+; CHECK-NEXT: [[FOUND_CONFLICT8:%.*]] = and i1 [[BOUND06]], [[BOUND1]]
; CHECK-NEXT: [[CONFLICT_RDX:%.*]] = or i1 [[FOUND_CONFLICT]], [[FOUND_CONFLICT8]]
-; CHECK-NEXT: [[BOUND09:%.*]] = icmp ult ptr [[P2]], [[SCEVGEP5]]
-; CHECK-NEXT: [[BOUND110:%.*]] = icmp ult ptr [[P2]], [[SCEVGEP4]]
-; CHECK-NEXT: [[FOUND_CONFLICT11:%.*]] = and i1 [[BOUND09]], [[BOUND110]]
+; CHECK-NEXT: [[FOUND_CONFLICT11:%.*]] = and i1 [[BOUND06]], [[BOUND0]]
; CHECK-NEXT: [[CONFLICT_RDX12:%.*]] = or i1 [[CONFLICT_RDX]], [[FOUND_CONFLICT11]]
; CHECK-NEXT: br i1 [[CONFLICT_RDX12]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/type-info-cache-evl-crash.ll b/llvm/test/Transforms/LoopVectorize/RISCV/type-info-cache-evl-crash.ll
index a9c793406032e4..063722538d1efb 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/type-info-cache-evl-crash.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/type-info-cache-evl-crash.ll
@@ -12,8 +12,7 @@ define void @type_info_cache_clobber(ptr %dstv, ptr %src, i64 %wide.trip.count)
; CHECK-NEXT: br label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DSTV]], i64 1
-; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[WIDE_TRIP_COUNT]], 1
-; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP5]]
+; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP0]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[DSTV]], [[SCEVGEP1]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[SRC]], [[SCEVGEP]]
; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll b/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
index 344ccf7d03685b..f52a8edeacef99 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
@@ -9,31 +9,30 @@ define void @three_groups_shared_bounds(ptr %a, ptr %b, ptr %c, i64 %n) {
; CHECK-NEXT: EMIT-SCALAR vp<[[VP2:%[0-9]+]]> = call i64 @llvm.umax(ir<%n>, ir<1>)
; CHECK-NEXT: EMIT vp<%min.iters.check> = icmp ult vp<[[VP2]]>, ir<4>
; CHECK-NEXT: EMIT branch-on-cond vp<%min.iters.check>
-; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, ir-bb<vector.memcheck>
+; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, vector.memcheck
; CHECK-EMPTY:
-; CHECK-NEXT: ir-bb<vector.memcheck>:
-; CHECK-NEXT: IR %umax = call i64 @llvm.umax.i64(i64 %n, i64 1)
-; CHECK-NEXT: IR %0 = shl i64 %umax, 2
-; CHECK-NEXT: IR %scevgep = getelementptr i8, ptr %c, i64 %0
-; CHECK-NEXT: IR %scevgep1 = getelementptr i8, ptr %a, i64 %0
-; CHECK-NEXT: IR %scevgep2 = getelementptr i8, ptr %b, i64 %0
-; CHECK-NEXT: IR %bound0 = icmp ult ptr %c, %scevgep1
-; CHECK-NEXT: IR %bound1 = icmp ult ptr %a, %scevgep
-; CHECK-NEXT: IR %found.conflict = and i1 %bound0, %bound1
-; CHECK-NEXT: IR %bound03 = icmp ult ptr %c, %scevgep2
-; CHECK-NEXT: IR %bound14 = icmp ult ptr %b, %scevgep
-; CHECK-NEXT: IR %found.conflict5 = and i1 %bound03, %bound14
-; CHECK-NEXT: IR %conflict.rdx = or i1 %found.conflict, %found.conflict5
-; CHECK-NEXT: IR %bound06 = icmp ult ptr %a, %scevgep2
-; CHECK-NEXT: IR %bound17 = icmp ult ptr %b, %scevgep1
-; CHECK-NEXT: IR %found.conflict8 = and i1 %bound06, %bound17
-; CHECK-NEXT: IR %conflict.rdx9 = or i1 %conflict.rdx, %found.conflict8
-; CHECK-NEXT: EMIT branch-on-cond ir<%conflict.rdx9>
+; CHECK-NEXT: vector.memcheck:
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = shl vp<[[VP2]]>, ir<2>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = ptradd ir<%c>, vp<[[VP4]]>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = ptradd ir<%a>, vp<[[VP4]]>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = ptradd ir<%b>, vp<[[VP4]]>
+; CHECK-NEXT: EMIT vp<%bound0> = icmp ult ir<%c>, vp<[[VP6]]>
+; CHECK-NEXT: EMIT vp<%bound1> = icmp ult ir<%a>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<%found.conflict> = and vp<%bound0>, vp<%bound1>
+; CHECK-NEXT: EMIT vp<%bound0>.1 = icmp ult ir<%c>, vp<[[VP7]]>
+; CHECK-NEXT: EMIT vp<%bound1>.1 = icmp ult ir<%b>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<%found.conflict>.1 = and vp<%bound0>.1, vp<%bound1>.1
+; CHECK-NEXT: EMIT vp<%conflict.rdx> = or vp<%found.conflict>, vp<%found.conflict>.1
+; CHECK-NEXT: EMIT vp<%bound0>.2 = icmp ult ir<%a>, vp<[[VP7]]>
+; CHECK-NEXT: EMIT vp<%bound1>.2 = icmp ult ir<%b>, vp<[[VP6]]>
+; CHECK-NEXT: EMIT vp<%found.conflict>.2 = and vp<%bound0>.2, vp<%bound1>.2
+; CHECK-NEXT: EMIT vp<%conflict.rdx>.1 = or vp<%conflict.rdx>, vp<%found.conflict>.2
+; CHECK-NEXT: EMIT branch-on-cond vp<%conflict.rdx>.1
; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, vector.ph
; CHECK-EMPTY:
; CHECK-NEXT: vector.ph:
-; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = and vp<[[VP2]]>, ir<3>
-; CHECK-NEXT: EMIT vp<%n.vec> = sub vp<[[VP2]]>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = and vp<[[VP2]]>, ir<3>
+; CHECK-NEXT: EMIT vp<%n.vec> = sub vp<[[VP2]]>, vp<[[VP9]]>
; CHECK-NEXT: Successor(s): vector.body
; CHECK-EMPTY:
; CHECK-NEXT: vector.body:
@@ -68,33 +67,33 @@ define void @ptr_minmax_bounds(ptr %a, ptr %b, i64 %n, i64 %s, i64 %t) {
; CHECK-NEXT: IR %step = mul i64 %s, %t
; CHECK-NEXT: EMIT vp<%min.iters.check> = icmp ult ir<%n>, ir<4>
; CHECK-NEXT: EMIT branch-on-cond vp<%min.iters.check>
-; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, ir-bb<vector.memcheck>
+; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, vector.memcheck
; CHECK-EMPTY:
-; CHECK-NEXT: ir-bb<vector.memcheck>:
-; CHECK-NEXT: IR %0 = shl i64 %n, 2
-; CHECK-NEXT: IR %scevgep = getelementptr i8, ptr %b, i64 %0
-; CHECK-NEXT: IR %1 = mul i64 %t, %s
-; CHECK-NEXT: IR %2 = add i64 %n, -1
-; CHECK-NEXT: IR %3 = mul i64 %1, %2
-; CHECK-NEXT: IR %4 = shl i64 %3, 2
-; CHECK-NEXT: IR %scevgep1 = getelementptr i8, ptr %a, i64 %4
-; CHECK-NEXT: IR %5 = icmp ult ptr %a, %scevgep1
-; CHECK-NEXT: IR %umin = select i1 %5, ptr %a, ptr %scevgep1
-; CHECK-NEXT: IR %6 = icmp ugt ptr %a, %scevgep1
-; CHECK-NEXT: IR %umax = select i1 %6, ptr %a, ptr %scevgep1
-; CHECK-NEXT: IR %scevgep2 = getelementptr i8, ptr %umax, i64 4
-; CHECK-NEXT: IR %bound0 = icmp ult ptr %b, %scevgep2
-; CHECK-NEXT: IR %bound1 = icmp ult ptr %umin, %scevgep
-; CHECK-NEXT: IR %found.conflict = and i1 %bound0, %bound1
-; CHECK-NEXT: EMIT branch-on-cond ir<%found.conflict>
+; CHECK-NEXT: vector.memcheck:
+; CHECK-NEXT: EMIT vp<[[VP3:%[0-9]+]]> = shl ir<%n>, ir<2>
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = ptradd ir<%b>, vp<[[VP3]]>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = add ir<%n>, ir<-1>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = mul ir<%t>, ir<%s>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = mul vp<[[VP6]]>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = shl vp<[[VP7]]>, ir<2>
+; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = ptradd ir<%a>, vp<[[VP8]]>
+; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = icmp ult ir<%a>, vp<[[VP9]]>
+; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = select vp<[[VP10]]>, ir<%a>, vp<[[VP9]]>
+; CHECK-NEXT: EMIT vp<[[VP12:%[0-9]+]]> = icmp ugt ir<%a>, vp<[[VP9]]>
+; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = select vp<[[VP12]]>, ir<%a>, vp<[[VP9]]>
+; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = ptradd vp<[[VP13]]>, ir<4>
+; CHECK-NEXT: EMIT vp<%bound0> = icmp ult ir<%b>, vp<[[VP14]]>
+; CHECK-NEXT: EMIT vp<%bound1> = icmp ult vp<[[VP11]]>, vp<[[VP4]]>
+; CHECK-NEXT: EMIT vp<%found.conflict> = and vp<%bound0>, vp<%bound1>
+; CHECK-NEXT: EMIT branch-on-cond vp<%found.conflict>
; CHECK-NEXT: Successor(s): ir-bb<scalar.ph>, vector.ph
; CHECK-EMPTY:
; CHECK-NEXT: vector.ph:
-; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = and ir<%n>, ir<3>
-; CHECK-NEXT: EMIT vp<%n.vec> = sub ir<%n>, vp<[[VP4]]>
-; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = broadcast ir<%step>
-; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = step-vector i64
-; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = broadcast ir<4>
+; CHECK-NEXT: EMIT vp<[[VP16:%[0-9]+]]> = and ir<%n>, ir<3>
+; CHECK-NEXT: EMIT vp<%n.vec> = sub ir<%n>, vp<[[VP16]]>
+; CHECK-NEXT: EMIT vp<[[VP17:%[0-9]+]]> = broadcast ir<%step>
+; CHECK-NEXT: EMIT vp<[[VP18:%[0-9]+]]> = step-vector i64
+; CHECK-NEXT: EMIT vp<[[VP19:%[0-9]+]]> = broadcast ir<4>
; CHECK-NEXT: Successor(s): vector.body
; CHECK-EMPTY:
; CHECK-NEXT: vector.body:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll b/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
index fe53027045aa23..42c8fe3011f7d2 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
@@ -473,10 +473,7 @@ define void @multi_exit(ptr %dst, ptr %src.1, ptr %src.2, i64 %A, i64 %B) #0 {
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 1
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[SRC_2]], i64 8
-; CHECK-NEXT: [[TMP15:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[B]], i64 1)
-; CHECK-NEXT: [[TMP16:%.*]] = freeze i64 [[TMP15]]
-; CHECK-NEXT: [[UMIN4:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP16]], i64 [[A]])
-; CHECK-NEXT: [[TMP17:%.*]] = shl i64 [[UMIN4]], 3
+; CHECK-NEXT: [[TMP17:%.*]] = shl i64 [[UMIN10]], 3
; CHECK-NEXT: [[TMP18:%.*]] = add i64 [[TMP17]], 8
; CHECK-NEXT: [[SCEVGEP5:%.*]] = getelementptr i8, ptr [[SRC_1]], i64 [[TMP18]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[DST]], [[SCEVGEP2]]
diff --git a/llvm/test/Transforms/LoopVectorize/X86/interleave-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/interleave-cost.ll
index aa5a0f5be24be8..a4d7aa15dc266b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/interleave-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/interleave-cost.ll
@@ -250,9 +250,8 @@ define void @geps_feeding_interleave_groups_with_reuse2(ptr %A, ptr %B, i64 %N)
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP36]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[A]], [[SCEVGEP35]]
; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
-; CHECK-NEXT: [[BOUND038:%.*]] = icmp ult ptr [[B]], [[SCEVGEP36]]
; CHECK-NEXT: [[BOUND139:%.*]] = icmp ult ptr [[A]], [[SCEVGEP37]]
-; CHECK-NEXT: [[FOUND_CONFLICT40:%.*]] = and i1 [[BOUND038]], [[BOUND139]]
+; CHECK-NEXT: [[FOUND_CONFLICT40:%.*]] = and i1 [[BOUND0]], [[BOUND139]]
; CHECK-NEXT: [[CONFLICT_RDX:%.*]] = or i1 [[FOUND_CONFLICT]], [[FOUND_CONFLICT40]]
; CHECK-NEXT: br i1 [[CONFLICT_RDX]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
diff --git a/llvm/test/Transforms/LoopVectorize/consecutive-ptr-uniforms.ll b/llvm/test/Transforms/LoopVectorize/consecutive-ptr-uniforms.ll
index 50b72c49bf6bf9..a82a0dfcc6185e 100644
--- a/llvm/test/Transforms/LoopVectorize/consecutive-ptr-uniforms.ll
+++ b/llvm/test/Transforms/LoopVectorize/consecutive-ptr-uniforms.ll
@@ -1276,10 +1276,9 @@ define i32 @pointer_iv_mixed(ptr %a, ptr %b, i64 %n) {
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 3
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 3
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
-; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[A]], [[SCEVGEP]]
@@ -1342,10 +1341,9 @@ define i32 @pointer_iv_mixed(ptr %a, ptr %b, i64 %n) {
; INTER-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; INTER-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; INTER: [[VECTOR_MEMCHECK]]:
-; INTER-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; INTER-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 3
+; INTER-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 3
; INTER-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
-; INTER-NEXT: [[TMP1:%.*]] = shl i64 [[SMAX]], 2
+; INTER-NEXT: [[TMP1:%.*]] = shl i64 [[SMAX2]], 2
; INTER-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
; INTER-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
; INTER-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[A]], [[SCEVGEP]]
@@ -1436,9 +1434,8 @@ define void @pointer_operand_geps_with_different_indexed_types(ptr %A, ptr %B, i
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[SMAX]]
-; CHECK-NEXT: [[TMP6:%.*]] = shl i64 [[SMAX]], 3
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[SMAX2]]
+; CHECK-NEXT: [[TMP6:%.*]] = shl i64 [[SMAX2]], 3
; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[TMP6]], -4
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP0]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[TMP1]]
@@ -1512,9 +1509,8 @@ define void @pointer_operand_geps_with_different_indexed_types(ptr %A, ptr %B, i
; INTER-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ule i64 [[SMAX2]], 4
; INTER-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; INTER: [[VECTOR_MEMCHECK]]:
-; INTER-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; INTER-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[SMAX]]
-; INTER-NEXT: [[TMP8:%.*]] = shl i64 [[SMAX]], 3
+; INTER-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[SMAX2]]
+; INTER-NEXT: [[TMP8:%.*]] = shl i64 [[SMAX2]], 3
; INTER-NEXT: [[TMP0:%.*]] = add i64 [[TMP8]], -4
; INTER-NEXT: [[TMP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP0]]
; INTER-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[TMP1]]
diff --git a/llvm/test/Transforms/LoopVectorize/if-conversion.ll b/llvm/test/Transforms/LoopVectorize/if-conversion.ll
index 948746df3f5807..b972d41dbbabbf 100644
--- a/llvm/test/Transforms/LoopVectorize/if-conversion.ll
+++ b/llvm/test/Transforms/LoopVectorize/if-conversion.ll
@@ -34,10 +34,7 @@ define void @function0(ptr nocapture %a, ptr nocapture %b, i32 %start, i32 %end)
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[TMP5:%.*]] = shl nsw i64 [[TMP0]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP5]]
-; CHECK-NEXT: [[TMP6:%.*]] = add i32 [[END]], -1
-; CHECK-NEXT: [[TMP7:%.*]] = sub i32 [[TMP6]], [[START]]
-; CHECK-NEXT: [[TMP8:%.*]] = zext i32 [[TMP7]] to i64
-; CHECK-NEXT: [[TMP9:%.*]] = shl nuw nsw i64 [[TMP8]], 2
+; CHECK-NEXT: [[TMP9:%.*]] = shl nuw nsw i64 [[TMP3]], 2
; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[TMP5]], [[TMP9]]
; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[TMP10]], 4
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP11]]
diff --git a/llvm/test/Transforms/LoopVectorize/induction.ll b/llvm/test/Transforms/LoopVectorize/induction.ll
index 1683f784daf75f..7d29db44a43b63 100644
--- a/llvm/test/Transforms/LoopVectorize/induction.ll
+++ b/llvm/test/Transforms/LoopVectorize/induction.ll
@@ -1551,12 +1551,10 @@ define void @scalarize_induction_variable_04(ptr %a, ptr %p, i32 %n) {
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[P:%.*]], i64 4
-; CHECK-NEXT: [[TMP3:%.*]] = add i32 [[N]], -1
-; CHECK-NEXT: [[TMP4:%.*]] = zext i32 [[TMP3]] to i64
-; CHECK-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP4]], 3
+; CHECK-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP1]], 3
; CHECK-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 8
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP6]]
-; CHECK-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP4]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP1]], 4
; CHECK-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 4
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 [[TMP8]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP2]]
@@ -1615,12 +1613,10 @@ define void @scalarize_induction_variable_04(ptr %a, ptr %p, i32 %n) {
; IND-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; IND: vector.memcheck:
; IND-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[P:%.*]], i64 4
-; IND-NEXT: [[TMP3:%.*]] = add i32 [[N]], -1
-; IND-NEXT: [[TMP4:%.*]] = zext i32 [[TMP3]] to i64
-; IND-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP4]], 3
+; IND-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP1]], 3
; IND-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 8
; IND-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP6]]
-; IND-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP4]], 4
+; IND-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP1]], 4
; IND-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 4
; IND-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 [[TMP8]]
; IND-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP2]]
@@ -1679,12 +1675,10 @@ define void @scalarize_induction_variable_04(ptr %a, ptr %p, i32 %n) {
; UNROLL-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; UNROLL: vector.memcheck:
; UNROLL-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[P:%.*]], i64 4
-; UNROLL-NEXT: [[TMP3:%.*]] = add i32 [[N]], -1
-; UNROLL-NEXT: [[TMP4:%.*]] = zext i32 [[TMP3]] to i64
-; UNROLL-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP4]], 3
+; UNROLL-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP1]], 3
; UNROLL-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 8
; UNROLL-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP6]]
-; UNROLL-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP4]], 4
+; UNROLL-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP1]], 4
; UNROLL-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 4
; UNROLL-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 [[TMP8]]
; UNROLL-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP2]]
@@ -1757,12 +1751,10 @@ define void @scalarize_induction_variable_04(ptr %a, ptr %p, i32 %n) {
; UNROLL-NO-IC-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; UNROLL-NO-IC: vector.memcheck:
; UNROLL-NO-IC-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[P:%.*]], i64 4
-; UNROLL-NO-IC-NEXT: [[TMP3:%.*]] = add i32 [[N]], -1
-; UNROLL-NO-IC-NEXT: [[TMP4:%.*]] = zext i32 [[TMP3]] to i64
-; UNROLL-NO-IC-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP4]], 3
+; UNROLL-NO-IC-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP1]], 3
; UNROLL-NO-IC-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 8
; UNROLL-NO-IC-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP6]]
-; UNROLL-NO-IC-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP4]], 4
+; UNROLL-NO-IC-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP1]], 4
; UNROLL-NO-IC-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 4
; UNROLL-NO-IC-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 [[TMP8]]
; UNROLL-NO-IC-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP2]]
@@ -1835,12 +1827,10 @@ define void @scalarize_induction_variable_04(ptr %a, ptr %p, i32 %n) {
; INTERLEAVE-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; INTERLEAVE: vector.memcheck:
; INTERLEAVE-NEXT: [[SCEVGEP:%.*]] = getelementptr nuw i8, ptr [[P:%.*]], i64 4
-; INTERLEAVE-NEXT: [[TMP3:%.*]] = add i32 [[N]], -1
-; INTERLEAVE-NEXT: [[TMP4:%.*]] = zext i32 [[TMP3]] to i64
-; INTERLEAVE-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP4]], 3
+; INTERLEAVE-NEXT: [[TMP5:%.*]] = shl nuw nsw i64 [[TMP1]], 3
; INTERLEAVE-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 8
; INTERLEAVE-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP6]]
-; INTERLEAVE-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP4]], 4
+; INTERLEAVE-NEXT: [[TMP7:%.*]] = shl nuw nsw i64 [[TMP1]], 4
; INTERLEAVE-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 4
; INTERLEAVE-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 [[TMP8]]
; INTERLEAVE-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP2]]
diff --git a/llvm/test/Transforms/LoopVectorize/interleaved-accesses-metadata.ll b/llvm/test/Transforms/LoopVectorize/interleaved-accesses-metadata.ll
index df243aea6bbdef..f09ceadc126294 100644
--- a/llvm/test/Transforms/LoopVectorize/interleaved-accesses-metadata.ll
+++ b/llvm/test/Transforms/LoopVectorize/interleaved-accesses-metadata.ll
@@ -76,8 +76,8 @@ define void @ir_tbaa_different(ptr %base, ptr %end, ptr %src) {
; CHECK-LABEL: define void @ir_tbaa_different(
; CHECK-SAME: ptr [[BASE:%.*]], ptr [[END:%.*]], ptr [[SRC:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
-; CHECK-NEXT: [[BASE2:%.*]] = ptrtoaddr ptr [[BASE]] to i64
; CHECK-NEXT: [[END2:%.*]] = ptrtoaddr ptr [[END]] to i64
+; CHECK-NEXT: [[BASE2:%.*]] = ptrtoaddr ptr [[BASE]] to i64
; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[END2]], -8
; CHECK-NEXT: [[TMP1:%.*]] = sub i64 [[TMP0]], [[BASE2]]
; CHECK-NEXT: [[TMP2:%.*]] = lshr i64 [[TMP1]], 3
@@ -85,9 +85,7 @@ define void @ir_tbaa_different(ptr %base, ptr %end, ptr %src) {
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP3]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT: [[TMP17:%.*]] = add i64 [[END2]], -8
-; CHECK-NEXT: [[TMP12:%.*]] = sub i64 [[TMP17]], [[BASE2]]
-; CHECK-NEXT: [[TMP14:%.*]] = and i64 [[TMP12]], -8
+; CHECK-NEXT: [[TMP14:%.*]] = and i64 [[TMP1]], -8
; CHECK-NEXT: [[TMP15:%.*]] = add i64 [[TMP14]], 8
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP15]]
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[SRC]], i64 4
diff --git a/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization-2.ll b/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization-2.ll
index 9c31a47e748fcc..4401a0e00392f5 100644
--- a/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization-2.ll
@@ -25,8 +25,7 @@ define void @inv_val_store_to_inv_address_conditional_diff_values_ic(ptr %a, i64
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
@@ -128,8 +127,7 @@ define void @inv_val_store_to_inv_address_conditional_inv(ptr %a, i64 %n, ptr %b
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
@@ -218,8 +216,7 @@ define i32 @variant_val_store_to_inv_address(ptr %a, i64 %n, ptr %b, i32 %k) {
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[A]], [[SCEVGEP1]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[B]], [[SCEVGEP]]
diff --git a/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll b/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll
index 3929994915bf4f..638b3780c8adc6 100644
--- a/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll
+++ b/llvm/test/Transforms/LoopVectorize/invariant-store-vectorization.ll
@@ -25,8 +25,7 @@ define i32 @inv_val_store_to_inv_address_with_reduction(ptr %a, i64 %n, ptr %b)
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[A]], [[SCEVGEP1]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[B]], [[SCEVGEP]]
@@ -101,8 +100,7 @@ define void @inv_val_store_to_inv_address(ptr %a, i64 %n, ptr %b) {
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[A]], [[SCEVGEP1]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[B]], [[SCEVGEP]]
@@ -177,8 +175,7 @@ define void @inv_val_store_to_inv_address_conditional(ptr %a, i64 %n, ptr %b, i3
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX2]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX]], 2
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[SMAX2]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP0]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 4
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[B]], [[SCEVGEP1]]
@@ -355,8 +352,8 @@ define void @multiple_uniform_stores(ptr nocapture %var1, ptr nocapture readonly
; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[TMP2]], 4
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[VAR1:%.*]], i64 [[TMP3]]
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[VAR2:%.*]], i64 4
-; CHECK-NEXT: [[TMP5:%.*]] = add i32 [[ITR]], -1
; CHECK-NEXT: [[TMP10:%.*]] = add i32 [[ITR]], -1
+; CHECK-NEXT: [[TMP5:%.*]] = add i32 [[ITR]], -1
; CHECK-NEXT: br label [[FOR_COND1_PREHEADER:%.*]]
; CHECK: for.cond1.preheader:
; CHECK-NEXT: [[INDVARS_IV23:%.*]] = phi i64 [ [[INDVARS_IV_NEXT24:%.*]], [[FOR_INC8:%.*]] ], [ 0, [[FOR_COND1_PREHEADER_PREHEADER]] ]
@@ -367,7 +364,7 @@ define void @multiple_uniform_stores(ptr nocapture %var1, ptr nocapture readonly
; CHECK-NEXT: [[ARRAYIDX5:%.*]] = getelementptr inbounds i32, ptr [[VAR1]], i64 [[INDVARS_IV23]]
; CHECK-NEXT: [[TMP4:%.*]] = zext i32 [[J_022]] to i64
; CHECK-NEXT: [[ARRAYIDX5_PROMOTED:%.*]] = load i32, ptr [[ARRAYIDX5]], align 4
-; CHECK-NEXT: [[TMP6:%.*]] = sub i32 [[TMP10]], [[J_022]]
+; CHECK-NEXT: [[TMP6:%.*]] = sub i32 [[TMP5]], [[J_022]]
; CHECK-NEXT: [[TMP7:%.*]] = zext i32 [[TMP6]] to i64
; CHECK-NEXT: [[TMP8:%.*]] = add nuw nsw i64 [[TMP7]], 1
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP8]], 4
@@ -375,7 +372,7 @@ define void @multiple_uniform_stores(ptr nocapture %var1, ptr nocapture readonly
; CHECK: vector.memcheck:
; CHECK-NEXT: [[TMP9:%.*]] = shl nuw nsw i64 [[TMP4]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[VAR2]], i64 [[TMP9]]
-; CHECK-NEXT: [[TMP11:%.*]] = sub i32 [[TMP5]], [[J_022]]
+; CHECK-NEXT: [[TMP11:%.*]] = sub i32 [[TMP10]], [[J_022]]
; CHECK-NEXT: [[TMP12:%.*]] = zext i32 [[TMP11]] to i64
; CHECK-NEXT: [[TMP13:%.*]] = shl nuw nsw i64 [[TMP12]], 2
; CHECK-NEXT: [[TMP14:%.*]] = add i64 [[TMP9]], [[TMP13]]
diff --git a/llvm/test/Transforms/LoopVectorize/metadata.ll b/llvm/test/Transforms/LoopVectorize/metadata.ll
index 4cc50326cdb2d5..76b790d4279355 100644
--- a/llvm/test/Transforms/LoopVectorize/metadata.ll
+++ b/llvm/test/Transforms/LoopVectorize/metadata.ll
@@ -501,8 +501,8 @@ define void @noalias_metadata(ptr align 8 %dst, ptr align 8 %src) {
; CHECK-LABEL: define void @noalias_metadata(
; CHECK-SAME: ptr align 8 [[DST:%.*]], ptr align 8 [[SRC:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
-; CHECK-NEXT: [[SRC2:%.*]] = ptrtoaddr ptr [[SRC]] to i64
; CHECK-NEXT: [[DST1:%.*]] = ptrtoaddr ptr [[DST]] to i64
+; CHECK-NEXT: [[SRC2:%.*]] = ptrtoaddr ptr [[SRC]] to i64
; CHECK-NEXT: [[TMP0:%.*]] = sub i64 [[DST1]], [[SRC2]]
; CHECK-NEXT: [[TMP1:%.*]] = lshr i64 [[TMP0]], 3
; CHECK-NEXT: [[TMP2:%.*]] = add nuw nsw i64 [[TMP1]], 1
@@ -510,8 +510,8 @@ define void @noalias_metadata(ptr align 8 %dst, ptr align 8 %src) {
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 8
-; CHECK-NEXT: [[TMP3:%.*]] = add i64 [[DST1]], 8
-; CHECK-NEXT: [[TMP4:%.*]] = sub i64 [[TMP3]], [[SRC2]]
+; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[DST1]], 8
+; CHECK-NEXT: [[TMP4:%.*]] = sub i64 [[TMP11]], [[SRC2]]
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP4]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[DST]], [[SCEVGEP3]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[SRC]], [[SCEVGEP]]
@@ -552,8 +552,8 @@ define void @noalias_metadata(ptr align 8 %dst, ptr align 8 %src) {
; INTERLEAVE-LABEL: define void @noalias_metadata(
; INTERLEAVE-SAME: ptr align 8 [[DST:%.*]], ptr align 8 [[SRC:%.*]]) {
; INTERLEAVE-NEXT: [[ENTRY:.*]]:
-; INTERLEAVE-NEXT: [[SRC2:%.*]] = ptrtoaddr ptr [[SRC]] to i64
; INTERLEAVE-NEXT: [[DST1:%.*]] = ptrtoaddr ptr [[DST]] to i64
+; INTERLEAVE-NEXT: [[SRC2:%.*]] = ptrtoaddr ptr [[SRC]] to i64
; INTERLEAVE-NEXT: [[TMP0:%.*]] = sub i64 [[DST1]], [[SRC2]]
; INTERLEAVE-NEXT: [[TMP1:%.*]] = lshr i64 [[TMP0]], 3
; INTERLEAVE-NEXT: [[TMP2:%.*]] = add nuw nsw i64 [[TMP1]], 1
@@ -561,8 +561,8 @@ define void @noalias_metadata(ptr align 8 %dst, ptr align 8 %src) {
; INTERLEAVE-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; INTERLEAVE: [[VECTOR_MEMCHECK]]:
; INTERLEAVE-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 8
-; INTERLEAVE-NEXT: [[TMP3:%.*]] = add i64 [[DST1]], 8
-; INTERLEAVE-NEXT: [[TMP4:%.*]] = sub i64 [[TMP3]], [[SRC2]]
+; INTERLEAVE-NEXT: [[TMP12:%.*]] = add i64 [[DST1]], 8
+; INTERLEAVE-NEXT: [[TMP4:%.*]] = sub i64 [[TMP12]], [[SRC2]]
; INTERLEAVE-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP4]]
; INTERLEAVE-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[DST]], [[SCEVGEP3]]
; INTERLEAVE-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[SRC]], [[SCEVGEP]]
diff --git a/llvm/test/Transforms/LoopVectorize/opaque-ptr.ll b/llvm/test/Transforms/LoopVectorize/opaque-ptr.ll
index 37dd2afb7da4c5..ebcf8407da35f8 100644
--- a/llvm/test/Transforms/LoopVectorize/opaque-ptr.ll
+++ b/llvm/test/Transforms/LoopVectorize/opaque-ptr.ll
@@ -20,9 +20,7 @@ define void @test_ptr_iv_no_inbounds(ptr %p1.start, ptr %p2.start, ptr %p1.end)
; CHECK-NEXT: [[IDENT_CHECK:%.*]] = icmp ne i64 [[TMP7]], 0
; CHECK-NEXT: br i1 [[IDENT_CHECK]], label [[SCALAR_PH]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP15:%.*]] = add i64 [[TMP8]], -4
-; CHECK-NEXT: [[TMP9:%.*]] = sub i64 [[TMP15]], [[P1_END1]]
-; CHECK-NEXT: [[TMP11:%.*]] = and i64 [[TMP9]], -4
+; CHECK-NEXT: [[TMP11:%.*]] = and i64 [[TMP1]], -4
; CHECK-NEXT: [[TMP12:%.*]] = add i64 [[TMP11]], 4
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[P1_END]], i64 [[TMP12]]
; CHECK-NEXT: [[SCEVGEP5:%.*]] = getelementptr i8, ptr [[P2_START:%.*]], i64 [[TMP12]]
@@ -94,20 +92,18 @@ define void @test_ptr_iv_with_inbounds(ptr %p1.start, ptr %p2.start, ptr %p1.end
; CHECK-NEXT: entry:
; CHECK-NEXT: [[P1_END1:%.*]] = ptrtoaddr ptr [[P1_END:%.*]] to i64
; CHECK-NEXT: [[TMP4:%.*]] = ptrtoaddr ptr [[P1_START:%.*]] to i64
-; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[TMP4]], -4
-; CHECK-NEXT: [[TMP1:%.*]] = sub i64 [[TMP0]], [[P1_END1]]
+; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[P1_END1]], -4
+; CHECK-NEXT: [[TMP1:%.*]] = sub i64 [[TMP5]], [[TMP4]]
; CHECK-NEXT: [[TMP2:%.*]] = lshr i64 [[TMP1]], 2
; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[TMP2]], 1
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP3]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[TMP4]], -4
-; CHECK-NEXT: [[TMP5:%.*]] = sub i64 [[TMP11]], [[P1_END1]]
-; CHECK-NEXT: [[TMP7:%.*]] = and i64 [[TMP5]], -4
+; CHECK-NEXT: [[TMP7:%.*]] = and i64 [[TMP1]], -4
; CHECK-NEXT: [[TMP8:%.*]] = add i64 [[TMP7]], 4
-; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[P1_END]], i64 [[TMP8]]
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[P1_START]], i64 [[TMP8]]
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[P2_START:%.*]], i64 [[TMP8]]
-; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[P1_END]], [[SCEVGEP3]]
+; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[P1_START]], [[SCEVGEP3]]
; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[P2_START]], [[SCEVGEP]]
; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
; CHECK-NEXT: br i1 [[FOUND_CONFLICT]], label [[SCALAR_PH]], label [[VECTOR_PH:%.*]]
@@ -115,13 +111,13 @@ define void @test_ptr_iv_with_inbounds(ptr %p1.start, ptr %p2.start, ptr %p1.end
; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i64 [[TMP3]], 1
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP3]], [[N_MOD_VF]]
; CHECK-NEXT: [[TMP9:%.*]] = shl i64 [[N_VEC]], 2
-; CHECK-NEXT: [[TMP12:%.*]] = getelementptr i8, ptr [[P1_END]], i64 [[TMP9]]
+; CHECK-NEXT: [[TMP14:%.*]] = getelementptr i8, ptr [[P1_START]], i64 [[TMP9]]
; CHECK-NEXT: [[IND_END6:%.*]] = getelementptr i8, ptr [[P2_START]], i64 [[TMP9]]
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[OFFSET_IDX:%.*]] = shl i64 [[INDEX]], 2
-; CHECK-NEXT: [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[P1_END]], i64 [[OFFSET_IDX]]
+; CHECK-NEXT: [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[P1_START]], i64 [[OFFSET_IDX]]
; CHECK-NEXT: [[NEXT_GEP6:%.*]] = getelementptr i8, ptr [[P2_START]], i64 [[OFFSET_IDX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x float>, ptr [[NEXT_GEP]], align 4, !alias.scope [[META9:![0-9]+]], !noalias [[META12:![0-9]+]]
; CHECK-NEXT: [[WIDE_LOAD10:%.*]] = load <2 x float>, ptr [[NEXT_GEP6]], align 4, !alias.scope [[META12]]
@@ -134,7 +130,7 @@ define void @test_ptr_iv_with_inbounds(ptr %p1.start, ptr %p2.start, ptr %p1.end
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP3]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label [[EXIT:%.*]], label [[SCALAR_PH]]
; CHECK: scalar.ph:
-; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi ptr [ [[TMP12]], [[MIDDLE_BLOCK]] ], [ [[P1_END]], [[ENTRY:%.*]] ], [ [[P1_END]], [[VECTOR_MEMCHECK]] ]
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi ptr [ [[TMP14]], [[MIDDLE_BLOCK]] ], [ [[P1_START]], [[ENTRY:%.*]] ], [ [[P1_START]], [[VECTOR_MEMCHECK]] ]
; CHECK-NEXT: [[BC_RESUME_VAL7:%.*]] = phi ptr [ [[IND_END6]], [[MIDDLE_BLOCK]] ], [ [[P2_START]], [[ENTRY]] ], [ [[P2_START]], [[VECTOR_MEMCHECK]] ]
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
@@ -146,7 +142,7 @@ define void @test_ptr_iv_with_inbounds(ptr %p1.start, ptr %p2.start, ptr %p1.end
; CHECK-NEXT: store float [[SUM]], ptr [[P1]], align 4
; CHECK-NEXT: [[P1_NEXT]] = getelementptr inbounds float, ptr [[P1]], i64 1
; CHECK-NEXT: [[P2_NEXT]] = getelementptr inbounds float, ptr [[P2]], i64 1
-; CHECK-NEXT: [[C:%.*]] = icmp ne ptr [[P1_NEXT]], [[P1_START]]
+; CHECK-NEXT: [[C:%.*]] = icmp ne ptr [[P1_NEXT]], [[P1_END]]
; CHECK-NEXT: br i1 [[C]], label [[LOOP]], label [[EXIT]], !llvm.loop [[LOOP15:![0-9]+]]
; CHECK: exit:
; CHECK-NEXT: ret void
diff --git a/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll b/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll
index 432571456a8885..afd4fc57c46427 100644
--- a/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll
+++ b/llvm/test/Transforms/LoopVectorize/pointer-select-runtime-checks.ll
@@ -11,8 +11,7 @@ define void @test1_select_invariant(ptr %src.1, ptr %src.2, ptr %dst, i1 %c, i8
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP3:%.*]] = add i8 [[N]], -1
-; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP3]] to i64
+; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP0]] to i64
; CHECK-NEXT: [[TMP5:%.*]] = add nuw nsw i64 [[TMP4]], 1
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST:%.*]], i64 [[TMP5]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[PTR_SEL]], i64 1
@@ -166,8 +165,7 @@ define void @test_loop_dependent_select2(ptr %src.1, ptr %src.2, ptr %dst, i8 %n
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP3:%.*]] = add i8 [[N]], -1
-; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP3]] to i64
+; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP0]] to i64
; CHECK-NEXT: [[TMP5:%.*]] = add nuw nsw i64 [[TMP4]], 1
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST:%.*]], i64 [[TMP5]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[SRC_1:%.*]], i64 1
@@ -197,12 +195,12 @@ define void @test_loop_dependent_select2(ptr %src.1, ptr %src.2, ptr %dst, i8 %n
; CHECK-NEXT: [[TMP10:%.*]] = icmp ult i8 [[TMP8]], [[X]]
; CHECK-NEXT: [[TMP11:%.*]] = select i1 [[TMP9]], ptr [[SRC_1]], ptr [[SRC_2]]
; CHECK-NEXT: [[TMP12:%.*]] = select i1 [[TMP10]], ptr [[SRC_1]], ptr [[SRC_2]]
-; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META11:![0-9]+]]
-; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META11]]
+; CHECK-NEXT: [[TMP20:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META11:![0-9]+]]
+; CHECK-NEXT: [[TMP21:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META11]]
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP7]]
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP8]]
-; CHECK-NEXT: store i8 [[TMP14]], ptr [[TMP15]], align 2, !alias.scope [[META14:![0-9]+]], !noalias [[META16:![0-9]+]]
-; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP16]], align 2, !alias.scope [[META14]], !noalias [[META16]]
+; CHECK-NEXT: store i8 [[TMP20]], ptr [[TMP15]], align 2, !alias.scope [[META14:![0-9]+]], !noalias [[META16:![0-9]+]]
+; CHECK-NEXT: store i8 [[TMP21]], ptr [[TMP16]], align 2, !alias.scope [[META14]], !noalias [[META16]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 2
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP17]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP18:![0-9]+]]
@@ -252,8 +250,7 @@ define void @test_loop_dependent_select_first_ptr_noundef(ptr noundef %src.1, pt
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP3:%.*]] = add i8 [[N]], -1
-; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP3]] to i64
+; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP0]] to i64
; CHECK-NEXT: [[TMP5:%.*]] = add nuw nsw i64 [[TMP4]], 1
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST:%.*]], i64 [[TMP5]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[SRC_1:%.*]], i64 1
@@ -281,12 +278,12 @@ define void @test_loop_dependent_select_first_ptr_noundef(ptr noundef %src.1, pt
; CHECK-NEXT: [[TMP10:%.*]] = icmp ult i8 [[TMP8]], [[X]]
; CHECK-NEXT: [[TMP11:%.*]] = select i1 [[TMP9]], ptr [[SRC_1]], ptr [[SRC_2]]
; CHECK-NEXT: [[TMP12:%.*]] = select i1 [[TMP10]], ptr [[SRC_1]], ptr [[SRC_2]]
-; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META20:![0-9]+]]
-; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META20]]
+; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META20:![0-9]+]]
+; CHECK-NEXT: [[TMP19:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META20]]
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP7]]
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP8]]
-; CHECK-NEXT: store i8 [[TMP14]], ptr [[TMP15]], align 2, !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
-; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP16]], align 2, !alias.scope [[META23]], !noalias [[META25]]
+; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP15]], align 2, !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
+; CHECK-NEXT: store i8 [[TMP19]], ptr [[TMP16]], align 2, !alias.scope [[META23]], !noalias [[META25]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 2
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP17]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP27:![0-9]+]]
@@ -336,8 +333,7 @@ define void @test_loop_dependent_select_second_ptr_noundef(ptr %src.1, ptr nound
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 2
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[TMP3:%.*]] = add i8 [[N]], -1
-; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP3]] to i64
+; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP0]] to i64
; CHECK-NEXT: [[TMP5:%.*]] = add nuw nsw i64 [[TMP4]], 1
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST:%.*]], i64 [[TMP5]]
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[SRC_1:%.*]], i64 1
@@ -365,12 +361,12 @@ define void @test_loop_dependent_select_second_ptr_noundef(ptr %src.1, ptr nound
; CHECK-NEXT: [[TMP10:%.*]] = icmp ult i8 [[TMP8]], [[X]]
; CHECK-NEXT: [[TMP11:%.*]] = select i1 [[TMP9]], ptr [[SRC_1]], ptr [[SRC_2]]
; CHECK-NEXT: [[TMP12:%.*]] = select i1 [[TMP10]], ptr [[SRC_1]], ptr [[SRC_2]]
-; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META29:![0-9]+]]
-; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META29]]
+; CHECK-NEXT: [[TMP18:%.*]] = load i8, ptr [[TMP11]], align 8, !alias.scope [[META29:![0-9]+]]
+; CHECK-NEXT: [[TMP19:%.*]] = load i8, ptr [[TMP12]], align 8, !alias.scope [[META29]]
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP7]]
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr i8, ptr [[DST]], i8 [[TMP8]]
-; CHECK-NEXT: store i8 [[TMP14]], ptr [[TMP15]], align 2, !alias.scope [[META32:![0-9]+]], !noalias [[META34:![0-9]+]]
-; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP16]], align 2, !alias.scope [[META32]], !noalias [[META34]]
+; CHECK-NEXT: store i8 [[TMP18]], ptr [[TMP15]], align 2, !alias.scope [[META32:![0-9]+]], !noalias [[META34:![0-9]+]]
+; CHECK-NEXT: store i8 [[TMP19]], ptr [[TMP16]], align 2, !alias.scope [[META32]], !noalias [[META34]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 2
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP17]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
diff --git a/llvm/test/Transforms/LoopVectorize/pr59319-loop-access-info-invalidation.ll b/llvm/test/Transforms/LoopVectorize/pr59319-loop-access-info-invalidation.ll
index 0caa6b2e1d6210..6a4c3bc2c1fdee 100644
--- a/llvm/test/Transforms/LoopVectorize/pr59319-loop-access-info-invalidation.ll
+++ b/llvm/test/Transforms/LoopVectorize/pr59319-loop-access-info-invalidation.ll
@@ -69,7 +69,7 @@ define void @reduced(ptr %0, ptr %1, i64 %iv, ptr %2, i64 %iv76, i64 %iv93) {
; CHECK-NEXT: [[ARRAYIDX_I_I62:%.*]] = getelementptr i32, ptr [[TMP0]], i64 [[IDXPROM_I_I61]]
; CHECK-NEXT: [[MIN_ITERS_CHECK20:%.*]] = icmp ult i64 [[TMP3]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK20]], label [[SCALAR_PH19:%.*]], label [[VECTOR_MEMCHECK13:%.*]]
-; CHECK: vector.memcheck13:
+; CHECK: vector.memcheck20:
; CHECK-NEXT: [[SCEVGEP14:%.*]] = getelementptr i8, ptr [[TMP1]], i64 4
; CHECK-NEXT: [[TMP10:%.*]] = shl nuw nsw i64 [[IDXPROM_I_I61]], 2
; CHECK-NEXT: [[TMP11:%.*]] = add nuw nsw i64 [[TMP10]], 4
@@ -78,20 +78,20 @@ define void @reduced(ptr %0, ptr %1, i64 %iv, ptr %2, i64 %iv76, i64 %iv93) {
; CHECK-NEXT: [[BOUND117:%.*]] = icmp ult ptr [[ARRAYIDX_I_I62]], [[SCEVGEP14]]
; CHECK-NEXT: [[FOUND_CONFLICT18:%.*]] = and i1 [[BOUND016]], [[BOUND117]]
; CHECK-NEXT: br i1 [[FOUND_CONFLICT18]], label [[SCALAR_PH19]], label [[VECTOR_PH21:%.*]]
-; CHECK: vector.ph21:
+; CHECK: vector.ph24:
; CHECK-NEXT: [[TMP12:%.*]] = and i64 [[TMP3]], 3
; CHECK-NEXT: [[N_VEC22:%.*]] = sub i64 [[TMP3]], [[TMP12]]
; CHECK-NEXT: br label [[VECTOR_BODY23:%.*]]
-; CHECK: vector.body23:
+; CHECK: vector.body26:
; CHECK-NEXT: [[INDEX24:%.*]] = phi i64 [ 0, [[VECTOR_PH21]] ], [ [[INDEX_NEXT25:%.*]], [[VECTOR_BODY23]] ]
; CHECK-NEXT: [[INDEX_NEXT25]] = add nuw i64 [[INDEX24]], 4
; CHECK-NEXT: [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT25]], [[N_VEC22]]
; CHECK-NEXT: br i1 [[TMP13]], label [[MIDDLE_BLOCK26:%.*]], label [[VECTOR_BODY23]], !llvm.loop [[LOOP10:![0-9]+]]
-; CHECK: middle.block26:
+; CHECK: middle.block29:
; CHECK-NEXT: store i32 0, ptr [[TMP1]], align 4, !alias.scope [[META11:![0-9]+]], !noalias [[META14:![0-9]+]]
; CHECK-NEXT: [[CMP_N27:%.*]] = icmp eq i64 [[TMP3]], [[N_VEC22]]
; CHECK-NEXT: br i1 [[CMP_N27]], label [[LOOP_CLEANUP:%.*]], label [[SCALAR_PH19]]
-; CHECK: scalar.ph19:
+; CHECK: scalar.ph18:
; CHECK-NEXT: [[BC_RESUME_VAL28:%.*]] = phi i64 [ [[N_VEC22]], [[MIDDLE_BLOCK26]] ], [ 0, [[LOOP_3_LR_PH]] ], [ 0, [[VECTOR_MEMCHECK13]] ]
; CHECK-NEXT: br label [[LOOP_3:%.*]]
; CHECK: loop.2:
diff --git a/llvm/test/Transforms/LoopVectorize/runtime-check.ll b/llvm/test/Transforms/LoopVectorize/runtime-check.ll
index 21e8a3ccdfa2f7..beb8aa2c998117 100644
--- a/llvm/test/Transforms/LoopVectorize/runtime-check.ll
+++ b/llvm/test/Transforms/LoopVectorize/runtime-check.ll
@@ -310,10 +310,9 @@ define void @different_load_store_pairs(ptr %src.1, ptr %src.2, ptr %dst.1, ptr
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
-; CHECK-NEXT: [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[UMAX]], 2
+; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[TMP0]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST_1:%.*]], i64 [[TMP1]]
-; CHECK-NEXT: [[TMP2:%.*]] = shl i64 [[UMAX]], 3
+; CHECK-NEXT: [[TMP2:%.*]] = shl i64 [[TMP0]], 3
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[DST_2:%.*]], i64 [[TMP2]]
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[SRC_1:%.*]], i64 [[TMP1]]
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[SRC_2:%.*]], i64 [[TMP2]]
@@ -346,11 +345,11 @@ define void @different_load_store_pairs(ptr %src.1, ptr %src.2, ptr %dst.1, ptr
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr i32, ptr [[SRC_1]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META22:![0-9]+]]
; CHECK-NEXT: [[TMP5:%.*]] = getelementptr i64, ptr [[SRC_2]], i64 [[INDEX]]
-; CHECK-NEXT: [[WIDE_LOAD19:%.*]] = load <4 x i64>, ptr [[TMP5]], align 8, !alias.scope [[META25:![0-9]+]]
+; CHECK-NEXT: [[WIDE_LOAD34:%.*]] = load <4 x i64>, ptr [[TMP5]], align 8, !alias.scope [[META25:![0-9]+]]
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr nusw i32, ptr [[DST_1]], i64 [[INDEX]]
; CHECK-NEXT: store <4 x i32> [[WIDE_LOAD]], ptr [[TMP6]], align 4, !alias.scope [[META27:![0-9]+]], !noalias [[META29:![0-9]+]]
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr nusw i64, ptr [[DST_2]], i64 [[INDEX]]
-; CHECK-NEXT: store <4 x i64> [[WIDE_LOAD19]], ptr [[TMP7]], align 8, !alias.scope [[META31:![0-9]+]], !noalias [[META32:![0-9]+]]
+; CHECK-NEXT: store <4 x i64> [[WIDE_LOAD34]], ptr [[TMP7]], align 8, !alias.scope [[META31:![0-9]+]], !noalias [[META32:![0-9]+]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP8]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
@@ -494,8 +493,7 @@ define void @test_scev_check_mul_add_expansion(ptr %out, ptr %in, i32 %len, i32
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_MEMCHECK:%.*]]
; CHECK: vector.memcheck:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[OUT:%.*]], i64 12
-; CHECK-NEXT: [[SMAX:%.*]] = call i32 @llvm.smax.i32(i32 [[LEN]], i32 7)
-; CHECK-NEXT: [[TMP2:%.*]] = zext nneg i32 [[SMAX]] to i64
+; CHECK-NEXT: [[TMP2:%.*]] = zext nneg i32 [[TMP0]] to i64
; CHECK-NEXT: [[TMP3:%.*]] = shl nuw nsw i64 [[TMP2]], 1
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[OUT]], i64 [[TMP3]]
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[IN:%.*]], i64 4
>From 317abefdb0d344bce51f4ec84851552c3b43c7ad Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 15 Sep 2026 10:44:58 +0100
Subject: [PATCH 2/5] !fixup address comments, thanks
---
.../Transforms/Vectorize/LoopVectorize.cpp | 21 +++++++++----------
.../Transforms/Vectorize/VPlanTransforms.cpp | 8 +++++++
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 14 -------------
3 files changed, 18 insertions(+), 25 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index c6062de3042b85..e72d62f1b1d8e3 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7016,14 +7016,6 @@ void LoopVectorizationPlanner::addReductionResultComputation(
RUN_VPLAN_PASS(VPlanTransforms::clearReductionWrapFlags, *Plan);
}
-/// Return true if \p CG's bounds can be expanded in the check block.
-/// VPSCEVExpander only expands AddRecs of loops enclosing the plan's scope.
-static bool boundsAreVPlanExpandable(const RuntimeCheckingPtrGroup &CG,
- ScalarEvolution &SE) {
- return !SE.containsAddRecurrence(CG.Low) &&
- !SE.containsAddRecurrence(CG.High);
-}
-
/// Return true if \p RtPtrChecking's memory checks for \p OrigLoop can be
/// modelled as VPlan recipes.
static bool
@@ -7037,10 +7029,17 @@ canModelMemChecksInVPlan(const RuntimePointerChecking &RtPtrChecking,
if (OrigLoop.getParentLoop())
return false;
+ // Return true if \p CG's bounds can be expanded in the check block.
+ // VPSCEVExpander only expands AddRecs of loops enclosing the plan's scope.
+ auto BoundsAreVPlanExpandable = [&SE](const RuntimeCheckingPtrGroup &CG) {
+ return !SE.containsAddRecurrence(CG.Low) &&
+ !SE.containsAddRecurrence(CG.High);
+ };
+
ArrayRef<RuntimePointerCheck> Checks = RtPtrChecking.getChecks();
- return !Checks.empty() && all_of(Checks, [&SE](const RuntimePointerCheck &C) {
- return boundsAreVPlanExpandable(*C.first, SE) &&
- boundsAreVPlanExpandable(*C.second, SE);
+ return !Checks.empty() && all_of(Checks, [&](const RuntimePointerCheck &C) {
+ return BoundsAreVPlanExpandable(*C.first) &&
+ BoundsAreVPlanExpandable(*C.second);
});
}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index a2f2a2997086e3..c49976d35f0ec2 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -1541,6 +1541,14 @@ static VPSingleDefRecipe *combineRecipe(VPlan &Plan, VPSingleDefRecipe *Def) {
{X, Plan.getConstantInt(APC->getBitWidth(), APC->exactLogBase2())},
*cast<VPRecipeWithIRFlags>(Def), Def->getDebugLoc());
+ // (X >> C) << C -> X & (-1 << C).
+ if (CanCreateNewRecipe &&
+ match(Def, m_Shl(m_LShr(m_VPValue(X), m_VPValue(Y, m_APInt(APC))),
+ m_Deferred(Y))))
+ return Builder.createAnd(
+ X, Plan.getConstantInt(APInt::getAllOnes(APC->getBitWidth()) << *APC),
+ Def->getDebugLoc());
+
if (match(Def, m_Not(m_VPValue(X)))) {
// Try to fold Not into compares by adjusting the predicate in-place.
CmpPredicate Pred;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index edc2f212e25092..7d6a261518998a 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -954,20 +954,6 @@ VPValue *VPSCEVExpander::expand(const SCEV *S) {
}
case scMulExpr: {
auto *MulE = cast<SCEVMulExpr>(S);
-
- // mul(PowerOf2C, udiv(X, PowerOf2C)) == (X >> C) << C -> X & (-1 << C),
- // matching SCEVExpander::visitMulExpr.
- const SCEVConstant *C1, *C2;
- const SCEV *Val;
- if (match(S, m_scev_Mul(m_SCEVConstant(C1),
- m_scev_UDiv(m_SCEV(Val), m_SCEVConstant(C2)))) &&
- C1 == C2 && C1->getAPInt().isPowerOf2()) {
- VPValue *LHS = expand(Val);
- APInt Mask = APInt::getBitsSetFrom(MulE->getType()->getScalarSizeInBits(),
- C1->getAPInt().logBase2());
- return Builder.createAnd(LHS, Builder.getPlan().getConstantInt(Mask), DL);
- }
-
VPIRFlags::WrapFlagsTy WrapFlags(MulE->hasNoUnsignedWrap(),
MulE->hasNoSignedWrap());
SmallVector<VPValue *, 2> Ops;
>From 59a5235edfdd4a61dbdbaf072a7e9485932d9949 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 22 Sep 2026 10:18:00 +0100
Subject: [PATCH 3/5] !fixup adjust to createScalarFreeze -> createFreeze
rename
---
llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 982ac67c17bfd6..11d063948d5686 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1515,8 +1515,8 @@ void VPlanTransforms::addMemoryRuntimeChecks(
VPValue *Start = Expander.expand(CG->Low);
VPValue *End = Expander.expand(CG->High);
if (CG->NeedsFreeze) {
- Start = Builder.createScalarFreeze(Start, DL);
- End = Builder.createScalarFreeze(End, DL);
+ Start = Builder.createFreeze(Start, DL);
+ End = Builder.createFreeze(End, DL);
}
GroupToBounds.try_emplace(CG, Start, End);
}
>From e698ce9ab8181960e7c996b88e1665bde2269c22 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 22 Sep 2026 20:58:34 +0100
Subject: [PATCH 4/5] !fixup update tests
---
.../predicated-inductions-vs-first-order-recurrences.ll | 5 ++---
1 file changed, 2 insertions(+), 3 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/predicated-inductions-vs-first-order-recurrences.ll b/llvm/test/Transforms/LoopVectorize/predicated-inductions-vs-first-order-recurrences.ll
index 0f6ba95f9aee31..4777d6ef825b2f 100644
--- a/llvm/test/Transforms/LoopVectorize/predicated-inductions-vs-first-order-recurrences.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicated-inductions-vs-first-order-recurrences.ll
@@ -907,11 +907,10 @@ define void @for_and_ind_indupdate_feeds_gep_index(ptr %dst, ptr %dst2, i64 %n)
; CHECK-NEXT: br i1 [[TMP7]], label %[[SCALAR_PH]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[DST]], i64 3
-; CHECK-NEXT: [[SMAX1:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT: [[TMP8:%.*]] = mul i64 [[SMAX1]], 3
+; CHECK-NEXT: [[TMP8:%.*]] = mul i64 [[TMP0]], 3
; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[TMP8]], 1
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[DST]], i64 [[TMP9]]
-; CHECK-NEXT: [[TMP10:%.*]] = mul i64 [[SMAX1]], 24
+; CHECK-NEXT: [[TMP10:%.*]] = mul i64 [[TMP0]], 24
; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[TMP10]], -16
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[DST2]], i64 [[TMP11]]
; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP3]]
>From 4b5451ead95edecaecc89dffafa29249b615a2cd Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 22 Sep 2026 22:11:25 +0100
Subject: [PATCH 5/5] !fixup preserve branch weights for expanded selects
---
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 7 ++--
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 16 +++++++++-
.../LoopVectorize/scev-check-unknown-prof.ll | 32 +++++++++----------
3 files changed, 36 insertions(+), 19 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 38c94512f05464..efbec89c901174 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -808,8 +808,11 @@ Value *VPInstruction::generate(VPTransformState &State) {
OnlyFirstLaneUsed || vputils::isSingleScalar(getOperand(0)));
Value *Op1 = State.get(getOperand(1), OnlyFirstLaneUsed);
Value *Op2 = State.get(getOperand(2), OnlyFirstLaneUsed);
- return Builder.CreateSelectFMF(Cond, Op1, Op2, getFastMathFlagsOrNone(),
- Name);
+ Value *Sel =
+ Builder.CreateSelectFMF(Cond, Op1, Op2, getFastMathFlagsOrNone(), Name);
+ if (auto *I = dyn_cast<Instruction>(Sel))
+ applyMetadata(*I);
+ return Sel;
}
case VPInstruction::ActiveLaneMask:
case VPInstruction::WideActiveLaneMask: {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 7d6a261518998a..ec301abd0b58ca 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -23,6 +23,7 @@
#include "llvm/Analysis/ScalarEvolutionExpressions.h"
#include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
#include "llvm/IR/Dominators.h"
+#include "llvm/IR/MDBuilder.h"
#include "llvm/IR/ProfDataUtils.h"
#include "llvm/Transforms/Utils/ScalarEvolutionExpander.h"
@@ -1083,7 +1084,20 @@ VPValue *VPSCEVExpander::expand(const SCEV *S) {
// pointer-typed min/max as cmp + select, matching SCEVExpander.
VPValue *Cmp = Builder.createICmp(
MinMaxIntrinsic::getPredicate(IntrinsicID), Result, Op, DL);
- Result = Builder.createSelect(Cmp, Result, Op, DL);
+ VPInstruction *Sel = Builder.createSelect(Cmp, Result, Op, DL);
+ Function *F =
+ Builder.getPlan().getScalarHeader()->getIRBasicBlock()->getParent();
+ std::optional<uint64_t> EC = F->getEntryCount();
+ if (EC && *EC > 0) {
+ MDBuilder MDB(SE.getContext());
+ Sel->setMetadata(
+ LLVMContext::MD_prof,
+ MDNode::get(SE.getContext(),
+ {MDB.createString(
+ MDProfLabels::UnknownBranchWeightsMarker),
+ MDB.createString("scev-expander")}));
+ }
+ Result = Sel;
} else {
Result = Builder.createScalarIntrinsic(IntrinsicID, {Result, Op},
ResultTy, DL);
diff --git a/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll b/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll
index f2d2f13a39a6bf..05ef106661e4c3 100644
--- a/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll
+++ b/llvm/test/Transforms/LoopVectorize/scev-check-unknown-prof.ll
@@ -16,12 +16,12 @@ define void @wrap_check(i32 %n, i32 %step) !prof !0 {
; CHECK: br i1 [[TMP16:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK: [[VECTOR_BODY:.*]]:
-; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP1:![0-9]+]]
+; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP2:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK: br i1 [[CMP_N:%.*]], label %[[EXIT_LOOPEXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK: [[LOOP:.*]]:
-; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[EXIT_LOOPEXIT]]:
; CHECK: [[EXIT]]:
;
@@ -61,17 +61,17 @@ define void @runtime_step_memcheck(ptr %in, ptr %out, i64 %n, i64 %step) !prof !
; CHECK: [[TMP9:%.*]] = select i1 [[TMP3]], i1 [[TMP8:%.*]], i1 [[TMP7:%.*]], !prof [[PROF1]]
; CHECK: br i1 [[TMP16:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK: [[UMIN:%.*]] = select i1 [[TMP23:%.*]], ptr [[IN]], ptr [[SCEVGEP1:%.*]], !prof [[PROF1]]
-; CHECK: [[UMAX:%.*]] = select i1 [[TMP24:%.*]], ptr [[IN]], ptr [[SCEVGEP1]], !prof [[PROF1]]
+; CHECK: [[TMP26:%.*]] = select i1 [[TMP25:%.*]], ptr [[IN]], ptr [[TMP24:%.*]], !prof [[PROF1]]
+; CHECK: [[TMP28:%.*]] = select i1 [[TMP27:%.*]], ptr [[IN]], ptr [[TMP24]], !prof [[PROF1]]
; CHECK: br i1 [[FOUND_CONFLICT:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK: [[VECTOR_BODY:.*]]:
-; CHECK: br i1 [[TMP47:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK: br i1 [[TMP52:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK: br i1 [[CMP_N:%.*]], label %[[EXIT_LOOPEXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK: [[LOOP:.*]]:
-; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP12:![0-9]+]]
; CHECK: [[EXIT_LOOPEXIT]]:
; CHECK: [[EXIT]]:
;
@@ -113,12 +113,12 @@ define void @wrap_check_not_profiled(i32 %n, i32 %step) {
; CHECK: br i1 [[TMP16:%.*]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK: [[VECTOR_BODY:.*]]:
-; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK: br i1 [[TMP23:%.*]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK: br i1 [[CMP_N:%.*]], label %[[EXIT_LOOPEXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK: [[LOOP:.*]]:
-; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK: br i1 [[EC:%.*]], label %[[EXIT_LOOPEXIT]], label %[[LOOP]], !llvm.loop [[LOOP14:![0-9]+]]
; CHECK: [[EXIT_LOOPEXIT]]:
; CHECK: [[EXIT]]:
;
@@ -150,12 +150,12 @@ exit:
;.
; CHECK: [[PROF0]] = !{!"function_entry_count", i64 1000}
; CHECK: [[PROF1]] = !{!"unknown", !"scev-expander"}
-; CHECK: [[LOOP1]] = distinct !{[[LOOP1]], [[META2:![0-9]+]], [[META3:![0-9]+]]}
-; CHECK: [[META2]] = !{!"llvm.loop.isvectorized", i32 1}
-; CHECK: [[META3]] = !{!"llvm.loop.unroll.runtime.disable"}
-; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META2]]}
-; CHECK: [[LOOP10]] = distinct !{[[LOOP10]], [[META2]], [[META3]]}
-; CHECK: [[LOOP11]] = distinct !{[[LOOP11]], [[META2]]}
-; CHECK: [[LOOP12]] = distinct !{[[LOOP12]], [[META2]], [[META3]]}
-; CHECK: [[LOOP13]] = distinct !{[[LOOP13]], [[META2]]}
+; CHECK: [[LOOP2]] = distinct !{[[LOOP2]], [[META3:![0-9]+]], [[META4:![0-9]+]]}
+; CHECK: [[META3]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META4]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META3]]}
+; CHECK: [[LOOP11]] = distinct !{[[LOOP11]], [[META3]], [[META4]]}
+; CHECK: [[LOOP12]] = distinct !{[[LOOP12]], [[META3]]}
+; CHECK: [[LOOP13]] = distinct !{[[LOOP13]], [[META3]], [[META4]]}
+; CHECK: [[LOOP14]] = distinct !{[[LOOP14]], [[META3]]}
;.
More information about the llvm-commits
mailing list