[llvm] [SLP]Support copyable GEP/zext main ops and relax gather pointer compatibility (PR #221599)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 13 10:18:59 PDT 2026
https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/221599
>From ca6485493dddebb92ee5c5061059c4c2c4bc22c0 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sun, 6 Sep 2026 12:27:30 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 282 ++++++++++++------
.../SLPCompatibilityAnalysis.cpp | 21 +-
.../Vectorize/SLPVectorizer/SLPUtils.cpp | 92 ++++++
.../Vectorize/SLPVectorizer/SLPUtils.h | 34 +++
.../AArch64/InstructionsState-is-invalid-0.ll | 21 +-
.../AArch64/long-non-power-of-2.ll | 66 ++--
.../runtime-alias-checks-scheduled-order.ll | 13 +-
.../RISCV/gather-node-with-no-users.ll | 2 +-
.../gather-runtime-stride-with-base-lane.ll | 33 +-
.../RISCV/gep-mixed-index-types.ll | 56 +---
.../SLPVectorizer/RISCV/revec-strided-load.ll | 12 +-
.../X86/copyable-operands-reordering.ll | 39 ++-
.../X86/copyable-zext-poison-lane.ll | 8 +-
.../SLPVectorizer/X86/split-node-throttled.ll | 22 +-
llvm/test/Transforms/SLPVectorizer/revec.ll | 12 +-
15 files changed, 450 insertions(+), 263 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 6d1241fe4408f..73da3c3dba7d6 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -6762,9 +6762,15 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
if (!IsMaskedGatherLegal())
return LoadsState::Gather;
- if (!all_of(PointerOps, [&](Value *P) {
- return arePointersCompatible(P, PointerOps.front(), *TLI);
- }))
+ // The gather address vector requires pointers of the same type.
+ Value *FrontPtr = PointerOps.front();
+ Type *PtrTy = FrontPtr->getType();
+ if (!all_of(PointerOps,
+ [&](Value *P) {
+ return P->getType() == PtrTy &&
+ arePointersCompatible(P, FrontPtr, *TLI);
+ }) &&
+ !isCopyableGEPAddressVector(PointerOps))
return LoadsState::Gather;
} else {
@@ -10201,7 +10207,7 @@ BoUpSLP::TreeEntry::EntryState BoUpSLP::getScalarsVectorizationState(
case Instruction::BitCast: {
Type *SrcTy = VL0->getOperand(0)->getType();
for (Value *V : VL) {
- if (isa<PoisonValue>(V))
+ if (isa<PoisonValue>(V) || S.isCopyableElement(V))
continue;
Type *Ty = cast<Instruction>(V)->getOperand(0)->getType();
if (Ty != SrcTy || !isValidElementType(Ty, SLPReVec)) {
@@ -10277,12 +10283,14 @@ BoUpSLP::TreeEntry::EntryState BoUpSLP::getScalarsVectorizationState(
return TreeEntry::NeedToGather;
return TreeEntry::Vectorize;
case Instruction::GetElementPtr: {
+ auto IsMatchingGEPLane = [&](Value *V) {
+ return isa<GetElementPtrInst>(V) && !S.isCopyableElement(V);
+ };
// We don't combine GEPs with complicated (nested) indexing.
for (Value *V : VL) {
- auto *I = dyn_cast<GetElementPtrInst>(V);
- if (!I)
+ if (!IsMatchingGEPLane(V))
continue;
- if (I->getNumOperands() != 2) {
+ if (cast<GetElementPtrInst>(V)->getNumOperands() != 2) {
LLVM_DEBUG(dbgs() << "SLP: not-vectorizable GEP (nested indexes).\n");
return TreeEntry::NeedToGather;
}
@@ -10293,7 +10301,7 @@ BoUpSLP::TreeEntry::EntryState BoUpSLP::getScalarsVectorizationState(
Type *Ty0 = cast<GEPOperator>(VL0)->getSourceElementType();
for (Value *V : VL) {
auto *GEP = dyn_cast<GEPOperator>(V);
- if (!GEP)
+ if (!GEP || S.isCopyableElement(V))
continue;
Type *CurTy = GEP->getSourceElementType();
if (Ty0 != CurTy) {
@@ -10305,21 +10313,25 @@ BoUpSLP::TreeEntry::EntryState BoUpSLP::getScalarsVectorizationState(
// We don't combine GEPs with non-constant indexes.
Type *Ty1 = VL0->getOperand(1)->getType();
for (Value *V : VL) {
- auto *I = dyn_cast<GetElementPtrInst>(V);
- if (!I)
+ if (!IsMatchingGEPLane(V))
continue;
- auto *Op = I->getOperand(1);
+ auto *Op = cast<GetElementPtrInst>(V)->getOperand(1);
if ((!IsScatterVectorizeUserTE && !isa<ConstantInt>(Op)) ||
(Op->getType() != Ty1 &&
- ((IsScatterVectorizeUserTE && !isa<ConstantInt>(Op)) ||
- Op->getType()->getScalarSizeInBits() >
- DL->getIndexSizeInBits(
- V->getType()->getPointerAddressSpace())))) {
+ Op->getType()->getScalarSizeInBits() >
+ DL->getIndexSizeInBits(
+ V->getType()->getPointerAddressSpace()))) {
LLVM_DEBUG(
dbgs() << "SLP: not-vectorizable GEP (non-constant indexes).\n");
return TreeEntry::NeedToGather;
}
}
+ // The indices must be representable in a single type, the operands
+ // builder casts the constants to it.
+ if (!getCommonGEPIndexType(VL, VL0, IsMatchingGEPLane, *DL)) {
+ LLVM_DEBUG(dbgs() << "SLP: not-vectorizable GEP (mixed index types).\n");
+ return TreeEntry::NeedToGather;
+ }
return TreeEntry::Vectorize;
}
@@ -11079,12 +11091,19 @@ class InstructionsCompatibilityAnalysis {
/// call with a well-defined idempotent value (FP min/max lacks one because of
/// NaNs). An extractelement with constant index from a fixed vector is also
/// supported: the matching lanes reuse the source vector and the copyable
- /// lanes are inserted into it.
+ /// lanes are inserted into it. A single-index scalar GEP is supported: the
+ /// copyable lanes are modeled as zero-index GEPs of the lane values
+ /// themselves. A scalar zext is supported: the copyable lanes are constants
+ /// modeled as zexts of truncated constants.
static bool isSupportedMainOp(Instruction *I) {
return isSupportedOpcode(I->getOpcode()) || isa<MinMaxIntrinsic>(I) ||
RecurrenceDescriptor::isFMulAddIntrinsic(I) ||
(isa<ExtractElementInst>(I) && isVectorLikeInstWithConstOps(I) &&
- isa<FixedVectorType>(I->getOperand(0)->getType()));
+ isa<FixedVectorType>(I->getOperand(0)->getType())) ||
+ (isa<GetElementPtrInst>(I) && I->getNumOperands() == 2 &&
+ !I->getType()->isVectorTy()) ||
+ (isa<ZExtInst>(I) && !I->getType()->isVectorTy() &&
+ !isa<Constant>(I->getOperand(0)));
}
/// Identifies the best candidate value, which represents main opcode
@@ -11186,6 +11205,31 @@ class InstructionsCompatibilityAnalysis {
continue;
}
UsedOutside = PUsedOutside;
+ if (P.first == Instruction::GetElementPtr) {
+ // Only the GEPs of the main op shape (source element type and index
+ // type) are matching lanes, the others are copyable. Select the most
+ // frequent shape to minimize the number of copyable lanes.
+ SmallDenseMap<std::pair<Type *, Type *>, unsigned, 4> ShapeCounts;
+ auto GetShape = [](Instruction *I) {
+ return std::make_pair(cast<GEPOperator>(I)->getSourceElementType(),
+ I->getOperand(1)->getType());
+ };
+ for (Instruction *I : P.second)
+ if (IsSupportedInstruction(I, AnyUndef))
+ ++ShapeCounts[GetShape(I)];
+ unsigned BestShapeNum = 0;
+ for (Instruction *I : P.second) {
+ if (!IsSupportedInstruction(I, AnyUndef))
+ continue;
+ unsigned ShapeNum = ShapeCounts.lookup(GetShape(I));
+ if (ShapeNum > BestShapeNum) {
+ MainOp = I;
+ BestOpcodeNum = P.second.size();
+ BestShapeNum = ShapeNum;
+ }
+ }
+ continue;
+ }
for (Instruction *I : P.second) {
if (IsSupportedInstruction(I, AnyUndef)) {
MainOp = I;
@@ -11223,8 +11267,17 @@ class InstructionsCompatibilityAnalysis {
/// instruction and its actual operands should be returned, or it is a
/// copyable element and its should be represented as idempotent instruction.
SmallVector<Value *> getOperands(const InstructionsState &S, Value *V) const {
- if (isa<PoisonValue>(V))
- return {V, V};
+ if (isa<PoisonValue>(V)) {
+ // Poison operands of the main op operand types: a GEP index or a cast
+ // source has a type, different from the lane type. Excludes the
+ // trailing callee operand of a call.
+ auto *CI = dyn_cast<CallInst>(MainOp);
+ return map_to_vector(
+ seq<unsigned>(CI ? CI->arg_size() : MainOp->getNumOperands()),
+ [&](unsigned Idx) -> Value * {
+ return PoisonValue::get(MainOp->getOperand(Idx)->getType());
+ });
+ }
if (!S.isCopyableElement(V))
return convertTo(cast<Instruction>(V), S).second;
if (RecurrenceDescriptor::isFMulAddIntrinsic(MainOp)) {
@@ -11245,10 +11298,47 @@ class InstructionsCompatibilityAnalysis {
// fmuladd(0.0, -0.0, V) == V.
return {ConstantFP::getZero(Ty), ConstantFP::getNegativeZero(Ty), V};
}
+ // gep V, 0 == V.
+ if (MainOpcode == Instruction::GetElementPtr)
+ return {V, ConstantInt::get(MainOp->getOperand(1)->getType(), 0)};
+ // zext(trunc(V)) == V for the round-trippable constant lanes.
+ if (MainOpcode == Instruction::ZExt)
+ return {ConstantExpr::getTrunc(cast<Constant>(V),
+ cast<CastInst>(MainOp)->getSrcTy())};
assert(isSupportedMainOp(MainOp) && "Unsupported opcode");
return {V, selectBestIdempotentValue()};
}
+ /// Builds the index operands of a GEP node with the main op \p VL0. The
+ /// indices of the matching GEP lanes (for which \p IsGEPLane returns true)
+ /// are cast to the common index type, all other lanes (non-GEP pointers,
+ /// copyable and poison lanes, modeled as gep V, 0) get a zero index of that
+ /// type. Required to be able to find correct matches between different
+ /// gather nodes and reuse the vectorized values rather than trying to
+ /// gather them again. The vectorized nodes always have the common index
+ /// type (a legality invariant); the operands built for the profitability
+ /// analysis before that check may have none, the constants are cast to the
+ /// pointer index type then.
+ void buildGEPIndexOperands(ArrayRef<Value *> VL, Instruction *VL0,
+ function_ref<bool(Value *)> IsGEPLane,
+ BoUpSLP::ValueList &Indices) const {
+ constexpr unsigned IndexIdx = 1;
+ Type *Ty = getCommonGEPIndexType(VL, VL0, IsGEPLane, DL);
+ if (!Ty)
+ Ty = DL.getIndexType(VL0->getOperand(0)->getType()->getScalarType());
+ for (auto [Idx, V] : enumerate(VL)) {
+ if (!IsGEPLane(V)) {
+ Indices[Idx] = ConstantInt::getNullValue(Ty);
+ continue;
+ }
+ auto *Op = cast<GetElementPtrInst>(V)->getOperand(IndexIdx);
+ auto *CI = dyn_cast<ConstantInt>(Op);
+ Indices[Idx] = CI ? ConstantFoldIntegerCast(
+ CI, Ty, CI->getValue().isSignBitSet(), DL)
+ : Op;
+ }
+ }
+
/// Builds operands for the original instructions.
void
buildOriginalOperands(const InstructionsState &S, ArrayRef<Value *> VL,
@@ -11382,37 +11472,11 @@ class InstructionsCompatibilityAnalysis {
return;
case Instruction::GetElementPtr: {
Operands.assign(2, {VL.size(), nullptr});
- // Need to cast all indices to the same type before vectorization to
- // avoid crash.
- // Required to be able to find correct matches between different gather
- // nodes and reuse the vectorized values rather than trying to gather them
- // again.
- const unsigned IndexIdx = 1;
- Type *VL0Ty = VL0->getOperand(IndexIdx)->getType();
- Type *Ty =
- all_of(VL,
- [&](Value *V) {
- auto *GEP = dyn_cast<GetElementPtrInst>(V);
- return !GEP || VL0Ty == GEP->getOperand(IndexIdx)->getType();
- })
- ? VL0Ty
- : DL.getIndexType(cast<GetElementPtrInst>(VL0)
- ->getPointerOperandType()
- ->getScalarType());
for (auto [Idx, V] : enumerate(VL)) {
auto *GEP = dyn_cast<GetElementPtrInst>(V);
- if (!GEP) {
- Operands[0][Idx] = V;
- Operands[1][Idx] = ConstantInt::getNullValue(Ty);
- continue;
- }
- Operands[0][Idx] = GEP->getPointerOperand();
- auto *Op = GEP->getOperand(IndexIdx);
- auto *CI = dyn_cast<ConstantInt>(Op);
- Operands[1][Idx] = CI ? ConstantFoldIntegerCast(
- CI, Ty, CI->getValue().isSignBitSet(), DL)
- : Op;
+ Operands[0][Idx] = GEP ? GEP->getPointerOperand() : V;
}
+ buildGEPIndexOperands(VL, VL0, IsaPred<GetElementPtrInst>, Operands[1]);
return;
}
case Instruction::Call: {
@@ -11696,6 +11760,16 @@ class InstructionsCompatibilityAnalysis {
return S;
InstructionsState OrigS = S;
S = InstructionsState(MainOp, MainOp, /*HasCopyables=*/true);
+ // Cast nodes support only matching lanes from the main op block and
+ // copyable (round-trippable constant) lanes.
+ if (isa<CastInst>(MainOp) && any_of(VL, [&](Value *V) {
+ auto *I = dyn_cast<Instruction>(V);
+ return !isa<PoisonValue>(V) &&
+ !(I && I->getOpcode() == MainOpcode &&
+ I->getParent() == MainOp->getParent()) &&
+ !S.isCopyableElement(V);
+ }))
+ return OrigS;
if (OrigS && !isCopyablePreferable(VL, R, OrigS, S))
return OrigS;
if (!WithProfitabilityCheck)
@@ -11715,6 +11789,16 @@ class InstructionsCompatibilityAnalysis {
// Check if it is profitable to vectorize the instruction.
unsigned CopyableNum =
count_if(VL, [&](Value *V) { return S.isCopyableElement(V); });
+ // A cast copyable node just extends the sources of the matching lanes:
+ // the constant lanes are free in a gather of the original values, so it
+ // cannot be better unless the matching lanes are the majority and are not
+ // vectorized already (the gather reuses the vector otherwise).
+ if (isa<CastInst>(MainOp) &&
+ (CopyableNum * 2 >= VL.size() || none_of(VL, [&](Value *V) {
+ return !S.isCopyableElement(V) && !isa<PoisonValue>(V) &&
+ !R.isVectorized(V);
+ })))
+ return OrigS;
// Absorb copyable single-use fmuls/fadds as fmuladd(a, b, -0.0) or
// fmuladd(1.0, a, b) when every copyable is such a binop: the binops die
// instead of being computed and gathered.
@@ -11722,6 +11806,46 @@ class InstructionsCompatibilityAnalysis {
AbsorbCopyableFMulOrFAdds)
S.setAbsorbCopyableFMulOrFAdd(true);
SmallVector<BoUpSLP::ValueList> Operands = buildOperands(S, VL);
+ auto CheckOperand = [&](ArrayRef<Value *> Ops) {
+ if (allConstant(Ops) || isSplat(Ops))
+ return true;
+ // Non-instruction operands of a call (args, constants) are always a
+ // trivial gather, same as the constant/splat cases above.
+ if (MainOpcode == Instruction::Call && none_of(Ops, IsaPred<Instruction>))
+ return true;
+ // Check if it is "almost" splat, i.e. has >= 4 elements and only single
+ // one is different.
+ constexpr unsigned Limit = 4;
+ if (Operands.front().size() >= Limit) {
+ SmallDenseMap<const Value *, unsigned> Counters;
+ for (Value *V : Ops) {
+ if (isa<UndefValue>(V))
+ continue;
+ ++Counters[V];
+ }
+ if (Counters.size() == 2 &&
+ any_of(Counters, [&](const std::pair<const Value *, unsigned> &C) {
+ return C.second == 1;
+ }))
+ return true;
+ }
+ // First operand not a constant or splat? Last attempt - check for
+ // potential vectorization.
+ InstructionsCompatibilityAnalysis Analysis(DT, DL, TTI, TLI);
+ InstructionsState OpS = Analysis.buildInstructionsState(Ops, R);
+ if (!OpS || (OpS.getOpcode() == Instruction::PHI && !allSameBlock(Ops)))
+ return false;
+ unsigned CopyableNum =
+ count_if(Ops, [&](Value *V) { return OpS.isCopyableElement(V); });
+ return CopyableNum <= VL.size() / 2;
+ };
+ // A copyable GEP node is a vector GEP over the base pointers of the
+ // matching lanes and the values of the copyable lanes. It is profitable
+ // only if that base vector is cheap (a splat, constants or vectorizable),
+ // otherwise it is a gather not cheaper than the gather of the pointers.
+ if (MainOpcode == Instruction::GetElementPtr &&
+ !CheckOperand(Operands.front()))
+ return OrigS;
auto BuildCandidates =
[](SmallVectorImpl<std::pair<Value *, Value *>> &Candidates, Value *V1,
Value *V2) {
@@ -11735,9 +11859,11 @@ class InstructionsCompatibilityAnalysis {
Candidates.emplace_back(V1, (I1 || I2) ? V2 : V1);
};
if (VL.size() == 2) {
- // The operand-pairing heuristic below does not apply to calls; defer
- // to the full tree cost computation instead of pre-rejecting here.
- if (MainOpcode == Instruction::Call)
+ // The operand-pairing heuristic below does not apply to calls, GEPs
+ // and casts; defer to the full tree cost computation instead of
+ // pre-rejecting here.
+ if (MainOpcode == Instruction::Call ||
+ MainOpcode == Instruction::GetElementPtr || isa<CastInst>(MainOp))
return S;
// Check if the operands allow better vectorization.
SmallVector<std::pair<Value *, Value *>, 4> Candidates1, Candidates2;
@@ -11787,8 +11913,10 @@ class InstructionsCompatibilityAnalysis {
return OrigS;
return S;
}
- // fmuladd is the only 3-operand copyable.
+ // Casts are the only 1-operand and fmuladd is the only 3-operand
+ // copyable.
assert((Operands.size() == 2 ||
+ (Operands.size() == 1 && isa<CastInst>(MainOp)) ||
(Operands.size() == 3 &&
RecurrenceDescriptor::isFMulAddIntrinsic(MainOp))) &&
"Unexpected number of operands!");
@@ -11844,39 +11972,6 @@ class InstructionsCompatibilityAnalysis {
if (Operands.size() == 2 &&
count_if(Operands.back(), IsaPred<Instruction>) > 1)
return OrigS;
- auto CheckOperand = [&](ArrayRef<Value *> Ops) {
- if (allConstant(Ops) || isSplat(Ops))
- return true;
- // Non-instruction operands of a call (args, constants) are always a
- // trivial gather, same as the constant/splat cases above.
- if (MainOpcode == Instruction::Call && none_of(Ops, IsaPred<Instruction>))
- return true;
- // Check if it is "almost" splat, i.e. has >= 4 elements and only single
- // one is different.
- constexpr unsigned Limit = 4;
- if (Operands.front().size() >= Limit) {
- SmallDenseMap<const Value *, unsigned> Counters;
- for (Value *V : Ops) {
- if (isa<UndefValue>(V))
- continue;
- ++Counters[V];
- }
- if (Counters.size() == 2 &&
- any_of(Counters, [&](const std::pair<const Value *, unsigned> &C) {
- return C.second == 1;
- }))
- return true;
- }
- // First operand not a constant or splat? Last attempt - check for
- // potential vectorization.
- InstructionsCompatibilityAnalysis Analysis(DT, DL, TTI, TLI);
- InstructionsState OpS = Analysis.buildInstructionsState(Ops, R);
- if (!OpS || (OpS.getOpcode() == Instruction::PHI && !allSameBlock(Ops)))
- return false;
- unsigned CopyableNum =
- count_if(Ops, [&](Value *V) { return OpS.isCopyableElement(V); });
- return CopyableNum <= VL.size() / 2;
- };
// Check the operand holding the copyable values.
if (!CheckOperand(Operands[S.getCopyableOpIdx()])) {
if (Operands.size() == 2)
@@ -11922,6 +12017,15 @@ class InstructionsCompatibilityAnalysis {
for (auto [OperandIdx, Operand] : enumerate(OperandsForValue))
Operands[OperandIdx][Idx] = Operand;
}
+ // The copyable lanes of a GEP node are gep V, 0; cast the indices of
+ // the matching lanes and the zeroes to the common index type.
+ if (MainOpcode == Instruction::GetElementPtr)
+ buildGEPIndexOperands(
+ VL, MainOp,
+ [&](Value *V) {
+ return isa<GetElementPtrInst>(V) && !S.isCopyableElement(V);
+ },
+ Operands[1]);
// Operand-order normalization below swaps OpIdx 0 and OpIdx 1
// of non-copyable lanes. That is only safe when the main op is
// commutative (e.g. 0 - X is not X - 0, so `sub` must be
@@ -34750,6 +34854,16 @@ bool SLPVectorizerPass::vectorizeStoreChains(BoUpSLP &R) {
InstructionsState S = Analysis.buildInstructionsState(
NewVL, R, /*WithProfitabilityCheck=*/true,
/*SkipSameCodeCheck=*/!SameParent);
+ // A copyable GEP node with non-constant indices is vectorizable only as
+ // the address vector of a masked gather, not as a stored value.
+ if (S && S.getOpcode() == Instruction::GetElementPtr &&
+ S.areInstructionsWithCopyableElements() &&
+ any_of(NewVL, [&](Value *V) {
+ auto *GEP = dyn_cast<GetElementPtrInst>(V);
+ return GEP && !S.isCopyableElement(V) &&
+ !isa<ConstantInt>(GEP->getOperand(1));
+ }))
+ return false;
if (S)
return true;
if (!SameParent)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.cpp
index 133abe65208c9..36a757df8049c 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPCompatibilityAnalysis.cpp
@@ -543,11 +543,30 @@ bool InstructionsState::isCopyableElement(Value *V) const {
assert(valid() && "InstructionsState is invalid.");
if (!HasCopyables)
return false;
- if (isAltShuffle() || getOpcode() == Instruction::GetElementPtr)
+ if (isAltShuffle())
return false;
auto *I = dyn_cast<Instruction>(V);
+ // Copyable lanes of a cast node are limited to constants representable as
+ // the cast of a source-type constant.
+ if (isa<CastInst>(MainOp)) {
+ if (I || isa<PoisonValue>(V))
+ return false;
+ if (isa<UndefValue>(V))
+ return true;
+ auto *C = dyn_cast<ConstantInt>(V);
+ return C && C->getValue().getActiveBits() <=
+ cast<CastInst>(MainOp)->getSrcTy()->getIntegerBitWidth();
+ }
if (!I)
return !isa<PoisonValue>(V);
+ // For a GEP main op only single-index GEPs with the same source element
+ // type can be matching lanes; GEPs with a different shape are copyable.
+ if (MainOp->getOpcode() == Instruction::GetElementPtr)
+ if (auto *GEP = dyn_cast<GetElementPtrInst>(I);
+ GEP && (GEP->getNumOperands() != 2 ||
+ GEP->getSourceElementType() !=
+ cast<GetElementPtrInst>(MainOp)->getSourceElementType()))
+ return true;
if (I->getParent() != MainOp->getParent() &&
(!isVectorLikeInstWithConstOps(I) ||
!isVectorLikeInstWithConstOps(MainOp)))
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index af6f8032c7b0b..b238b6aad7a28 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -11,6 +11,7 @@
#include "llvm/ADT/APInt.h"
#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/Sequence.h"
+#include "llvm/ADT/SmallPtrSet.h"
#include "llvm/Analysis/ValueTracking.h"
#include "llvm/Analysis/VectorUtils.h"
#include "llvm/IR/Constants.h"
@@ -614,6 +615,97 @@ bool isSelectedBaseLoad(Type *ScalarTy, ArrayRef<Value *> PointerOps,
return TrueBase != nullptr;
}
+Type *getCommonGEPIndexType(ArrayRef<Value *> VL, Instruction *VL0,
+ function_ref<bool(Value *)> IsGEPLane,
+ const DataLayout &DL) {
+ constexpr unsigned IndexIdx = 1;
+ Type *VL0Ty = VL0->getOperand(IndexIdx)->getType();
+ Type *PtrIdxTy =
+ DL.getIndexType(VL0->getOperand(0)->getType()->getScalarType());
+ bool AllSameTy = true;
+ bool HasNonConstIdx = false;
+ bool ConstsFitVL0Ty = true;
+ for (Value *V : VL) {
+ if (!IsGEPLane(V))
+ continue;
+ Value *Op = cast<GetElementPtrInst>(V)->getOperand(IndexIdx);
+ if (Op->getType() != VL0Ty)
+ AllSameTy = false;
+ auto *CI = dyn_cast<ConstantInt>(Op);
+ if (!CI) {
+ // Non-constant indices are not cast, they must have the main op type.
+ if (Op->getType() != VL0Ty)
+ return nullptr;
+ HasNonConstIdx = true;
+ continue;
+ }
+ if (!CI->getValue().isSignedIntN(VL0Ty->getIntegerBitWidth()))
+ ConstsFitVL0Ty = false;
+ }
+ if (AllSameTy)
+ return VL0Ty;
+ if (!HasNonConstIdx || VL0Ty == PtrIdxTy)
+ return PtrIdxTy;
+ return ConstsFitVL0Ty ? VL0Ty : nullptr;
+}
+
+bool isCopyableGEPAddressVector(ArrayRef<Value *> PointerOps) {
+ SmallPtrSet<Value *, 16> UniquePtrs(llvm::from_range, PointerOps);
+ if (UniquePtrs.size() != PointerOps.size())
+ return false;
+ auto IsConstantOffsetPtr = [](Value *P) {
+ auto *GEP = dyn_cast<GetElementPtrInst>(P);
+ return !GEP ||
+ (GEP->getNumOperands() == 2 && isConstant(GEP->getOperand(1)));
+ };
+ auto *RefIt = find_if_not(PointerOps, IsConstantOffsetPtr);
+ if (RefIt == PointerOps.end())
+ return false;
+ auto *RefGEP = dyn_cast<GetElementPtrInst>(*RefIt);
+ if (!RefGEP || RefGEP->getNumOperands() != 2)
+ return false;
+ Value *Base = RefGEP->getPointerOperand();
+ Type *PtrTy = RefGEP->getType();
+ Type *SrcElemTy = RefGEP->getSourceElementType();
+ // The stride and the (optional) cast opcode of the runtime indices.
+ Value *Stride = nullptr;
+ unsigned CastOpcode = 0;
+ for (Value *P : PointerOps) {
+ if (P->getType() != PtrTy)
+ return false;
+ if (P == Base)
+ continue;
+ auto *GEP = dyn_cast<GetElementPtrInst>(P);
+ if (!GEP || GEP->getNumOperands() != 2 ||
+ GEP->getPointerOperand() != Base ||
+ GEP->getSourceElementType() != SrcElemTy)
+ return false;
+ Value *Idx = GEP->getOperand(1);
+ if (isConstant(Idx))
+ continue;
+ unsigned LaneCastOpcode = 0;
+ if (auto *Cast = dyn_cast<CastInst>(Idx)) {
+ LaneCastOpcode = Cast->getOpcode();
+ Idx = Cast->getOperand(0);
+ }
+ Value *LaneStride = Idx;
+ if (auto *BO = dyn_cast<BinaryOperator>(Idx)) {
+ if (isa<Constant>(BO->getOperand(1)))
+ LaneStride = BO->getOperand(0);
+ else if (isa<Constant>(BO->getOperand(0)))
+ LaneStride = BO->getOperand(1);
+ }
+ if (!Stride) {
+ Stride = LaneStride;
+ CastOpcode = LaneCastOpcode;
+ continue;
+ }
+ if (LaneStride != Stride || LaneCastOpcode != CastOpcode)
+ return false;
+ }
+ return Stride != nullptr;
+}
+
void addMask(SmallVectorImpl<int> &Mask, ArrayRef<int> SubMask,
bool ExtendingManyInputs) {
if (SubMask.empty())
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index e9aeff605eb91..b4bf5da4bf7db 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -18,6 +18,7 @@
#include "llvm/ADT/APInt.h"
#include "llvm/ADT/ArrayRef.h"
+#include "llvm/ADT/STLFunctionalExtras.h"
#include "llvm/ADT/SmallBitVector.h"
#include "llvm/ADT/SmallVector.h"
#include "llvm/Analysis/MemoryLocation.h"
@@ -288,6 +289,39 @@ bool isSelectedBaseLoad(Type *ScalarTy, ArrayRef<Value *> PointerOps,
Value *&FalseBase,
SmallVectorImpl<Value *> &Conditions);
+/// Returns the common type for the indices of the single-index GEP lanes of
+/// a GEP node with the main op \p VL0, or nullptr if no such type exists.
+/// \p IsGEPLane tells which lanes of \p VL are matching GEPs, whose index is
+/// used as is; all other lanes (copyable, poison, non-GEP pointers) are
+/// modeled as gep V, 0 and just take a zero index of the common type.
+/// The common type is the index type of \p VL0 if all matching lanes share
+/// it. Otherwise the constant indices are cast: to the pointer index type if
+/// there are no non-constant indices (or the index type of \p VL0 already is
+/// the pointer index type), or to the index type of \p VL0 if all the
+/// non-constant indices have that type and all the constants are
+/// representable in it (GEP indices are sign-extended to the pointer index
+/// width, so the value must fit as a signed number). A non-constant index of
+/// a different type cannot be cast without a new instruction, so no common
+/// type exists.
+Type *getCommonGEPIndexType(ArrayRef<Value *> VL, Instruction *VL0,
+ function_ref<bool(Value *)> IsGEPLane,
+ const DataLayout &DL);
+
+/// Checks if the pointers \p PointerOps of the gathered loads, which are not
+/// compatible in the usual sense (some of them are constant-offset pointers,
+/// some have runtime indices), still form a cheap address vector as a
+/// copyable GEP node: a splat of the common base and a vector of indices.
+/// The constant-offset lanes are the base itself or single-index GEPs of the
+/// base with a constant index (modeled as gep V, 0 or as matching lanes with
+/// constant indices), the other lanes are single-index GEPs of the base of
+/// the same shape, whose indices are affine in a single runtime value (the
+/// stride): optionally cast (all with the same cast opcode) values, each of
+/// which is the stride itself or a binary operation of the stride and a
+/// constant, like b[i * S] or b[S + i]. Duplicate lanes are not accepted:
+/// the node would be a shuffled non-full vector, gathering the loads is
+/// cheaper then.
+bool isCopyableGEPAddressVector(ArrayRef<Value *> PointerOps);
+
/// Shuffles \p Mask in accordance with the given \p SubMask.
/// \param ExtendingManyInputs Supports reshuffling of the mask with not only
/// one but two input vectors.
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/InstructionsState-is-invalid-0.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/InstructionsState-is-invalid-0.ll
index 2fc7f92a52b49..7820905a0c086 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/InstructionsState-is-invalid-0.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/InstructionsState-is-invalid-0.ll
@@ -7,20 +7,13 @@ target triple = "aarch64-unknown-linux-gnu"
define void @foo(ptr %0) {
; CHECK-LABEL: @foo(
; CHECK-NEXT: vector.scevcheck:
-; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[TMP0:%.*]], i64 4
-; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr null, i64 4
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x ptr> poison, ptr [[TMP0]], i64 1
-; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x ptr> [[TMP1]], ptr [[SCEVGEP]], i64 0
-; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <4 x ptr> [[TMP2]], <4 x ptr> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
-; CHECK-NEXT: [[TMP4:%.*]] = icmp ult <4 x ptr> [[TMP3]], splat (ptr null)
-; CHECK-NEXT: [[TMP5:%.*]] = and <4 x i1> [[TMP4]], zeroinitializer
-; CHECK-NEXT: [[TMP6:%.*]] = insertelement <4 x ptr> poison, ptr [[TMP0]], i64 0
-; CHECK-NEXT: [[TMP7:%.*]] = insertelement <4 x ptr> [[TMP6]], ptr [[SCEVGEP3]], i64 1
-; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <4 x ptr> [[TMP7]], <4 x ptr> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
-; CHECK-NEXT: [[TMP9:%.*]] = icmp ult <4 x ptr> [[TMP8]], splat (ptr null)
-; CHECK-NEXT: [[TMP10:%.*]] = and <4 x i1> [[TMP9]], zeroinitializer
-; CHECK-NEXT: [[RDX_OP:%.*]] = or <4 x i1> [[TMP5]], [[TMP10]]
-; CHECK-NEXT: [[OP_RDX:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[RDX_OP]])
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <3 x ptr> <ptr poison, ptr null, ptr poison>, ptr [[TMP0:%.*]], i64 0
+; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <3 x ptr> [[TMP1]], <3 x ptr> poison, <3 x i32> <i32 0, i32 0, i32 1>
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i8, <3 x ptr> [[TMP2]], <3 x i64> <i64 4, i64 0, i64 4>
+; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <3 x ptr> [[TMP3]], <3 x ptr> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 1, i32 1, i32 1, i32 1, i32 2>
+; CHECK-NEXT: [[TMP5:%.*]] = icmp ult <8 x ptr> [[TMP4]], splat (ptr null)
+; CHECK-NEXT: [[TMP6:%.*]] = and <8 x i1> [[TMP5]], zeroinitializer
+; CHECK-NEXT: [[OP_RDX:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP6]])
; CHECK-NEXT: br i1 [[OP_RDX]], label [[DOTLR_PH:%.*]], label [[VECTOR_PH:%.*]]
; CHECK: vector.ph:
; CHECK-NEXT: ret void
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/long-non-power-of-2.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/long-non-power-of-2.ll
index c6489c2085000..944194a9d6529 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/long-non-power-of-2.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/long-non-power-of-2.ll
@@ -7,8 +7,6 @@ define i1 @test(ptr %arg, ptr %arg1, i64 %arg2, ptr %arg3) {
; CHECK-NEXT: [[BB:.*:]]
; CHECK-NEXT: [[GETELEMENTPTR:%.*]] = getelementptr i8, ptr [[ARG1]], i64 [[ARG2]]
; CHECK-NEXT: [[GETELEMENTPTR4:%.*]] = getelementptr i8, ptr null, i64 0
-; CHECK-NEXT: [[TMP2:%.*]] = insertelement <2 x ptr> <ptr null, ptr poison>, ptr [[ARG3]], i64 1
-; CHECK-NEXT: [[TMP1:%.*]] = getelementptr i8, <2 x ptr> [[TMP2]], <2 x i64> <i64 -32, i64 -432>
; CHECK-NEXT: [[GETELEMENTPTR5:%.*]] = getelementptr i8, ptr null, i64 -32
; CHECK-NEXT: [[TMP32:%.*]] = getelementptr i8, ptr [[ARG3]], i64 -440
; CHECK-NEXT: [[GETELEMENTPTR7:%.*]] = getelementptr i8, ptr [[ARG1]], i64 0
@@ -16,52 +14,50 @@ define i1 @test(ptr %arg, ptr %arg1, i64 %arg2, ptr %arg3) {
; CHECK-NEXT: [[ICMP:%.*]] = icmp ult ptr null, [[GETELEMENTPTR8]]
; CHECK-NEXT: [[AND:%.*]] = and i1 false, [[ICMP]]
; CHECK-NEXT: [[ICMP9:%.*]] = icmp ult ptr [[GETELEMENTPTR7]], null
-; CHECK-NEXT: [[AND10:%.*]] = and i1 [[ICMP9]], false
+; CHECK-NEXT: [[OP_RDX18:%.*]] = and i1 [[ICMP9]], false
; CHECK-NEXT: [[ICMP11:%.*]] = icmp ult ptr [[GETELEMENTPTR7]], null
-; CHECK-NEXT: [[AND37:%.*]] = and i1 [[ICMP11]], false
+; CHECK-NEXT: [[OP_RDX3:%.*]] = and i1 [[ICMP11]], false
; CHECK-NEXT: [[ICMP14:%.*]] = icmp ult ptr [[GETELEMENTPTR4]], [[GETELEMENTPTR8]]
-; CHECK-NEXT: [[AND85:%.*]] = and i1 false, [[ICMP14]]
-; CHECK-NEXT: [[ICMP17:%.*]] = icmp ult ptr null, null
-; CHECK-NEXT: [[ICMP36:%.*]] = icmp ult ptr null, [[GETELEMENTPTR8]]
-; CHECK-NEXT: [[TMP33:%.*]] = and i1 [[ICMP17]], [[ICMP36]]
+; CHECK-NEXT: [[AND15:%.*]] = and i1 false, [[ICMP14]]
+; CHECK-NEXT: [[ICMP71:%.*]] = icmp ult ptr null, null
+; CHECK-NEXT: [[ICMP18:%.*]] = icmp ult ptr null, [[GETELEMENTPTR8]]
+; CHECK-NEXT: [[AND19:%.*]] = and i1 [[ICMP71]], [[ICMP18]]
; CHECK-NEXT: [[ICMP21:%.*]] = icmp ult ptr null, [[GETELEMENTPTR8]]
-; CHECK-NEXT: [[AND22:%.*]] = and i1 false, [[ICMP21]]
+; CHECK-NEXT: [[OP_RDX6:%.*]] = and i1 false, [[ICMP21]]
; CHECK-NEXT: [[ICMP24:%.*]] = icmp ult ptr null, [[GETELEMENTPTR8]]
-; CHECK-NEXT: [[AND25:%.*]] = and i1 false, [[ICMP24]]
-; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x ptr> <ptr poison, ptr poison, ptr null, ptr null>, ptr [[ARG]], i64 1
-; CHECK-NEXT: [[TMP19:%.*]] = insertelement <4 x ptr> [[TMP0]], ptr [[GETELEMENTPTR7]], i64 0
-; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <2 x ptr> [[TMP1]], <2 x ptr> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <4 x ptr> <ptr poison, ptr poison, ptr null, ptr null>, <4 x ptr> [[TMP4]], <4 x i32> <i32 4, i32 5, i32 2, i32 3>
-; CHECK-NEXT: [[TMP20:%.*]] = icmp ult <4 x ptr> [[TMP19]], [[TMP7]]
+; CHECK-NEXT: [[AND86:%.*]] = and i1 false, [[ICMP24]]
+; CHECK-NEXT: [[ICMP39:%.*]] = icmp ult ptr [[GETELEMENTPTR7]], [[GETELEMENTPTR5]]
+; CHECK-NEXT: [[ICMP44:%.*]] = icmp ult ptr [[ARG]], [[GETELEMENTPTR8]]
; CHECK-NEXT: [[ICMP58:%.*]] = icmp ult ptr [[GETELEMENTPTR]], null
; CHECK-NEXT: [[ICMP62:%.*]] = icmp ult ptr [[GETELEMENTPTR]], null
; CHECK-NEXT: [[ICMP84:%.*]] = icmp ult ptr null, [[TMP32]]
; CHECK-NEXT: [[ICMP70:%.*]] = icmp ult ptr [[ARG]], [[GETELEMENTPTR5]]
; CHECK-NEXT: [[ICMP75:%.*]] = icmp ult ptr [[ARG1]], [[TMP32]]
-; CHECK-NEXT: [[TMP5:%.*]] = insertelement <16 x ptr> <ptr poison, ptr poison, ptr poison, ptr poison, ptr null, ptr null, ptr poison, ptr poison, ptr poison, ptr poison, ptr poison, ptr null, ptr null, ptr null, ptr poison, ptr poison>, ptr [[GETELEMENTPTR8]], i64 0
-; CHECK-NEXT: [[TMP6:%.*]] = insertelement <16 x ptr> [[TMP5]], ptr [[TMP32]], i64 8
-; CHECK-NEXT: [[TMP18:%.*]] = shufflevector <16 x ptr> [[TMP6]], <16 x ptr> poison, <16 x i32> <i32 0, i32 0, i32 0, i32 0, i32 4, i32 5, i32 0, i32 0, i32 8, i32 8, i32 8, i32 11, i32 12, i32 13, i32 8, i32 8>
-; CHECK-NEXT: [[TMP8:%.*]] = icmp ult <16 x ptr> splat (ptr null), [[TMP18]]
-; CHECK-NEXT: [[TMP9:%.*]] = shufflevector <4 x i1> [[TMP20]], <4 x i1> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP10:%.*]] = shufflevector <16 x i1> <i1 false, i1 false, i1 false, i1 false, i1 poison, i1 poison, i1 poison, i1 poison, i1 false, i1 poison, i1 poison, i1 poison, i1 poison, i1 poison, i1 false, i1 false>, <16 x i1> [[TMP9]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 16, i32 17, i32 18, i32 19, i32 8, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 14, i32 15>
-; CHECK-NEXT: [[TMP11:%.*]] = insertelement <16 x i1> [[TMP10]], i1 [[ICMP58]], i64 9
-; CHECK-NEXT: [[TMP12:%.*]] = insertelement <16 x i1> [[TMP11]], i1 [[ICMP62]], i64 10
-; CHECK-NEXT: [[TMP13:%.*]] = insertelement <16 x i1> [[TMP12]], i1 [[ICMP84]], i64 11
-; CHECK-NEXT: [[TMP14:%.*]] = insertelement <16 x i1> [[TMP13]], i1 [[ICMP70]], i64 12
-; CHECK-NEXT: [[TMP15:%.*]] = insertelement <16 x i1> [[TMP14]], i1 [[ICMP75]], i64 13
-; CHECK-NEXT: [[TMP16:%.*]] = and <16 x i1> [[TMP15]], [[TMP8]]
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <16 x ptr> <ptr poison, ptr poison, ptr poison, ptr poison, ptr null, ptr null, ptr poison, ptr poison, ptr poison, ptr poison, ptr poison, ptr null, ptr null, ptr null, ptr poison, ptr poison>, ptr [[GETELEMENTPTR8]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <16 x ptr> [[TMP0]], ptr [[TMP32]], i64 8
+; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <16 x ptr> [[TMP1]], <16 x ptr> poison, <16 x i32> <i32 0, i32 0, i32 0, i32 0, i32 4, i32 5, i32 0, i32 0, i32 8, i32 8, i32 8, i32 11, i32 12, i32 13, i32 8, i32 8>
+; CHECK-NEXT: [[TMP3:%.*]] = icmp ult <16 x ptr> splat (ptr null), [[TMP2]]
+; CHECK-NEXT: [[TMP4:%.*]] = insertelement <16 x i1> <i1 false, i1 false, i1 false, i1 false, i1 poison, i1 poison, i1 poison, i1 poison, i1 false, i1 poison, i1 poison, i1 poison, i1 poison, i1 poison, i1 false, i1 false>, i1 [[ICMP39]], i64 4
+; CHECK-NEXT: [[TMP5:%.*]] = insertelement <16 x i1> [[TMP4]], i1 [[ICMP44]], i64 5
+; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <16 x i1> [[TMP5]], <16 x i1> <i1 false, i1 false, i1 undef, i1 undef, i1 undef, i1 undef, i1 undef, i1 undef, i1 undef, i1 undef, i1 undef, i1 undef, i1 undef, i1 undef, i1 undef, i1 undef>, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 16, i32 17, i32 8, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 14, i32 15>
+; CHECK-NEXT: [[TMP7:%.*]] = insertelement <16 x i1> [[TMP6]], i1 [[ICMP58]], i64 9
+; CHECK-NEXT: [[TMP8:%.*]] = insertelement <16 x i1> [[TMP7]], i1 [[ICMP62]], i64 10
+; CHECK-NEXT: [[TMP9:%.*]] = insertelement <16 x i1> [[TMP8]], i1 [[ICMP84]], i64 11
+; CHECK-NEXT: [[TMP10:%.*]] = insertelement <16 x i1> [[TMP9]], i1 [[ICMP70]], i64 12
+; CHECK-NEXT: [[TMP11:%.*]] = insertelement <16 x i1> [[TMP10]], i1 [[ICMP75]], i64 13
+; CHECK-NEXT: [[TMP16:%.*]] = and <16 x i1> [[TMP11]], [[TMP3]]
; CHECK-NEXT: [[ICMP85:%.*]] = icmp ult ptr null, [[TMP32]]
-; CHECK-NEXT: [[AND86:%.*]] = and i1 false, [[ICMP85]]
+; CHECK-NEXT: [[AND85:%.*]] = and i1 false, [[ICMP85]]
; CHECK-NEXT: [[TMP37:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP16]])
-; CHECK-NEXT: [[OP_RDX16:%.*]] = or i1 [[TMP37]], [[AND]]
-; CHECK-NEXT: [[OP_RDX17:%.*]] = or i1 [[AND10]], [[AND37]]
-; CHECK-NEXT: [[OP_RDX18:%.*]] = or i1 [[AND85]], [[TMP33]]
-; CHECK-NEXT: [[OP_RDX3:%.*]] = or i1 [[AND22]], [[AND25]]
-; CHECK-NEXT: [[OP_RDX19:%.*]] = or i1 [[OP_RDX16]], [[OP_RDX17]]
+; CHECK-NEXT: [[OP_RDX19:%.*]] = or i1 [[TMP37]], [[AND]]
; CHECK-NEXT: [[OP_RDX5:%.*]] = or i1 [[OP_RDX18]], [[OP_RDX3]]
-; CHECK-NEXT: [[OP_RDX6:%.*]] = or i1 [[OP_RDX19]], [[OP_RDX5]]
+; CHECK-NEXT: [[OP_RDX2:%.*]] = or i1 [[AND15]], [[AND19]]
; CHECK-NEXT: [[OP_RDX20:%.*]] = or i1 [[OP_RDX6]], [[AND86]]
-; CHECK-NEXT: ret i1 [[OP_RDX20]]
+; CHECK-NEXT: [[OP_RDX4:%.*]] = or i1 [[OP_RDX19]], [[OP_RDX5]]
+; CHECK-NEXT: [[OP_RDX8:%.*]] = or i1 [[OP_RDX2]], [[OP_RDX20]]
+; CHECK-NEXT: [[OP_RDX9:%.*]] = or i1 [[OP_RDX4]], [[OP_RDX8]]
+; CHECK-NEXT: [[OP_RDX7:%.*]] = or i1 [[OP_RDX9]], [[AND85]]
+; CHECK-NEXT: ret i1 [[OP_RDX7]]
;
bb:
%getelementptr = getelementptr i8, ptr %arg1, i64 %arg2
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/runtime-alias-checks-scheduled-order.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/runtime-alias-checks-scheduled-order.ll
index 253318ba2aa8e..0b71f270d426a 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/runtime-alias-checks-scheduled-order.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/runtime-alias-checks-scheduled-order.ll
@@ -106,14 +106,9 @@ define void @versioned_block_check_reuses_no_body_scalars(ptr %arg, ptr nofree r
; CHECK-NEXT: [[RT_GUARD:%.*]] = freeze i1 [[RT_CONFLICT]]
; CHECK-NEXT: br i1 [[RT_GUARD]], label %[[ENTRY_RTSCALAR:.*]], label %[[ENTRY_RTVEC:.*]], !prof [[PROF0]]
; CHECK: [[ENTRY_RTVEC]]:
-; CHECK-NEXT: [[GEP4_1:%.*]] = getelementptr i8, ptr [[ARG]], i64 4
-; CHECK-NEXT: [[TMP2:%.*]] = insertelement <2 x ptr> poison, ptr [[ARG]], i64 0
-; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <2 x ptr> [[TMP2]], <2 x ptr> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP4:%.*]] = getelementptr i8, <2 x ptr> [[TMP3]], <2 x i64> <i64 8, i64 12>
-; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x ptr> [[TMP3]], <2 x ptr> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP6:%.*]] = insertelement <4 x ptr> [[TMP5]], ptr [[GEP4_1]], i64 1
-; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <2 x ptr> [[TMP4]], <2 x ptr> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <4 x ptr> [[TMP6]], <4 x ptr> [[TMP7]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x ptr> poison, ptr [[ARG]], i64 0
+; CHECK-NEXT: [[TMP16:%.*]] = shufflevector <4 x ptr> [[TMP2]], <4 x ptr> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP8:%.*]] = getelementptr i8, <4 x ptr> [[TMP16]], <4 x i64> <i64 0, i64 4, i64 8, i64 12>
; CHECK-NEXT: [[TMP9:%.*]] = ptrtoint <4 x ptr> [[TMP8]] to <4 x i64>
; CHECK-NEXT: [[TMP10:%.*]] = load <4 x i32>, ptr [[ARG2]], align 4
; CHECK-NEXT: [[TMP11:%.*]] = zext <4 x i32> [[TMP10]] to <4 x i64>
@@ -122,8 +117,6 @@ define void @versioned_block_check_reuses_no_body_scalars(ptr %arg, ptr nofree r
; CHECK-NEXT: [[TMP14:%.*]] = zext <4 x i1> [[TMP13]] to <4 x i32>
; CHECK-NEXT: store <4 x i32> [[TMP14]], ptr [[ARG]], align 4
; CHECK-NEXT: [[GEP_4:%.*]] = getelementptr i8, ptr [[ARG2]], i64 16
-; CHECK-NEXT: [[TMP15:%.*]] = insertelement <4 x ptr> poison, ptr [[ARG]], i64 0
-; CHECK-NEXT: [[TMP16:%.*]] = shufflevector <4 x ptr> [[TMP15]], <4 x ptr> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: [[TMP17:%.*]] = getelementptr i8, <4 x ptr> [[TMP16]], <4 x i64> <i64 16, i64 20, i64 24, i64 28>
; CHECK-NEXT: [[GEP4_4:%.*]] = getelementptr i8, ptr [[ARG]], i64 16
; CHECK-NEXT: [[TMP18:%.*]] = ptrtoint <4 x ptr> [[TMP17]] to <4 x i64>
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/gather-node-with-no-users.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/gather-node-with-no-users.ll
index 9c9e4210bb4db..4e77500b3a041 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/gather-node-with-no-users.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/gather-node-with-no-users.ll
@@ -8,8 +8,8 @@ define void @test(ptr %c) {
; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x ptr> poison, ptr [[C]], i64 0
; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <8 x ptr> [[TMP0]], <8 x ptr> poison, <8 x i32> zeroinitializer
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr i8, <8 x ptr> [[TMP1]], <8 x i64> <i64 222, i64 228, i64 276, i64 279, i64 282, i64 285, i64 288, i64 0>
-; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i8, <8 x ptr> [[TMP1]], <8 x i64> <i64 0, i64 345, i64 348, i64 351, i64 354, i64 357, i64 360, i64 363>
; CHECK-NEXT: [[TMP4:%.*]] = call <8 x i8> @llvm.masked.gather.v8i8.v8p0(<8 x ptr> align 1 [[TMP2]], <8 x i1> splat (i1 true), <8 x i8> poison)
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i8, <8 x ptr> [[TMP1]], <8 x i64> <i64 0, i64 345, i64 348, i64 351, i64 354, i64 357, i64 360, i64 363>
; CHECK-NEXT: [[TMP5:%.*]] = call <8 x i8> @llvm.masked.gather.v8i8.v8p0(<8 x ptr> align 1 [[TMP3]], <8 x i1> splat (i1 true), <8 x i8> poison)
; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <8 x i8> [[TMP5]], <8 x i8> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <8 x i8> [[TMP4]], <8 x i8> [[TMP5]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/gather-runtime-stride-with-base-lane.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/gather-runtime-stride-with-base-lane.ll
index 40a53e6ce40bf..ac6fd0169deca 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/gather-runtime-stride-with-base-lane.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/gather-runtime-stride-with-base-lane.ll
@@ -10,31 +10,14 @@ define i32 @sum_of_absolute_diff_8(ptr %b, i32 %S) {
; CHECK-LABEL: define i32 @sum_of_absolute_diff_8(
; CHECK-SAME: ptr [[B:%.*]], i32 [[S:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[IDXPROM_1:%.*]] = zext i32 [[S]] to i64
-; CHECK-NEXT: [[MUL_2:%.*]] = shl i32 [[S]], 1
-; CHECK-NEXT: [[IDXPROM_2:%.*]] = zext i32 [[MUL_2]] to i64
-; CHECK-NEXT: [[ARRAYIDX_2:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IDXPROM_2]]
-; CHECK-NEXT: [[TMP0:%.*]] = load i8, ptr [[ARRAYIDX_2]], align 1
-; CHECK-NEXT: [[MUL_3:%.*]] = mul i32 [[S]], 3
-; CHECK-NEXT: [[IDXPROM_3:%.*]] = zext i32 [[MUL_3]] to i64
-; CHECK-NEXT: [[ARRAYIDX_3:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IDXPROM_3]]
-; CHECK-NEXT: [[TMP1:%.*]] = load i8, ptr [[ARRAYIDX_3]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = mul i64 [[IDXPROM_1]], 1
-; CHECK-NEXT: [[TMP3:%.*]] = call <2 x i8> @llvm.experimental.vp.strided.load.v2i8.p0.i64(ptr align 1 [[B]], i64 [[TMP2]], <2 x i1> splat (i1 true), i32 2)
-; CHECK-NEXT: [[TMP4:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
-; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <4 x i32> [[TMP4]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP6:%.*]] = mul <4 x i32> [[TMP5]], <i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT: [[TMP17:%.*]] = zext <4 x i32> [[TMP6]] to <4 x i64>
-; CHECK-NEXT: [[TMP18:%.*]] = insertelement <4 x ptr> poison, ptr [[B]], i64 0
-; CHECK-NEXT: [[TMP19:%.*]] = shufflevector <4 x ptr> [[TMP18]], <4 x ptr> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP20:%.*]] = getelementptr inbounds nuw i8, <4 x ptr> [[TMP19]], <4 x i64> [[TMP17]]
-; CHECK-NEXT: [[TMP11:%.*]] = call <4 x i8> @llvm.masked.gather.v4i8.v4p0(<4 x ptr> align 1 [[TMP20]], <4 x i1> splat (i1 true), <4 x i8> poison)
-; CHECK-NEXT: [[TMP12:%.*]] = insertelement <8 x i8> poison, i8 [[TMP0]], i64 2
-; CHECK-NEXT: [[TMP13:%.*]] = insertelement <8 x i8> [[TMP12]], i8 [[TMP1]], i64 3
-; CHECK-NEXT: [[TMP14:%.*]] = shufflevector <4 x i8> [[TMP11]], <4 x i8> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP15:%.*]] = shufflevector <8 x i8> [[TMP13]], <8 x i8> [[TMP14]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
-; CHECK-NEXT: [[TMP16:%.*]] = shufflevector <2 x i8> [[TMP3]], <2 x i8> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <8 x i8> [[TMP15]], <8 x i8> [[TMP16]], <8 x i32> <i32 8, i32 9, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[S]], i64 1
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP0]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; CHECK-NEXT: [[TMP2:%.*]] = mul <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = zext <8 x i32> [[TMP2]] to <8 x i64>
+; CHECK-NEXT: [[TMP4:%.*]] = insertelement <8 x ptr> poison, ptr [[B]], i64 0
+; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <8 x ptr> [[TMP4]], <8 x ptr> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds nuw i8, <8 x ptr> [[TMP5]], <8 x i64> [[TMP3]]
+; CHECK-NEXT: [[TMP7:%.*]] = call <8 x i8> @llvm.masked.gather.v8i8.v8p0(<8 x ptr> align 1 [[TMP6]], <8 x i1> splat (i1 true), <8 x i8> poison)
; CHECK-NEXT: [[TMP8:%.*]] = sub <8 x i8> zeroinitializer, [[TMP7]]
; CHECK-NEXT: [[TMP9:%.*]] = sext <8 x i8> [[TMP8]] to <8 x i32>
; CHECK-NEXT: [[TMP10:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP9]])
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/gep-mixed-index-types.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/gep-mixed-index-types.ll
index fdfac8ce05fda..82a4c72e80dad 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/gep-mixed-index-types.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/gep-mixed-index-types.ll
@@ -9,53 +9,23 @@ define i32 @copyable_gep_node_mixed_index_types(ptr %b, i32 %S) {
; CHECK-LABEL: define i32 @copyable_gep_node_mixed_index_types(
; CHECK-SAME: ptr [[B:%.*]], i32 [[S:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = load i8, ptr [[B]], align 1
-; CHECK-NEXT: [[SUB:%.*]] = sub i8 0, [[TMP0]]
-; CHECK-NEXT: [[CONV:%.*]] = sext i8 [[SUB]] to i32
-; CHECK-NEXT: [[IDXPROM_1:%.*]] = zext i32 [[S]] to i64
-; CHECK-NEXT: [[ARRAYIDX_1:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IDXPROM_1]]
-; CHECK-NEXT: [[TMP1:%.*]] = load i8, ptr [[ARRAYIDX_1]], align 1
-; CHECK-NEXT: [[SUB_1:%.*]] = sub i8 0, [[TMP1]]
-; CHECK-NEXT: [[CONV_1:%.*]] = sext i8 [[SUB_1]] to i32
-; CHECK-NEXT: [[ADD_1:%.*]] = add nsw i32 [[CONV]], [[CONV_1]]
; CHECK-NEXT: [[MUL_2:%.*]] = shl i32 [[S]], 1
-; CHECK-NEXT: [[IDXPROM_2:%.*]] = zext i32 [[MUL_2]] to i64
-; CHECK-NEXT: [[ARRAYIDX_2:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IDXPROM_2]]
-; CHECK-NEXT: [[TMP2:%.*]] = load i8, ptr [[ARRAYIDX_2]], align 1
-; CHECK-NEXT: [[SUB_2:%.*]] = sub i8 0, [[TMP2]]
-; CHECK-NEXT: [[CONV_2:%.*]] = sext i8 [[SUB_2]] to i32
-; CHECK-NEXT: [[ADD_2:%.*]] = add nsw i32 [[ADD_1]], [[CONV_2]]
-; CHECK-NEXT: [[ARRAYIDX_3:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i32 7
-; CHECK-NEXT: [[TMP3:%.*]] = load i8, ptr [[ARRAYIDX_3]], align 1
-; CHECK-NEXT: [[SUB_3:%.*]] = sub i8 0, [[TMP3]]
-; CHECK-NEXT: [[CONV_3:%.*]] = sext i8 [[SUB_3]] to i32
-; CHECK-NEXT: [[ADD_3:%.*]] = add nsw i32 [[ADD_2]], [[CONV_3]]
; CHECK-NEXT: [[MUL_4:%.*]] = shl i32 [[S]], 2
-; CHECK-NEXT: [[IDXPROM_4:%.*]] = zext i32 [[MUL_4]] to i64
-; CHECK-NEXT: [[ARRAYIDX_4:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IDXPROM_4]]
-; CHECK-NEXT: [[TMP4:%.*]] = load i8, ptr [[ARRAYIDX_4]], align 1
-; CHECK-NEXT: [[SUB_4:%.*]] = sub i8 0, [[TMP4]]
-; CHECK-NEXT: [[CONV_4:%.*]] = sext i8 [[SUB_4]] to i32
-; CHECK-NEXT: [[ADD_4:%.*]] = add nsw i32 [[ADD_3]], [[CONV_4]]
-; CHECK-NEXT: [[ARRAYIDX_5:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i32 5
-; CHECK-NEXT: [[TMP5:%.*]] = load i8, ptr [[ARRAYIDX_5]], align 1
-; CHECK-NEXT: [[SUB_5:%.*]] = sub i8 0, [[TMP5]]
-; CHECK-NEXT: [[CONV_5:%.*]] = sext i8 [[SUB_5]] to i32
-; CHECK-NEXT: [[ADD_5:%.*]] = add nsw i32 [[ADD_4]], [[CONV_5]]
; CHECK-NEXT: [[MUL_6:%.*]] = mul i32 [[S]], 6
-; CHECK-NEXT: [[IDXPROM_6:%.*]] = zext i32 [[MUL_6]] to i64
-; CHECK-NEXT: [[ARRAYIDX_6:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IDXPROM_6]]
-; CHECK-NEXT: [[TMP6:%.*]] = load i8, ptr [[ARRAYIDX_6]], align 1
-; CHECK-NEXT: [[SUB_6:%.*]] = sub i8 0, [[TMP6]]
-; CHECK-NEXT: [[CONV_6:%.*]] = sext i8 [[SUB_6]] to i32
-; CHECK-NEXT: [[ADD_6:%.*]] = add nsw i32 [[ADD_5]], [[CONV_6]]
; CHECK-NEXT: [[MUL_7:%.*]] = mul i32 [[S]], 7
-; CHECK-NEXT: [[IDXPROM_7:%.*]] = zext i32 [[MUL_7]] to i64
-; CHECK-NEXT: [[ARRAYIDX_7:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IDXPROM_7]]
-; CHECK-NEXT: [[TMP7:%.*]] = load i8, ptr [[ARRAYIDX_7]], align 1
-; CHECK-NEXT: [[SUB_7:%.*]] = sub i8 0, [[TMP7]]
-; CHECK-NEXT: [[CONV_7:%.*]] = sext i8 [[SUB_7]] to i32
-; CHECK-NEXT: [[TMP12:%.*]] = add nsw i32 [[ADD_6]], [[CONV_7]]
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 7, i32 poison, i32 5, i32 poison, i32 poison>, i32 [[S]], i64 1
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <8 x i32> [[TMP0]], i32 [[MUL_2]], i64 2
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <8 x i32> [[TMP1]], i32 [[MUL_4]], i64 4
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <8 x i32> [[TMP2]], i32 [[MUL_6]], i64 6
+; CHECK-NEXT: [[TMP4:%.*]] = insertelement <8 x i32> [[TMP3]], i32 [[MUL_7]], i64 7
+; CHECK-NEXT: [[TMP5:%.*]] = zext <8 x i32> [[TMP4]] to <8 x i64>
+; CHECK-NEXT: [[TMP6:%.*]] = insertelement <8 x ptr> poison, ptr [[B]], i64 0
+; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <8 x ptr> [[TMP6]], <8 x ptr> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds nuw i8, <8 x ptr> [[TMP7]], <8 x i64> [[TMP5]]
+; CHECK-NEXT: [[TMP9:%.*]] = call <8 x i8> @llvm.masked.gather.v8i8.v8p0(<8 x ptr> align 1 [[TMP8]], <8 x i1> splat (i1 true), <8 x i8> poison)
+; CHECK-NEXT: [[TMP10:%.*]] = sub <8 x i8> zeroinitializer, [[TMP9]]
+; CHECK-NEXT: [[TMP11:%.*]] = sext <8 x i8> [[TMP10]] to <8 x i32>
+; CHECK-NEXT: [[TMP12:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP11]])
; CHECK-NEXT: ret i32 [[TMP12]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/revec-strided-load.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/revec-strided-load.ll
index 920f39c1cb762..c21b460027a69 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/revec-strided-load.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/revec-strided-load.ll
@@ -125,9 +125,9 @@ entry:
define void @too_wide(ptr %in0, ptr %out0) {
; CHECK-LABEL: @too_wide(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[IN1:%.*]] = getelementptr i16, ptr [[IN0:%.*]], i64 16
-; CHECK-NEXT: [[TMP4:%.*]] = insertelement <2 x ptr> poison, ptr [[IN0]], i64 0
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x ptr> [[TMP4]], ptr [[IN1]], i64 1
+; CHECK-NEXT: [[TMP4:%.*]] = insertelement <2 x ptr> poison, ptr [[IN0:%.*]], i64 0
+; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x ptr> [[TMP4]], <2 x ptr> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr i16, <2 x ptr> [[TMP5]], <2 x i64> <i64 0, i64 16>
; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x ptr> [[TMP1]], <2 x ptr> poison, <16 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i16, <16 x ptr> [[TMP2]], <16 x i64> <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7>
; CHECK-NEXT: [[TMP0:%.*]] = call <16 x i16> @llvm.masked.gather.v16i16.v16p0(<16 x ptr> align 2 [[TMP3]], <16 x i1> splat (i1 true), <16 x i16> poison)
@@ -170,9 +170,9 @@ entry:
define void @non_aligned_stride(ptr %in0, ptr %out0) {
; CHECK-LABEL: @non_aligned_stride(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[IN1:%.*]] = getelementptr i8, ptr [[IN0:%.*]], i64 3
-; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x ptr> poison, ptr [[IN0]], i64 0
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x ptr> [[TMP0]], ptr [[IN1]], i64 1
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x ptr> poison, ptr [[IN0:%.*]], i64 0
+; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x ptr> [[TMP0]], <2 x ptr> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr i8, <2 x ptr> [[TMP5]], <2 x i64> <i64 0, i64 3>
; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x ptr> [[TMP1]], <2 x ptr> poison, <4 x i32> <i32 0, i32 0, i32 1, i32 1>
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i8, <4 x ptr> [[TMP2]], <4 x i64> <i64 0, i64 1, i64 0, i64 1>
; CHECK-NEXT: [[TMP4:%.*]] = call <4 x i8> @llvm.masked.gather.v4i8.v4p0(<4 x ptr> align 2 [[TMP3]], <4 x i1> splat (i1 true), <4 x i8> poison)
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/copyable-operands-reordering.ll b/llvm/test/Transforms/SLPVectorizer/X86/copyable-operands-reordering.ll
index bfb9f92528233..e654b82baf744 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/copyable-operands-reordering.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/copyable-operands-reordering.ll
@@ -5,30 +5,25 @@ define i64 @test(ptr %buf) {
; CHECK-LABEL: define i64 @test(
; CHECK-SAME: ptr [[BUF:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[TMP0:%.*]] = load i8, ptr [[BUF]], align 1
-; CHECK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr [[BUF]], i64 1
-; CHECK-NEXT: [[ARRAYIDX8:%.*]] = getelementptr inbounds nuw i8, ptr [[BUF]], i64 3
-; CHECK-NEXT: [[TMP3:%.*]] = zext i8 [[TMP0]] to i32
-; CHECK-NEXT: [[TMP5:%.*]] = insertelement <4 x i32> <i32 poison, i32 1, i32 1, i32 1>, i32 [[TMP3]], i64 0
-; CHECK-NEXT: [[TMP6:%.*]] = mul <4 x i32> <i32 24, i32 0, i32 0, i32 0>, [[TMP5]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i8>, ptr [[BUF]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i16>
+; CHECK-NEXT: [[TMP2:%.*]] = mul <4 x i16> [[TMP1]], <i16 24, i16 16, i16 8, i16 1>
+; CHECK-NEXT: [[TMP3:%.*]] = call i16 @llvm.vector.reduce.or.v4i16(<4 x i16> [[TMP2]])
+; CHECK-NEXT: [[TMP4:%.*]] = zext i16 [[TMP3]] to i64
+; CHECK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw i8, ptr [[BUF]], i64 4
; CHECK-NEXT: [[TMP8:%.*]] = load <2 x i8>, ptr [[ARRAYIDX1]], align 1
-; CHECK-NEXT: [[TMP10:%.*]] = zext <2 x i8> [[TMP8]] to <2 x i32>
-; CHECK-NEXT: [[TMP17:%.*]] = shufflevector <2 x i32> [[TMP10]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP20:%.*]] = shufflevector <4 x i32> <i32 poison, i32 1, i32 1, i32 1>, <4 x i32> [[TMP17]], <4 x i32> <i32 4, i32 1, i32 2, i32 3>
-; CHECK-NEXT: [[TMP4:%.*]] = mul <4 x i32> <i32 16, i32 0, i32 0, i32 0>, [[TMP20]]
-; CHECK-NEXT: [[TMP7:%.*]] = or disjoint <4 x i32> [[TMP4]], [[TMP6]]
-; CHECK-NEXT: [[TMP22:%.*]] = shufflevector <4 x i32> <i32 poison, i32 1, i32 1, i32 1>, <4 x i32> [[TMP17]], <4 x i32> <i32 5, i32 1, i32 2, i32 3>
-; CHECK-NEXT: [[TMP9:%.*]] = mul <4 x i32> <i32 8, i32 0, i32 0, i32 0>, [[TMP22]]
-; CHECK-NEXT: [[TMP18:%.*]] = or disjoint <4 x i32> [[TMP7]], [[TMP9]]
-; CHECK-NEXT: [[TMP19:%.*]] = load <4 x i8>, ptr [[ARRAYIDX8]], align 1
-; CHECK-NEXT: [[TMP12:%.*]] = zext <4 x i8> [[TMP19]] to <4 x i32>
-; CHECK-NEXT: [[TMP13:%.*]] = or disjoint <4 x i32> [[TMP18]], [[TMP12]]
-; CHECK-NEXT: [[TMP14:%.*]] = mul <4 x i32> [[TMP13]], <i32 32, i32 24, i32 16, i32 8>
-; CHECK-NEXT: [[ARRAYIDX27:%.*]] = getelementptr inbounds nuw i8, ptr [[BUF]], i64 7
-; CHECK-NEXT: [[TMP16:%.*]] = load i8, ptr [[ARRAYIDX27]], align 1
-; CHECK-NEXT: [[CONV28:%.*]] = zext i8 [[TMP16]] to i64
-; CHECK-NEXT: [[TMP15:%.*]] = zext <4 x i32> [[TMP14]] to <4 x i64>
+; CHECK-NEXT: [[ARRAYIDX22:%.*]] = getelementptr inbounds nuw i8, ptr [[BUF]], i64 6
+; CHECK-NEXT: [[TMP6:%.*]] = zext <2 x i8> [[TMP8]] to <2 x i64>
+; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <2 x i64> [[TMP6]], <2 x i64> poison, <4 x i32> <i32 1, i32 0, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP14:%.*]] = insertelement <4 x i64> [[TMP7]], i64 [[TMP4]], i64 2
+; CHECK-NEXT: [[TMP9:%.*]] = load <2 x i8>, ptr [[ARRAYIDX22]], align 1
+; CHECK-NEXT: [[TMP10:%.*]] = zext <2 x i8> [[TMP9]] to <2 x i64>
+; CHECK-NEXT: [[TMP16:%.*]] = shufflevector <2 x i64> [[TMP10]], <2 x i64> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <2 x i64> [[TMP10]], <2 x i64> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <4 x i64> [[TMP14]], <4 x i64> [[TMP12]], <4 x i32> <i32 0, i32 1, i32 2, i32 4>
+; CHECK-NEXT: [[TMP15:%.*]] = mul <4 x i64> [[TMP13]], <i64 16, i64 24, i64 32, i64 8>
; CHECK-NEXT: [[TMP11:%.*]] = call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP15]])
+; CHECK-NEXT: [[CONV28:%.*]] = extractelement <2 x i64> [[TMP10]], i64 1
; CHECK-NEXT: [[OP_RDX:%.*]] = or disjoint i64 [[TMP11]], [[CONV28]]
; CHECK-NEXT: ret i64 [[OP_RDX]]
;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/copyable-zext-poison-lane.ll b/llvm/test/Transforms/SLPVectorizer/X86/copyable-zext-poison-lane.ll
index 8c942e7f1c7f9..37b4629eaf984 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/copyable-zext-poison-lane.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/copyable-zext-poison-lane.ll
@@ -6,11 +6,9 @@
define void @zext_lanes_with_poison_and_constant(ptr %p, i8 %a, i8 %b, i32 %x) {
; CHECK-LABEL: define void @zext_lanes_with_poison_and_constant(
; CHECK-SAME: ptr [[P:%.*]], i8 [[A:%.*]], i8 [[B:%.*]], i32 [[X:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x i8> poison, i8 [[A]], i64 0
-; CHECK-NEXT: [[TMP2:%.*]] = insertelement <2 x i8> [[TMP1]], i8 [[B]], i64 1
-; CHECK-NEXT: [[TMP7:%.*]] = zext <2 x i8> [[TMP2]] to <2 x i32>
-; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <2 x i32> [[TMP7]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <4 x i32> <i32 poison, i32 poison, i32 poison, i32 200>, <4 x i32> [[TMP8]], <4 x i32> <i32 4, i32 5, i32 poison, i32 3>
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x i8> <i8 poison, i8 poison, i8 poison, i8 -56>, i8 [[A]], i64 0
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x i8> [[TMP1]], i8 [[B]], i64 1
+; CHECK-NEXT: [[TMP3:%.*]] = zext <4 x i8> [[TMP2]] to <4 x i32>
; CHECK-NEXT: [[TMP4:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <4 x i32> [[TMP4]], <4 x i32> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: [[TMP6:%.*]] = mul <4 x i32> [[TMP3]], [[TMP5]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/split-node-throttled.ll b/llvm/test/Transforms/SLPVectorizer/X86/split-node-throttled.ll
index 5c882d80fc870..df72913751c46 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/split-node-throttled.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/split-node-throttled.ll
@@ -14,25 +14,25 @@ define fastcc void @test(i32 %arg) {
; CHECK-NEXT: [[TMP24:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> <i32 0, i32 poison>, <2 x i32> <i32 2, i32 0>
; CHECK-NEXT: [[TMP25:%.*]] = zext <2 x i32> [[TMP24]] to <2 x i64>
; CHECK-NEXT: [[TMP26:%.*]] = shufflevector <2 x i64> [[TMP25]], <2 x i64> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 1>
-; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> <i64 0, i64 0, i64 0, i64 0, i64 undef, i64 undef, i64 undef, i64 undef>, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
-; CHECK-NEXT: [[TMP27:%.*]] = shufflevector <4 x i64> [[TMP26]], <4 x i64> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP27:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: [[TMP28:%.*]] = shufflevector <8 x i64> [[TMP27]], <8 x i64> <i64 0, i64 0, i64 0, i64 0, i64 undef, i64 undef, i64 undef, i64 undef>, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
-; CHECK-NEXT: [[TMP10:%.*]] = mul <8 x i64> zeroinitializer, [[TMP9]]
-; CHECK-NEXT: [[TMP11:%.*]] = trunc <8 x i64> [[TMP10]] to <8 x i1>
+; CHECK-NEXT: [[TMP9:%.*]] = shufflevector <4 x i64> [[TMP26]], <4 x i64> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP10:%.*]] = shufflevector <8 x i64> [[TMP9]], <8 x i64> <i64 0, i64 0, i64 0, i64 0, i64 undef, i64 undef, i64 undef, i64 undef>, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; CHECK-NEXT: [[TMP14:%.*]] = mul <8 x i64> zeroinitializer, [[TMP28]]
+; CHECK-NEXT: [[TMP11:%.*]] = trunc <8 x i64> [[TMP14]] to <8 x i1>
; CHECK-NEXT: [[TMP12:%.*]] = or <8 x i1> zeroinitializer, [[TMP11]]
; CHECK-NEXT: [[TMP13:%.*]] = or <8 x i1> [[TMP12]], zeroinitializer
-; CHECK-NEXT: [[TMP14:%.*]] = shufflevector <8 x i1> [[TMP13]], <8 x i1> poison, <8 x i32> <i32 0, i32 4, i32 2, i32 6, i32 1, i32 5, i32 3, i32 7>
-; CHECK-NEXT: [[TMP15:%.*]] = mul <8 x i64> zeroinitializer, [[TMP28]]
-; CHECK-NEXT: [[TMP16:%.*]] = trunc <8 x i64> [[TMP15]] to <8 x i1>
-; CHECK-NEXT: [[TMP17:%.*]] = or <8 x i1> [[TMP14]], [[TMP16]]
+; CHECK-NEXT: [[TMP15:%.*]] = shufflevector <8 x i1> [[TMP13]], <8 x i1> poison, <8 x i32> <i32 0, i32 4, i32 2, i32 6, i32 1, i32 5, i32 3, i32 7>
+; CHECK-NEXT: [[TMP23:%.*]] = mul <8 x i64> zeroinitializer, [[TMP10]]
+; CHECK-NEXT: [[TMP29:%.*]] = trunc <8 x i64> [[TMP23]] to <8 x i1>
+; CHECK-NEXT: [[TMP17:%.*]] = or <8 x i1> [[TMP15]], [[TMP29]]
; CHECK-NEXT: [[TMP18:%.*]] = lshr <8 x i1> [[TMP17]], zeroinitializer
; CHECK-NEXT: [[TMP19:%.*]] = zext <8 x i1> [[TMP18]] to <8 x i32>
; CHECK-NEXT: [[TMP20:%.*]] = call <8 x i32> @llvm.smax.v8i32(<8 x i32> [[TMP19]], <8 x i32> zeroinitializer)
; CHECK-NEXT: [[TMP21:%.*]] = call <8 x i32> @llvm.smin.v8i32(<8 x i32> [[TMP20]], <8 x i32> zeroinitializer)
; CHECK-NEXT: [[TMP22:%.*]] = trunc <8 x i32> [[TMP21]] to <8 x i16>
-; CHECK-NEXT: [[TMP23:%.*]] = shufflevector <8 x i16> [[TMP22]], <8 x i16> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; CHECK-NEXT: store <8 x i16> [[TMP23]], ptr [[GETELEMENTPTR]], align 2
+; CHECK-NEXT: [[TMP30:%.*]] = shufflevector <8 x i16> [[TMP22]], <8 x i16> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-NEXT: store <8 x i16> [[TMP30]], ptr [[GETELEMENTPTR]], align 2
; CHECK-NEXT: ret void
;
bb:
diff --git a/llvm/test/Transforms/SLPVectorizer/revec.ll b/llvm/test/Transforms/SLPVectorizer/revec.ll
index 659b1ae79ebb8..af0c55e49aa3d 100644
--- a/llvm/test/Transforms/SLPVectorizer/revec.ll
+++ b/llvm/test/Transforms/SLPVectorizer/revec.ll
@@ -123,13 +123,13 @@ entry:
define void @test5(ptr %ptr0, ptr %ptr1) {
; CHECK-LABEL: @test5(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[GETELEMENTPTR0:%.*]] = getelementptr i8, ptr null, i64 0
-; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x ptr> <ptr null, ptr null, ptr undef, ptr undef>, ptr [[GETELEMENTPTR0]], i32 2
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x ptr> [[TMP0]], ptr null, i32 3
-; CHECK-NEXT: [[TMP2:%.*]] = icmp ult <4 x ptr> splat (ptr null), [[TMP1]]
; CHECK-NEXT: [[TMP3:%.*]] = insertelement <4 x ptr> <ptr poison, ptr null, ptr null, ptr null>, ptr [[PTR0:%.*]], i32 0
-; CHECK-NEXT: [[TMP4:%.*]] = insertelement <4 x ptr> [[TMP1]], ptr [[PTR1:%.*]], i32 3
-; CHECK-NEXT: [[TMP5:%.*]] = icmp ult <4 x ptr> [[TMP3]], [[TMP4]]
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x ptr> splat (ptr null), ptr [[PTR1:%.*]], i32 3
+; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <4 x ptr> [[TMP3]], <4 x ptr> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <8 x ptr> <ptr null, ptr null, ptr null, ptr null, ptr undef, ptr undef, ptr undef, ptr undef>, <8 x ptr> [[TMP2]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <4 x ptr> [[TMP1]], <4 x ptr> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <8 x ptr> <ptr null, ptr null, ptr null, ptr null, ptr undef, ptr undef, ptr undef, ptr undef>, <8 x ptr> [[TMP4]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; CHECK-NEXT: [[TMP6:%.*]] = icmp ult <8 x ptr> [[TMP7]], [[TMP5]]
; CHECK-NEXT: ret void
;
entry:
More information about the llvm-commits
mailing list