[llvm-branch-commits] [llvm] [LV][REVEC][AArch64] Proof of concept for re-vectorisation (PR #208213)
Gaëtan Bossu via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Wed Jul 8 06:07:33 PDT 2026
https://github.com/gbossu created https://github.com/llvm/llvm-project/pull/208213
⚠️ DRAFT ⚠️
This shows the changes required to enable basic re-vectorisation support in LoopVectorizer. Most of the diff comes from the added tests, the changes to LoopVectorizer files are rather minimal. This proof-of-concept has obvious limitations and only represents the first building block.
My hope is that this helps discussions and complements the RFC at https://discourse.llvm.org/t/rfc-re-vectorisation-to-wider-vectors-in-loopvectorizer/91071.
Support for re-vectorisation is hidden behind a -vectorize-vector-loops flag and LV will bail out if it encounters constructs that are not yet supported. For example:
- shufflevectors
- gather/scatter and interleaved accesses
- target intrinsics
- reductions
- if-conversion or tail folding
Note: This is part of a stacked PR to create a proof-of-concept for re-vectorisation in LoopVectorizer.
* https://github.com/llvm/llvm-project/pull/208212
* (this)
>From a7d3267e8fa0b0b08e1882483dcb8962d9dd2961 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 6 Jul 2026 10:32:07 +0000
Subject: [PATCH] [LV][REVEC][AArch64] Proof of concept for re-vectorisation
This shows the changes required to enable basic re-vectorisation support in LoopVectorizer. Most of the diff comes from the added tests, the changes to LoopVectorizer files are rather minimal. This proof-of-concept has obvious limitations and only represents the first building block.
My hope is that this helps discussions and complements the RFC at https://discourse.llvm.org/t/rfc-re-vectorisation-to-wider-vectors-in-loopvectorizer/91071.
Support for re-vectorisation is hidden behind a -vectorize-vector-loops flag and LV will bail out if it encounters constructs that are not yet supported. For example:
- shufflevectors
- gather/scatter and interleaved accesses
- target intrinsics
- reductions
- if-conversion or tail folding
---
llvm/include/llvm/IR/VectorTypeUtils.h | 25 +-
.../Vectorize/LoopVectorizationLegality.h | 4 +
llvm/lib/IR/IRBuilder.cpp | 20 ++
.../AArch64/AArch64TargetTransformInfo.h | 9 +
.../Vectorize/LoopVectorizationLegality.cpp | 74 ++++-
.../Vectorize/LoopVectorizationPlanner.cpp | 38 ++-
.../Transforms/Vectorize/LoopVectorize.cpp | 15 +-
llvm/lib/Transforms/Vectorize/VPlan.cpp | 24 +-
llvm/lib/Transforms/Vectorize/VPlanHelpers.h | 7 +
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 36 ++-
.../LoopVectorize/AArch64/revec-disabled.ll | 43 +++
.../LoopVectorize/AArch64/revec-invalid.ll | 73 +++++
.../LoopVectorize/AArch64/revec-livein.ll | 51 +++
.../AArch64/revec-memory-contiguous.ll | 90 ++++++
.../AArch64/revec-memory-gather-scatter.ll | 118 +++++++
.../AArch64/revec-memory-interleaved.ll | 89 ++++++
.../AArch64/revec-predication.ll | 102 ++++++
.../LoopVectorize/AArch64/revec-select.ll | 293 ++++++++++++++++++
.../LoopVectorize/AArch64/revec-unroll.ll | 100 ++++++
19 files changed, 1170 insertions(+), 41 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/revec-disabled.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/revec-invalid.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/revec-livein.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-contiguous.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-gather-scatter.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-interleaved.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/revec-predication.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/revec-select.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/revec-unroll.ll
diff --git a/llvm/include/llvm/IR/VectorTypeUtils.h b/llvm/include/llvm/IR/VectorTypeUtils.h
index e3d7fadad6089..05809f9e07233 100644
--- a/llvm/include/llvm/IR/VectorTypeUtils.h
+++ b/llvm/include/llvm/IR/VectorTypeUtils.h
@@ -14,12 +14,25 @@
namespace llvm {
-/// A helper function for converting Scalar types to vector types. If
+/// A helper function for converting scalar or vector types to vector types. If
/// the incoming type is void, we return void. If the EC represents a
-/// scalar, we return the scalar type.
+/// scalar, we return the input type. For vector inputs, the existing vector
+/// element count is multiplied by EC.
inline Type *toVectorTy(Type *Scalar, ElementCount EC) {
if (Scalar->isVoidTy() || Scalar->isMetadataTy() || EC.isScalar())
return Scalar;
+ if (auto *VTy = dyn_cast<VectorType>(Scalar)) {
+ assert(!(VTy->getElementCount().isScalable() && EC.isScalable()) &&
+ "Attempt to create <vscale x vscale x N x elt>!");
+
+ if (auto *FVTy = dyn_cast<FixedVectorType>(VTy))
+ return VectorType::get(VTy->getElementType(),
+ EC * FVTy->getNumElements());
+
+ return VectorType::get(VTy->getElementType(),
+ VTy->getElementCount() * EC.getKnownMinValue());
+ }
+
return VectorType::get(Scalar, EC);
}
@@ -27,6 +40,14 @@ inline Type *toVectorTy(Type *Scalar, unsigned VF) {
return toVectorTy(Scalar, ElementCount::getFixed(VF));
}
+/// Returns the ElementCount if Ty is a vector type, and 1 otherwise.
+inline ElementCount getElementCount(Type *Ty) {
+ // TODO: Support vectorized structs?
+ if (auto *VTy = dyn_cast<VectorType>(Ty))
+ return VTy->getElementCount();
+ return ElementCount::getFixed(1);
+}
+
/// A helper for converting structs of scalar types to structs of vector types.
/// Note:
/// - If \p EC is scalar, \p StructTy is returned unchanged
diff --git a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
index 3e8db73fd79d2..25ec10e71779e 100644
--- a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
+++ b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
@@ -48,6 +48,8 @@ class TargetLibraryInfo;
class TargetTransformInfo;
class Type;
+extern cl::opt<bool> VectorizeVectorLoops;
+
/// Utility class for getting and setting loop vectorizer hints in the form
/// of loop metadata.
/// This class keeps a number of loop annotations locally (as member variables)
@@ -501,6 +503,8 @@ class LoopVectorizationLegality {
return CountableExitingBlocks;
}
+ bool LoopContainsVectors = false;
+
private:
/// Return true if the pre-header, exiting and latch blocks of \p Lp and all
/// its nested loops are considered legal for vectorization. These legal
diff --git a/llvm/lib/IR/IRBuilder.cpp b/llvm/lib/IR/IRBuilder.cpp
index df738faad7842..644c9f38bbbae 100644
--- a/llvm/lib/IR/IRBuilder.cpp
+++ b/llvm/lib/IR/IRBuilder.cpp
@@ -30,6 +30,7 @@
#include "llvm/IR/Statepoint.h"
#include "llvm/IR/Type.h"
#include "llvm/IR/Value.h"
+#include "llvm/IR/VectorTypeUtils.h"
#include "llvm/Support/Casting.h"
#include <cassert>
#include <cstdint>
@@ -1286,6 +1287,25 @@ Value *IRBuilderBase::CreateVectorSplat(ElementCount EC, Value *V,
const Twine &Name) {
assert(EC.isNonZero() && "Cannot splat to an empty vector!");
+ if (V->getType()->isVectorTy()) {
+ auto *VectorTy = cast<VectorType>(V->getType());
+ assert(!isa<ScalableVectorType>(VectorTy));
+
+ // If the value was already a constant splat, just recreate it with a
+ // bigger element count.
+ if (auto *VectorConstant = dyn_cast<Constant>(V);
+ VectorConstant && VectorConstant->getSplatValue()) {
+ ElementCount ActualEC =
+ EC.multiplyCoefficientBy(VectorTy->getElementCount().getFixedValue());
+ return CreateVectorSplat(ActualEC, VectorConstant->getSplatValue(), Name);
+ }
+
+ // Otherwise, use vector.broadcast.
+ auto *WideVectorTy = toVectorTy(VectorTy, EC);
+ return CreateIntrinsic(Intrinsic::vector_broadcast,
+ {WideVectorTy, VectorTy}, {V}, {}, Name);
+ }
+
// First insert it into a poison vector so we can shuffle it.
Value *Poison = PoisonValue::get(VectorType::get(V->getType(), EC));
V = CreateInsertElement(Poison, V, getInt64(0), Name + ".splatinsert");
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index 24934cb01cdd1..a5d73578d0511 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -307,6 +307,15 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
Ty->isIntegerTy(32) || Ty->isIntegerTy(64))
return true;
+ // REVEC: Allow 64-bit and 128-bit vectors, as well as i1 vectors. We still
+ // avoid creating nxv1 types though, as codegen for those can be suboptimal.
+ if (auto *FVTy = dyn_cast<FixedVectorType>(Ty))
+ return (FVTy->getElementType()->isIntegerTy(1) ||
+ DL.getTypeSizeInBits(FVTy) == AArch64::SVEBitsPerBlock ||
+ DL.getTypeSizeInBits(FVTy) == AArch64::SVEBitsPerBlock / 2) &&
+ FVTy->getNumElements() > 1 &&
+ isElementTypeLegalForScalableVector(FVTy->getElementType());
+
return false;
}
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index 8d875b2b6e492..fcae1c3bb4a0f 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -79,6 +79,13 @@ static cl::opt<bool> EnableHistogramVectorization(
"enable-histogram-loop-vectorization", cl::init(false), cl::Hidden,
cl::desc("Enables autovectorization of some loops containing histograms"));
+namespace llvm {
+cl::opt<bool>
+ VectorizeVectorLoops("vectorize-vector-loops", cl::init(false), cl::Hidden,
+ cl::desc("Allow vectorization of loops with vector "
+ "instructions."));
+} // namespace llvm
+
/// Maximum vectorization interleave count.
static const unsigned MaxInterleaveFactor = 16;
@@ -953,8 +960,20 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
if (CI && !VFDatabase::getMappings(*CI).empty())
VecCallVariantsFound = true;
- auto CanWidenInstructionTy = [](Instruction const &Inst) {
+ // REVEC: Remember that a vector instruction was found for later checks.
+ if (I.getType()->isVectorTy() ||
+ any_of(I.operand_values(),
+ [](const Value *V) { return V->getType()->isVectorTy(); }))
+ LoopContainsVectors = true;
+
+ auto CanWidenInstructionTy = [TTI = TTI](Instruction const &Inst) {
Type *InstTy = Inst.getType();
+
+ // TODO-REVEC: To support fixed VFs, we'll need to query a diffent TTI hook.
+ if (isa<FixedVectorType>(InstTy))
+ return VectorizeVectorLoops &&
+ TTI->isElementTypeLegalForScalableVector(InstTy->getScalarType());
+
if (!isa<StructType>(InstTy))
return canVectorizeTy(InstTy);
@@ -965,13 +984,20 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
all_of(Inst.users(), IsaPred<ExtractValueInst>);
};
+ auto CanWidenCast = [&](const Instruction &CastI) {
+ assert(isa<CastInst>(CastI));
+ assert(CanWidenInstructionTy(CastI) &&
+ "CanWidenInstructionTy was not checked beforehand.");
+ Type *FromTy = CastI.getOperand(0)->getType();
+ return VectorType::isValidElementType(FromTy) ||
+ (isa<FixedVectorType>(FromTy) && VectorizeVectorLoops &&
+ TTI->isElementTypeLegalForScalableVector(FromTy->getScalarType()));
+ };
+
// Check that the instruction return type is vectorizable.
- // We can't vectorize casts from vector type to scalar type.
- // Also, we can't vectorize extractelement instructions.
- if (!CanWidenInstructionTy(I) ||
- (isa<CastInst>(I) &&
- !VectorType::isValidElementType(I.getOperand(0)->getType())) ||
- isa<ExtractElementInst>(I)) {
+ // Also, we cannot re-vectorize element or shuffle operations yet.
+ if (!CanWidenInstructionTy(I) || (isa<CastInst>(I) && !CanWidenCast(I)) ||
+ isa<ExtractElementInst, InsertElementInst, ShuffleVectorInst>(I)) {
reportVectorizationFailure("Found unvectorizable type",
"instruction return type cannot be vectorized",
"CantVectorizeInstructionReturnType", ORE,
@@ -982,7 +1008,11 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
// Check that the stored type is vectorizable.
if (auto *ST = dyn_cast<StoreInst>(&I)) {
Type *T = ST->getValueOperand()->getType();
- if (!VectorType::isValidElementType(T)) {
+ bool CanWidenStoreType =
+ VectorType::isValidElementType(T) ||
+ (isa<FixedVectorType>(T) && VectorizeVectorLoops &&
+ TTI->isElementTypeLegalForScalableVector(T->getScalarType()));
+ if (!CanWidenStoreType) {
reportVectorizationFailure("Store instruction cannot be vectorized",
"CantVectorizeStore", ORE, TheLoop, ST);
return false;
@@ -992,7 +1022,8 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
// supported on the target.
if (ST->getMetadata(LLVMContext::MD_nontemporal)) {
// Arbitrarily try a vector of 2 elements.
- auto *VecTy = FixedVectorType::get(T, /*NumElts=*/2);
+ Type *VecTy =
+ T->isVectorTy() ? T : FixedVectorType::get(T, /*NumElts=*/2);
assert(VecTy && "did not find vectorized version of stored type");
if (!TTI->isLegalNTStore(VecTy, ST->getAlign())) {
reportVectorizationFailure(
@@ -1006,7 +1037,9 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
if (LD->getMetadata(LLVMContext::MD_nontemporal)) {
// For nontemporal loads, check that a nontemporal vector version is
// supported on the target (arbitrarily try a vector of 2 elements).
- auto *VecTy = FixedVectorType::get(I.getType(), /*NumElts=*/2);
+ Type *VecTy = I.getType()->isVectorTy()
+ ? I.getType()
+ : FixedVectorType::get(I.getType(), /*NumElts=*/2);
assert(VecTy && "did not find vectorized version of load type");
if (!TTI->isLegalNTLoad(VecTy, LD->getAlign())) {
reportVectorizationFailure(
@@ -1912,6 +1945,19 @@ bool LoopVectorizationLegality::canVectorize(bool UseVPlanNativePath) {
return false;
}
+ if (LoopContainsVectors && any_of(TheLoop->blocks(), [this](BasicBlock *BB) {
+ return blockNeedsPredication(BB);
+ })) {
+ reportVectorizationFailure("Cannot if-convert vector loop",
+ "if-conversion is not supported for vector "
+ "instructions in loop",
+ "UnsupportedVectorInstruction", ORE, TheLoop);
+ if (DoExtraAnalysis)
+ Result = false;
+ else
+ return false;
+ }
+
if (isa<SCEVCouldNotCompute>(PSE.getBackedgeTakenCount())) {
if (TheLoop->getExitingBlock()) {
reportVectorizationFailure("Cannot vectorize uncountable loop",
@@ -1981,6 +2027,14 @@ bool LoopVectorizationLegality::canFoldTailByMasking() const {
LLVM_DEBUG(dbgs() << "LV: checking if tail can be folded by masking.\n");
+ // TODO-REVEC: Disable tail-folding for now. New intrinsics are needed for
+ // per-segment predication because the element count of the predicate and the
+ // data type do not match.
+ if (LoopContainsVectors) {
+ LLVM_DEBUG(dbgs() << "LV: Tail-folding disabled for REVEC.\n");
+ return false;
+ }
+
// The list of pointers that we can safely read and write to remains empty.
SmallPtrSet<Value *, 8> SafePointers;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index dbb5ad28fb4ed..bcf8bcac5ddb6 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -154,8 +154,11 @@ bool VFSelectionContext::isLegalGatherOrScatter(Value *V,
return false;
auto *Ty = getLoadStoreType(V);
Align Align = getLoadStoreAlignment(V);
+ // TODO-REVEC: Support non-contiguous accesses
+ if (Ty->isVectorTy())
+ return false;
if (VF.isVector())
- Ty = VectorType::get(Ty, VF);
+ Ty = toVectorTy(Ty, VF);
return ForceTargetSupportsGatherScatterOps ||
(LI && TTI.isLegalMaskedGather(Ty, Align)) ||
(SI && TTI.isLegalMaskedScatter(Ty, Align));
@@ -279,7 +282,9 @@ ElementCount VFSelectionContext::getMaximizedVFForTarget(
else
MaxPermissibleVFWithoutMaxBW.FixedVF = MaxVF;
- if (useMaxBandwidth(ComputeScalableMaxVF)) {
+ // REVEC: Avoid creating wider than expected scalable types by choosing
+ // VF > vscale x 1.
+ if (useMaxBandwidth(ComputeScalableMaxVF) && !Legal->LoopContainsVectors) {
auto MaxVectorElementCountMaxBW = ElementCount::get(
llvm::bit_floor(WidestRegister.getKnownMinValue() / SmallestType),
ComputeScalableMaxVF);
@@ -350,6 +355,12 @@ bool VFSelectionContext::isScalableVectorizationAllowed() {
// Disable scalable vectorization if the loop contains any instructions
// with element types not supported for scalable vectors.
if (any_of(ElementTypesInLoop, [&](Type *Ty) {
+ if (auto *FVTy = dyn_cast<FixedVectorType>(Ty)) {
+ if (!VectorizeVectorLoops)
+ return true;
+ Ty = FVTy->getElementType();
+ } else if (Ty->isVectorTy())
+ return true;
return !Ty->isVoidTy() && !TTI.isElementTypeLegalForScalableVector(Ty);
})) {
reportVectorizationInfo("Scalable vectorization is not supported "
@@ -412,6 +423,11 @@ FixedScalableVFPair VFSelectionContext::computeFeasibleMaxVF(
auto MaxSafeFixedVF = ElementCount::getFixed(MaxSafeElementsPowerOf2);
auto MaxSafeScalableVF = getMaxLegalScalableVF(MaxSafeElementsPowerOf2);
+ // A vector loop can only be widened to a scalable vector loop for now.
+ // TODO-REVEC: Support fixed-length REVEC.
+ if (Legal->LoopContainsVectors)
+ MaxSafeFixedVF = ElementCount::getFixed(1);
+
if (!Legal->isSafeForAnyVectorWidth())
MaxSafeElements = MaxSafeElementsPowerOf2;
@@ -427,9 +443,13 @@ FixedScalableVFPair VFSelectionContext::computeFeasibleMaxVF(
if (ElementCount::isKnownLE(UserVF, MaxSafeUserVF)) {
// If `VF=vscale x N` is safe, then so is `VF=N`
- if (UserVF.isScalable())
- return FixedScalableVFPair(
- ElementCount::getFixed(UserVF.getKnownMinValue()), UserVF);
+ // (unless it's wider than MaxSafeFixedVF).
+ if (UserVF.isScalable()) {
+ auto UserVFAsFixed = ElementCount::getFixed(UserVF.getKnownMinValue());
+ return ElementCount::isKnownLE(UserVFAsFixed, MaxSafeFixedVF)
+ ? FixedScalableVFPair(UserVFAsFixed, UserVF)
+ : FixedScalableVFPair(MaxSafeFixedVF, UserVF);
+ }
return UserVF;
}
@@ -525,10 +545,10 @@ VFSelectionContext::getSmallestAndWidestTypes() const {
}
} else {
for (Type *T : ElementTypesInLoop) {
- MinWidth = std::min<unsigned>(
- MinWidth, DL.getTypeSizeInBits(T->getScalarType()).getFixedValue());
- MaxWidth = std::max<unsigned>(
- MaxWidth, DL.getTypeSizeInBits(T->getScalarType()).getFixedValue());
+ MinWidth =
+ std::min<unsigned>(MinWidth, DL.getTypeSizeInBits(T).getFixedValue());
+ MaxWidth =
+ std::max<unsigned>(MaxWidth, DL.getTypeSizeInBits(T).getFixedValue());
}
}
return {MinWidth, MaxWidth};
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 6bb68d1f7bb19..c0efb67a2f6fb 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -4377,6 +4377,9 @@ LoopVectorizationCostModel::getInterleaveGroupCost(Instruction *I,
Instruction *InsertPos = Group->getInsertPos();
Type *ValTy = getLoadStoreType(InsertPos);
+ if (ValTy->isVectorTy())
+ return InstructionCost::getInvalid();
+
auto *VectorTy = cast<VectorType>(toVectorTy(ValTy, VF));
unsigned AS = getLoadStoreAddressSpace(InsertPos);
@@ -5133,6 +5136,7 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
// fold away. We can generalize this for all operations using the notion
// of neutral elements. (TODO)
if (I->getOpcode() == Instruction::Mul &&
+ PSE.getSE()->isSCEVable(I->getOperand(0)->getType()) &&
((TheLoop->isLoopInvariant(I->getOperand(0)) &&
PSE.getSCEV(I->getOperand(0))->isOne()) ||
(TheLoop->isLoopInvariant(I->getOperand(1)) &&
@@ -5171,8 +5175,12 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
}
case Instruction::Select: {
SelectInst *SI = cast<SelectInst>(I);
- const SCEV *CondSCEV = SE->getSCEV(SI->getCondition());
- bool ScalarCond = (SE->isLoopInvariant(CondSCEV, TheLoop));
+ Type *CondTy = SI->getCondition()->getType();
+ bool ScalarCond = false;
+ if (SE->isSCEVable(CondTy)) {
+ const SCEV *CondSCEV = SE->getSCEV(SI->getCondition());
+ ScalarCond = SE->isLoopInvariant(CondSCEV, TheLoop);
+ }
const Value *Op0, *Op1;
using namespace llvm::PatternMatch;
@@ -5191,9 +5199,8 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
I);
}
- Type *CondTy = SI->getCondition()->getType();
if (!ScalarCond)
- CondTy = VectorType::get(CondTy, VF);
+ CondTy = toVectorTy(CondTy, VF);
CmpInst::Predicate Pred = CmpInst::BAD_ICMP_PREDICATE;
if (auto *Cmp = dyn_cast<CmpInst>(SI->getCondition()))
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.cpp b/llvm/lib/Transforms/Vectorize/VPlan.cpp
index 292e3e85da9d5..01d6b9889af16 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlan.cpp
@@ -299,11 +299,31 @@ Value *VPTransformState::get(const VPValue *Def, const VPLane &Lane) {
assert(hasVectorValue(Def));
auto *VecPart = Data.VPV2Vector[Def];
- if (!VecPart->getType()->isVectorTy()) {
+ // If VecPart's type is the same as the initial pre-vectorisation type,
+ // Def hasn't been vectorised. We are done.
+ Type *InitialTy = Def->getScalarType();
+ if (VecPart->getType() == InitialTy) {
assert(Lane.isFirstLane() && "cannot get lane > 0 for scalar");
return VecPart;
}
+
+ // Last case: Extract a lane of type InitialTy from a vectorised value.
// TODO: Cache created scalar values.
+ if (auto *InitVTy = dyn_cast<VectorType>(InitialTy)) {
+ assert(!InitialTy->isScalableTy() && "REVEC: Unexpected scalable vector");
+ unsigned EC = InitVTy->getElementCount().getFixedValue();
+ if (Lane.getKind() == VPLane::Kind::First)
+ return Builder.CreateExtractVector(InitialTy, VecPart,
+ uint64_t(Lane.getKnownLane() * EC));
+
+ // Shift the demanded InitVTy-typed lane into lane 0.
+ unsigned NumLanesFromEnd =
+ VF.getKnownMinValue() - Lane.getOffsetInLastSubvec();
+ auto *ShiftLastSubvec = Builder.CreateVectorSpliceRight(
+ VecPart, PoisonValue::get(VecPart->getType()), NumLanesFromEnd * EC);
+ return Builder.CreateExtractVector(InitialTy, ShiftLastSubvec, uint64_t(0));
+ }
+
Value *LaneV = Lane.getAsRuntimeExpr(Builder, VF);
auto *Extract = Builder.CreateExtractElement(VecPart, LaneV);
// set(Def, Extract, Instance);
@@ -328,7 +348,7 @@ Value *VPTransformState::get(const VPValue *Def, bool NeedsScalar) {
auto GetBroadcastInstrs = [this](Value *V) {
if (VF.isScalar())
return V;
- // Broadcast the scalar into all locations in the vector.
+ // Broadcast the value into all locations in the vector.
Value *Shuf = Builder.CreateVectorSplat(VF, V, "broadcast");
return Shuf;
};
diff --git a/llvm/lib/Transforms/Vectorize/VPlanHelpers.h b/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
index 5ea0208e416ed..e9aaac9dedc45 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanHelpers.h
@@ -161,6 +161,13 @@ class VPLane {
return Lane;
}
+ /// Returns the offset of this lane from the start of the last
+ /// <N x ElTy> subvector.
+ unsigned getOffsetInLastSubvec() const {
+ assert(LaneKind == Kind::ScalableLast);
+ return Lane;
+ }
+
/// Returns an expression describing the lane index that can be used at
/// runtime.
Value *getAsRuntimeExpr(IRBuilderBase &Builder, const ElementCount &VF) const;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 258ca7296006b..09b597192e6ed 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -489,13 +489,9 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
case Instruction::Store:
return Type::getVoidTy(Ctx);
case Instruction::ICmp:
- assert(Op0Ty->isIntOrPtrTy() && "expected integer or pointer operand");
- AssertOperandType(1, Op0Ty);
- return IntegerType::get(Ctx, 1);
case Instruction::FCmp:
- assert(Op0Ty->isFloatingPointTy() && "expected floating-point operand");
AssertOperandType(1, Op0Ty);
- return IntegerType::get(Ctx, 1);
+ return CmpInst::makeCmpResultType(Op0Ty);
case VPInstruction::ActiveLaneMask:
assert(Op0Ty->isIntegerTy() && "expected integer operand");
AssertOperandType(1, Op0Ty);
@@ -517,8 +513,6 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
assert(Op0Ty->isIntegerTy() && "expected integer operand");
return IntegerType::get(Ctx, 32);
case Instruction::Select: {
- assert((!Op0Ty || Op0Ty->isIntegerTy(1)) &&
- "select condition must be bool");
Type *Op1Ty = Operands[1]->getScalarType();
AssertOperandType(2, Op1Ty);
return Op1Ty;
@@ -1259,8 +1253,16 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
}
case Instruction::Select: {
SelectInst *SI = cast_or_null<SelectInst>(getUnderlyingValue());
- bool IsScalarCond = getOperand(0)->isDefinedOutsideLoopRegions();
+ Type *InitialCondTy = getOperand(0)->getScalarType();
Type *ScalarTy = this->getScalarType();
+ bool IsScalarCond = InitialCondTy->isIntegerTy(1) &&
+ getOperand(0)->isDefinedOutsideLoopRegions();
+
+ // TODO-REVEC: Support all kinds of InitialCondTy
+ if (ScalarTy->isVectorTy() &&
+ (!IsScalarCond &&
+ getElementCount(ScalarTy) != getElementCount(InitialCondTy)))
+ return InstructionCost::getInvalid();
VPValue *Op0, *Op1;
bool IsLogicalAnd =
@@ -1291,9 +1293,7 @@ InstructionCost VPRecipeWithIRFlags::getCostForRecipeWithOpcode(
Ctx.CostKind, {Op1VK, Op1VP}, {Op2VK, Op2VP}, Operands, SI);
}
- Type *CondTy = getOperand(0)->getScalarType();
- if (!IsScalarCond && VF.isVector())
- CondTy = VectorType::get(CondTy, VF);
+ Type *CondTy = IsScalarCond ? InitialCondTy : toVectorTy(InitialCondTy, VF);
llvm::CmpPredicate Pred;
if (!match(getOperand(0), m_Cmp(Pred, m_VPValue(), m_VPValue())))
@@ -1332,6 +1332,8 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
match(getOperand(0), m_Cmp(Pred, m_VPValue(), m_VPValue()));
auto *CondTy = getOperand(0)->getScalarType();
auto *VecTy = getOperand(1)->getScalarType();
+ assert(!CondTy->isVectorTy() &&
+ "REVEC: Unexpected VPlan-created select at this stage");
if (!vputils::onlyFirstLaneUsed(this)) {
CondTy = toVectorTy(CondTy, VF);
VecTy = toVectorTy(VecTy, VF);
@@ -2795,7 +2797,13 @@ void VPWidenRecipe::execute(VPTransformState &State) {
}
case Instruction::Select: {
VPValue *CondOp = getOperand(0);
- Value *Cond = State.get(CondOp, vputils::isSingleScalar(CondOp));
+ Type *InitialCondTy = CondOp->getScalarType();
+ // REVEC: The initial CondOp might be a vector, so be sure we either
+ // generate a single i1 condition, or an i1 vector with the same EC as the
+ // data inputs.
+ bool BuildScalarCond =
+ vputils::isSingleScalar(CondOp) && InitialCondTy->isIntegerTy(1);
+ Value *Cond = State.get(CondOp, BuildScalarCond);
Value *Op0 = State.get(getOperand(1));
Value *Op1 = State.get(getOperand(2));
Value *Sel = State.Builder.CreateSelect(Cond, Op0, Op1);
@@ -2817,7 +2825,7 @@ void VPWidenRecipe::execute(VPTransformState &State) {
#if !defined(NDEBUG)
// Verify that VPlan type inference results agree with the type of the
// generated values.
- assert(VectorType::get(this->getScalarType(), State.VF) ==
+ assert(toVectorTy(this->getScalarType(), State.VF) ==
State.get(this)->getType() &&
"inferred type and type from generated instructions do not match");
#endif
@@ -4135,7 +4143,7 @@ InstructionCost VPWidenMemoryRecipe::computeCost(ElementCount VF,
void VPWidenLoadRecipe::execute(VPTransformState &State) {
Type *ScalarDataTy = getScalarType();
- auto *DataTy = VectorType::get(ScalarDataTy, State.VF);
+ auto *DataTy = toVectorTy(ScalarDataTy, State.VF);
bool CreateGather = !isConsecutive();
auto &Builder = State.Builder;
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/revec-disabled.ll b/llvm/test/Transforms/LoopVectorize/AArch64/revec-disabled.ll
new file mode 100644
index 0000000000000..ee50a9fca7b61
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/revec-disabled.ll
@@ -0,0 +1,43 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter-out-after "scalar.ph:" --replace-value-regex "!llvm.loop ![0-9]+" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -force-vector-interleave=1 -vectorize-vector-loops=false \
+; RUN: < %s | FileCheck %s --check-prefix=NO-REVEC
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -force-vector-interleave=2 -vectorize-vector-loops=false \
+; RUN: < %s | FileCheck %s --check-prefix=NO-REVEC
+
+; Make sure vector loops aren't modified when -vectorize-vector-loops is false.
+
+define void @copy_v4(ptr noalias %dst, ptr noalias %src, i64 %n) {
+; NO-REVEC-LABEL: define void @copy_v4(
+; NO-REVEC-SAME: ptr noalias [[DST:%.*]], ptr noalias [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; NO-REVEC-NEXT: [[ENTRY:.*]]:
+; NO-REVEC-NEXT: br label %[[LOOP:.*]]
+; NO-REVEC: [[LOOP]]:
+; NO-REVEC-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; NO-REVEC-NEXT: [[SRC_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[SRC]], i64 [[IV]]
+; NO-REVEC-NEXT: [[DST_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[IV]]
+; NO-REVEC-NEXT: [[V:%.*]] = load <4 x i32>, ptr [[SRC_GEP]], align 16
+; NO-REVEC-NEXT: store <4 x i32> [[V]], ptr [[DST_GEP]], align 16
+; NO-REVEC-NEXT: [[IV_NEXT]] = add nuw i64 [[IV]], 1
+; NO-REVEC-NEXT: [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; NO-REVEC-NEXT: br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; NO-REVEC: [[EXIT]]:
+; NO-REVEC-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %src.gep = getelementptr inbounds <4 x i32>, ptr %src, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %iv
+ %v = load <4 x i32>, ptr %src.gep, align 16
+ store <4 x i32> %v, ptr %dst.gep, align 16
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/revec-invalid.ll b/llvm/test/Transforms/LoopVectorize/AArch64/revec-invalid.ll
new file mode 100644
index 0000000000000..8695676c6e49d
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/revec-invalid.ll
@@ -0,0 +1,73 @@
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -scalable-vectorization=on -force-vector-interleave=1 \
+; RUN: -vectorize-vector-loops < %s | FileCheck %s --check-prefix=IR
+; RUN: opt -disable-output -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -scalable-vectorization=on -force-vector-interleave=1 \
+; RUN: -vectorize-vector-loops -pass-remarks-analysis=loop-vectorize \
+; RUN: < %s 2>&1 | FileCheck %s --check-prefix=REMARKS
+
+; IR-LABEL: @insertelement_cost(
+; IR-NOT: vector.body:
+; REMARKS: loop not vectorized: instruction return type cannot be vectorized
+define void @insertelement_cost(ptr noalias nocapture noundef writeonly %a, ptr nocapture noundef readonly %b) {
+entry:
+ br label %for.body
+
+for.cond.cleanup:
+ ret void
+
+for.body:
+ %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %for.body ]
+ %arrayidx = getelementptr inbounds i16, ptr %b, i64 %indvars.iv
+ %0 = load i16, ptr %arrayidx, align 16
+ %result = insertelement <8 x i16> zeroinitializer, i16 %0, i32 0
+ %arrayidx2 = getelementptr inbounds <8 x i16>, ptr %a, i64 %indvars.iv
+ store <8 x i16> %result, ptr %arrayidx2, align 16
+ %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+ %exitcond.not = icmp eq i64 %indvars.iv.next, 1024
+ br i1 %exitcond.not, label %for.cond.cleanup, label %for.body
+}
+
+; IR-LABEL: @extractelement_cost(
+; IR-NOT: vector.body:
+; REMARKS: loop not vectorized: instruction return type cannot be vectorized
+define void @extractelement_cost(ptr noalias nocapture noundef writeonly %a, ptr nocapture noundef readonly %b) {
+entry:
+ br label %for.body
+
+for.cond.cleanup:
+ ret void
+
+for.body:
+ %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %for.body ]
+ %arrayidx = getelementptr inbounds <4 x i32>, ptr %b, i64 %indvars.iv
+ %0 = load <4 x i32>, ptr %arrayidx, align 16
+ %result = extractelement <4 x i32> %0, i32 1
+ %arrayidx2 = getelementptr inbounds i32, ptr %a, i64 %indvars.iv
+ store i32 %result, ptr %arrayidx2, align 16
+ %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+ %exitcond.not = icmp eq i64 %indvars.iv.next, 1024
+ br i1 %exitcond.not, label %for.cond.cleanup, label %for.body
+}
+
+; IR-LABEL: @shufflevector_cost(
+; IR-NOT: vector.body:
+; REMARKS: loop not vectorized: instruction return type cannot be vectorized
+define void @shufflevector_cost(ptr noalias nocapture noundef writeonly %a, ptr nocapture noundef readonly %b) {
+entry:
+ br label %for.body
+
+for.cond.cleanup:
+ ret void
+
+for.body:
+ %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %for.body ]
+ %arrayidx = getelementptr inbounds <4 x i32>, ptr %b, i64 %indvars.iv
+ %0 = load <4 x i32>, ptr %arrayidx, align 16
+ %result = shufflevector <4 x i32> %0, <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+ %arrayidx2 = getelementptr inbounds <4 x i32>, ptr %a, i64 %indvars.iv
+ store <4 x i32> %result, ptr %arrayidx2, align 16
+ %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+ %exitcond.not = icmp eq i64 %indvars.iv.next, 1024
+ br i1 %exitcond.not, label %for.cond.cleanup, label %for.body
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/revec-livein.ll b/llvm/test/Transforms/LoopVectorize/AArch64/revec-livein.ll
new file mode 100644
index 0000000000000..20d284bb10579
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/revec-livein.ll
@@ -0,0 +1,51 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter-out-after "scalar.ph:" --replace-value-regex "!llvm.loop ![0-9]+" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -scalable-vectorization=on -force-vector-interleave=1 \
+; RUN: -vectorize-vector-loops < %s | FileCheck %s
+
+define void @select_livein_v4(ptr noalias %dst, ptr noalias %src, <4 x i32> %threshold, i64 %n) {
+; CHECK-LABEL: define void @select_livein_v4(
+; CHECK-SAME: ptr noalias [[DST:%.*]], ptr noalias [[SRC:%.*]], <4 x i32> [[THRESHOLD:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP0]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP0]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT: [[BROADCAST:%.*]] = call <vscale x 4 x i32> @llvm.vector.broadcast.nxv4i32.v4i32(<4 x i32> [[THRESHOLD]])
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds <4 x i32>, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP1]], align 16
+; CHECK-NEXT: [[TMP3:%.*]] = icmp sgt <vscale x 4 x i32> [[WIDE_LOAD]], [[BROADCAST]]
+; CHECK-NEXT: [[TMP4:%.*]] = select <vscale x 4 x i1> [[TMP3]], <vscale x 4 x i32> [[WIDE_LOAD]], <vscale x 4 x i32> [[BROADCAST]]
+; CHECK-NEXT: store <vscale x 4 x i32> [[TMP4]], ptr [[TMP2]], align 16
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP0]]
+; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP5]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], {{!llvm.loop ![0-9]+}}
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %src.gep = getelementptr inbounds <4 x i32>, ptr %src, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %iv
+ %v = load <4 x i32>, ptr %src.gep, align 16
+ %cmp = icmp sgt <4 x i32> %v, %threshold
+ %sel = select <4 x i1> %cmp, <4 x i32> %v, <4 x i32> %threshold
+ store <4 x i32> %sel, ptr %dst.gep, align 16
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-contiguous.ll b/llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-contiguous.ll
new file mode 100644
index 0000000000000..128443f63c079
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-contiguous.ll
@@ -0,0 +1,90 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "scalar.ph:" --replace-value-regex "!llvm.loop ![0-9]+" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -force-vector-interleave=1 -vectorize-vector-loops \
+; RUN: < %s | FileCheck %s --check-prefix=REVEC
+
+define void @copy_v4(ptr noalias %dst, ptr noalias %src, i64 %n) {
+; REVEC-LABEL: define void @copy_v4(
+; REVEC-SAME: ptr noalias [[DST:%.*]], ptr noalias [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; REVEC-NEXT: [[ENTRY:.*:]]
+; REVEC-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; REVEC-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP0]]
+; REVEC-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; REVEC: [[VECTOR_PH]]:
+; REVEC-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP0]]
+; REVEC-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; REVEC-NEXT: br label %[[VECTOR_BODY:.*]]
+; REVEC: [[VECTOR_BODY]]:
+; REVEC-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; REVEC-NEXT: [[TMP1:%.*]] = getelementptr inbounds <4 x i32>, ptr [[SRC]], i64 [[INDEX]]
+; REVEC-NEXT: [[TMP2:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[INDEX]]
+; REVEC-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP1]], align 16
+; REVEC-NEXT: store <vscale x 4 x i32> [[WIDE_LOAD]], ptr [[TMP2]], align 16
+; REVEC-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP0]]
+; REVEC-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; REVEC-NEXT: br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], {{!llvm.loop ![0-9]+}}
+; REVEC: [[MIDDLE_BLOCK]]:
+; REVEC-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; REVEC-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; REVEC: [[SCALAR_PH]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %src.gep = getelementptr inbounds <4 x i32>, ptr %src, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %iv
+ %v = load <4 x i32>, ptr %src.gep, align 16
+ store <4 x i32> %v, ptr %dst.gep, align 16
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @copy_v4_nontemporal(ptr noalias %dst, ptr noalias %src, i64 %n) {
+; REVEC-LABEL: define void @copy_v4_nontemporal(
+; REVEC-SAME: ptr noalias [[DST:%.*]], ptr noalias [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; REVEC-NEXT: [[ENTRY:.*:]]
+; REVEC-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; REVEC-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP0]]
+; REVEC-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; REVEC: [[VECTOR_PH]]:
+; REVEC-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP0]]
+; REVEC-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; REVEC-NEXT: br label %[[VECTOR_BODY:.*]]
+; REVEC: [[VECTOR_BODY]]:
+; REVEC-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; REVEC-NEXT: [[TMP1:%.*]] = getelementptr inbounds <4 x i32>, ptr [[SRC]], i64 [[INDEX]]
+; REVEC-NEXT: [[TMP2:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[INDEX]]
+; REVEC-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP1]], align 16, !nontemporal [[META4:![0-9]+]]
+; REVEC-NEXT: store <vscale x 4 x i32> [[WIDE_LOAD]], ptr [[TMP2]], align 16, !nontemporal [[META4]]
+; REVEC-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP0]]
+; REVEC-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; REVEC-NEXT: br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], {{!llvm.loop ![0-9]+}}
+; REVEC: [[MIDDLE_BLOCK]]:
+; REVEC-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; REVEC-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; REVEC: [[SCALAR_PH]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %src.gep = getelementptr inbounds <4 x i32>, ptr %src, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %iv
+ %v = load <4 x i32>, ptr %src.gep, align 16, !nontemporal !0
+ store <4 x i32> %v, ptr %dst.gep, align 16, !nontemporal !0
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+!0 = !{i32 1}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-gather-scatter.ll b/llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-gather-scatter.ll
new file mode 100644
index 0000000000000..cbb5aa7d8bd0a
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-gather-scatter.ll
@@ -0,0 +1,118 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter-out-after "scalar.ph:" --replace-value-regex "!llvm.loop ![0-9]+" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -scalable-vectorization=on -force-vector-interleave=1 \
+; RUN: -vectorize-vector-loops < %s | FileCheck %s
+
+define void @strided_v4(ptr noalias %dst, ptr noalias %src, i64 %n) {
+; CHECK-LABEL: define void @strided_v4(
+; CHECK-SAME: ptr noalias [[DST:%.*]], ptr noalias [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[IDX:%.*]] = shl nuw i64 [[IV]], 1
+; CHECK-NEXT: [[SRC_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[SRC]], i64 [[IDX]]
+; CHECK-NEXT: [[DST_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[IDX]]
+; CHECK-NEXT: [[V:%.*]] = load <4 x i32>, ptr [[SRC_GEP]], align 16
+; CHECK-NEXT: store <4 x i32> [[V]], ptr [[DST_GEP]], align 16
+; CHECK-NEXT: [[IV_NEXT]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %idx = shl nuw i64 %iv, 1
+ %src.gep = getelementptr inbounds <4 x i32>, ptr %src, i64 %idx
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %idx
+ %v = load <4 x i32>, ptr %src.gep, align 16
+ store <4 x i32> %v, ptr %dst.gep, align 16
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @gather_128(ptr noalias nocapture noundef writeonly %a, ptr nocapture noundef readonly %b, i64 %stride) {
+; CHECK-LABEL: define void @gather_128(
+; CHECK-SAME: ptr noalias noundef writeonly captures(none) [[A:%.*]], ptr noundef readonly captures(none) [[B:%.*]], i64 [[STRIDE:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[VECTOR_PH:.*]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[FOR_COND_CLEANUP:.*]]:
+; CHECK-NEXT: ret void
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDVARS_IV_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[OFFSET:%.*]] = mul i64 [[INDEX]], [[STRIDE]]
+; CHECK-NEXT: [[ARRAYIDX0:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[OFFSET]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <8 x i16>, ptr [[ARRAYIDX0]], align 16
+; CHECK-NEXT: [[RESULT:%.*]] = add <8 x i16> [[TMP0]], splat (i16 1)
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds <8 x i16>, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: store <8 x i16> [[RESULT]], ptr [[TMP6]], align 16
+; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDEX]], 1
+; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], 1024
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label %[[FOR_COND_CLEANUP]], label %[[VECTOR_BODY]]
+;
+entry:
+ br label %for.body
+
+for.cond.cleanup: ; preds = %for.body
+ ret void
+
+for.body: ; preds = %entry, %for.body
+ %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %for.body ]
+ %offset = mul i64 %indvars.iv, %stride
+ %arrayidx0 = getelementptr inbounds i16, ptr %b, i64 %offset
+ %0 = load <8 x i16>, ptr %arrayidx0, align 16
+ %result = add <8 x i16> %0, splat (i16 1)
+ %arrayidx2 = getelementptr inbounds <8 x i16>, ptr %a, i64 %indvars.iv
+ store <8 x i16> %result, ptr %arrayidx2, align 16
+ %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+ %exitcond.not = icmp eq i64 %indvars.iv.next, 1024
+ br i1 %exitcond.not, label %for.cond.cleanup, label %for.body
+}
+
+
+
+define void @scatter_128(ptr noalias nocapture noundef writeonly %a, ptr nocapture noundef readonly %b, i64 %stride) {
+; CHECK-LABEL: define void @scatter_128(
+; CHECK-SAME: ptr noalias noundef writeonly captures(none) [[A:%.*]], ptr noundef readonly captures(none) [[B:%.*]], i64 [[STRIDE:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[VECTOR_PH:.*]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[FOR_COND_CLEANUP:.*]]:
+; CHECK-NEXT: ret void
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDVARS_IV_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds <8 x i16>, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <8 x i16>, ptr [[TMP2]], align 16
+; CHECK-NEXT: [[RESULT:%.*]] = add <8 x i16> [[TMP0]], splat (i16 1)
+; CHECK-NEXT: [[OFFSET:%.*]] = mul i64 [[INDEX]], [[STRIDE]]
+; CHECK-NEXT: [[ARRAYIDX2:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[OFFSET]]
+; CHECK-NEXT: store <8 x i16> [[RESULT]], ptr [[ARRAYIDX2]], align 16
+; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDEX]], 1
+; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], 1024
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label %[[FOR_COND_CLEANUP]], label %[[VECTOR_BODY]]
+;
+entry:
+ br label %for.body
+
+for.cond.cleanup: ; preds = %for.body
+ ret void
+
+for.body: ; preds = %entry, %for.body
+ %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %for.body ]
+ %arrayidx0 = getelementptr inbounds <8 x i16>, ptr %b, i64 %indvars.iv
+ %0 = load <8 x i16>, ptr %arrayidx0, align 16
+ %result = add <8 x i16> %0, splat (i16 1)
+ %offset = mul i64 %indvars.iv, %stride
+ %arrayidx2 = getelementptr inbounds i16, ptr %a, i64 %offset
+ store <8 x i16> %result, ptr %arrayidx2, align 16
+ %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+ %exitcond.not = icmp eq i64 %indvars.iv.next, 1024
+ br i1 %exitcond.not, label %for.cond.cleanup, label %for.body
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-interleaved.ll b/llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-interleaved.ll
new file mode 100644
index 0000000000000..a78af8d6fe473
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/revec-memory-interleaved.ll
@@ -0,0 +1,89 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter-out-after "scalar.ph:" --replace-value-regex "!llvm.loop ![0-9]+" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -scalable-vectorization=on -force-vector-interleave=1 \
+; RUN: -vectorize-vector-loops < %s | FileCheck %s
+
+
+define void @ld2q(ptr noalias nocapture noundef writeonly %a, ptr nocapture noundef readonly %b) {
+; CHECK-LABEL: define void @ld2q(
+; CHECK-SAME: ptr noalias noundef writeonly captures(none) [[A:%.*]], ptr noundef readonly captures(none) [[B:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_COND_CLEANUP:.*]]:
+; CHECK-NEXT: ret void
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[INDVARS_IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[ARRAYIDX0:%.*]] = getelementptr inbounds [16 x i16], ptr [[B]], i64 [[INDVARS_IV]], i64 0
+; CHECK-NEXT: [[TMP0:%.*]] = load <8 x i16>, ptr [[ARRAYIDX0]], align 16
+; CHECK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds [16 x i16], ptr [[B]], i64 [[INDVARS_IV]], i64 8
+; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[ARRAYIDX1]], align 16
+; CHECK-NEXT: [[RESULT:%.*]] = add <8 x i16> [[TMP0]], [[TMP1]]
+; CHECK-NEXT: [[ARRAYIDX2:%.*]] = getelementptr inbounds <8 x i16>, ptr [[A]], i64 [[INDVARS_IV]]
+; CHECK-NEXT: store <8 x i16> [[RESULT]], ptr [[ARRAYIDX2]], align 16
+; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
+; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], 1024
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label %[[FOR_COND_CLEANUP]], label %[[FOR_BODY]]
+;
+entry:
+ br label %for.body
+
+for.cond.cleanup: ; preds = %for.body
+ ret void
+
+for.body: ; preds = %entry, %for.body
+ %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %for.body ]
+ %arrayidx0 = getelementptr inbounds [16 x i16], ptr %b, i64 %indvars.iv, i64 0
+ %0 = load <8 x i16>, ptr %arrayidx0, align 16
+ %arrayidx1 = getelementptr inbounds [16 x i16], ptr %b, i64 %indvars.iv, i64 8
+ %1 = load <8 x i16>, ptr %arrayidx1, align 16
+ %result = add <8 x i16> %0, %1
+ %arrayidx2 = getelementptr inbounds <8 x i16>, ptr %a, i64 %indvars.iv
+ store <8 x i16> %result, ptr %arrayidx2, align 16
+ %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+ %exitcond.not = icmp eq i64 %indvars.iv.next, 1024
+ br i1 %exitcond.not, label %for.cond.cleanup, label %for.body
+}
+
+define void @st2q(ptr noalias nocapture noundef writeonly %a, ptr nocapture noundef readonly %b) {
+; CHECK-LABEL: define void @st2q(
+; CHECK-SAME: ptr noalias noundef writeonly captures(none) [[A:%.*]], ptr noundef readonly captures(none) [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_COND_CLEANUP:.*]]:
+; CHECK-NEXT: ret void
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[INDVARS_IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[ARRAYIDX_IN:%.*]] = getelementptr inbounds <8 x i16>, ptr [[B]], i64 [[INDVARS_IV]]
+; CHECK-NEXT: [[TMP9:%.*]] = load <8 x i16>, ptr [[ARRAYIDX_IN]], align 16
+; CHECK-NEXT: [[V0:%.*]] = add <8 x i16> [[TMP9]], splat (i16 1)
+; CHECK-NEXT: [[V1:%.*]] = add <8 x i16> [[V0]], splat (i16 1)
+; CHECK-NEXT: [[ARRAYIDX_OUT0:%.*]] = getelementptr inbounds [16 x i16], ptr [[A]], i64 [[INDVARS_IV]], i64 0
+; CHECK-NEXT: store <8 x i16> [[V0]], ptr [[ARRAYIDX_OUT0]], align 16
+; CHECK-NEXT: [[ARRAYIDX_OUT1:%.*]] = getelementptr inbounds [16 x i16], ptr [[A]], i64 [[INDVARS_IV]], i64 8
+; CHECK-NEXT: store <8 x i16> [[V1]], ptr [[ARRAYIDX_OUT1]], align 16
+; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
+; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], 1024
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label %[[FOR_COND_CLEANUP]], label %[[FOR_BODY]]
+;
+entry:
+ br label %for.body
+
+for.cond.cleanup: ; preds = %for.body
+ ret void
+
+for.body: ; preds = %entry, %for.body
+ %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %for.body ]
+ %arrayidx.in = getelementptr inbounds <8 x i16>, ptr %b, i64 %indvars.iv
+ %0 = load <8 x i16>, ptr %arrayidx.in, align 16
+ %v0 = add <8 x i16> %0, splat (i16 1)
+ %v1 = add <8 x i16> %v0, splat (i16 1)
+
+ %arrayidx.out0 = getelementptr inbounds [16 x i16], ptr %a, i64 %indvars.iv, i64 0
+ store <8 x i16> %v0, ptr %arrayidx.out0, align 16
+ %arrayidx.out1 = getelementptr inbounds [16 x i16], ptr %a, i64 %indvars.iv, i64 8
+ store <8 x i16> %v1, ptr %arrayidx.out1, align 16
+
+ %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+ %exitcond.not = icmp eq i64 %indvars.iv.next, 1024
+ br i1 %exitcond.not, label %for.cond.cleanup, label %for.body
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/revec-predication.ll b/llvm/test/Transforms/LoopVectorize/AArch64/revec-predication.ll
new file mode 100644
index 0000000000000..f3a3ba7f00541
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/revec-predication.ll
@@ -0,0 +1,102 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter-out-after "scalar.ph:" --replace-value-regex "!llvm.loop ![0-9]+" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -scalable-vectorization=on -force-vector-interleave=1 \
+; RUN: -vectorize-vector-loops < %s | FileCheck %s
+
+define void @predicated_v4(ptr noalias %dst, ptr noalias %src, i64 %n, i64 %limit) {
+; CHECK-LABEL: define void @predicated_v4(
+; CHECK-SAME: ptr noalias [[DST:%.*]], ptr noalias [[SRC:%.*]], i64 [[N:%.*]], i64 [[LIMIT:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT: [[PRED:%.*]] = icmp ult i64 [[IV]], [[LIMIT]]
+; CHECK-NEXT: br i1 [[PRED]], label %[[THEN:.*]], label %[[LATCH]]
+; CHECK: [[THEN]]:
+; CHECK-NEXT: [[SRC_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT: [[DST_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT: [[V:%.*]] = load <4 x i32>, ptr [[SRC_GEP]], align 16
+; CHECK-NEXT: store <4 x i32> [[V]], ptr [[DST_GEP]], align 16
+; CHECK-NEXT: br label %[[LATCH]]
+; CHECK: [[LATCH]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %pred = icmp ult i64 %iv, %limit
+ br i1 %pred, label %then, label %latch
+
+then:
+ %src.gep = getelementptr inbounds <4 x i32>, ptr %src, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %iv
+ %v = load <4 x i32>, ptr %src.gep, align 16
+ store <4 x i32> %v, ptr %dst.gep, align 16
+ br label %latch
+
+latch:
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @predicated_load_uniform_cond(ptr noalias nocapture noundef writeonly %a, ptr nocapture noundef readonly %b, ptr nocapture noundef readonly %c, i1 %cond) {
+; CHECK-LABEL: define void @predicated_load_uniform_cond(
+; CHECK-SAME: ptr noalias noundef writeonly captures(none) [[A:%.*]], ptr noundef readonly captures(none) [[B:%.*]], ptr noundef readonly captures(none) [[C:%.*]], i1 [[COND:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_COND_CLEANUP:.*]]:
+; CHECK-NEXT: ret void
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDVARS_IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ARRAYIDX0:%.*]] = getelementptr inbounds <8 x i16>, ptr [[B]], i64 [[INDVARS_IV]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <8 x i16>, ptr [[ARRAYIDX0]], align 16
+; CHECK-NEXT: br i1 [[COND]], label %[[IF_THEN:.*]], label %[[FOR_INC]]
+; CHECK: [[IF_THEN]]:
+; CHECK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds <8 x i16>, ptr [[C]], i64 [[INDVARS_IV]]
+; CHECK-NEXT: [[PRED_1:%.*]] = load <8 x i16>, ptr [[ARRAYIDX1]], align 16
+; CHECK-NEXT: br label %[[FOR_INC]]
+; CHECK: [[FOR_INC]]:
+; CHECK-NEXT: [[TMP1:%.*]] = phi <8 x i16> [ zeroinitializer, %[[FOR_BODY]] ], [ [[PRED_1]], %[[IF_THEN]] ]
+; CHECK-NEXT: [[RESULT:%.*]] = add <8 x i16> [[TMP0]], [[TMP1]]
+; CHECK-NEXT: [[ARRAYIDX2:%.*]] = getelementptr inbounds <8 x i16>, ptr [[A]], i64 [[INDVARS_IV]]
+; CHECK-NEXT: store <8 x i16> [[RESULT]], ptr [[ARRAYIDX2]], align 16
+; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
+; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], 1024
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label %[[FOR_COND_CLEANUP]], label %[[FOR_BODY]]
+;
+entry:
+ br label %for.body
+
+for.cond.cleanup: ; preds = %for.body
+ ret void
+
+for.body: ; preds = %entry, %for.inc
+ %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %for.inc ]
+ %arrayidx0 = getelementptr inbounds <8 x i16>, ptr %b, i64 %indvars.iv
+ %0 = load <8 x i16>, ptr %arrayidx0, align 16
+ br i1 %cond, label %if.then, label %for.inc
+
+if.then:
+ %arrayidx1 = getelementptr inbounds <8 x i16>, ptr %c, i64 %indvars.iv
+ %pred.1 = load <8 x i16>, ptr %arrayidx1, align 16
+ br label %for.inc
+
+for.inc:
+ %1 = phi <8 x i16> [ zeroinitializer, %for.body ], [ %pred.1, %if.then ]
+ %result = add <8 x i16> %0, %1
+ %arrayidx2 = getelementptr inbounds <8 x i16>, ptr %a, i64 %indvars.iv
+ store <8 x i16> %result, ptr %arrayidx2, align 16
+ %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+ %exitcond.not = icmp eq i64 %indvars.iv.next, 1024
+ br i1 %exitcond.not, label %for.cond.cleanup, label %for.body
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/revec-select.ll b/llvm/test/Transforms/LoopVectorize/AArch64/revec-select.ll
new file mode 100644
index 0000000000000..2d837a5236e51
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/revec-select.ll
@@ -0,0 +1,293 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter-out-after "scalar.ph:" --replace-value-regex "!llvm.loop ![0-9]+" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -scalable-vectorization=on -force-vector-interleave=1 \
+; RUN: -vectorize-vector-loops < %s | FileCheck %s
+
+define void @icmp_select_v4(ptr noalias %dst, ptr noalias %lhs, ptr noalias %rhs, i64 %n) {
+; CHECK-LABEL: define void @icmp_select_v4(
+; CHECK-SAME: ptr noalias [[DST:%.*]], ptr noalias [[LHS:%.*]], ptr noalias [[RHS:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP0]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP0]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds <4 x i32>, ptr [[LHS]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds <4 x i32>, ptr [[RHS]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP1]], align 16
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 4 x i32>, ptr [[TMP2]], align 16
+; CHECK-NEXT: [[TMP4:%.*]] = icmp sgt <vscale x 4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP5:%.*]] = select <vscale x 4 x i1> [[TMP4]], <vscale x 4 x i32> [[WIDE_LOAD]], <vscale x 4 x i32> [[WIDE_LOAD1]]
+; CHECK-NEXT: store <vscale x 4 x i32> [[TMP5]], ptr [[TMP3]], align 16
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP0]]
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], {{!llvm.loop ![0-9]+}}
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %lhs.gep = getelementptr inbounds <4 x i32>, ptr %lhs, i64 %iv
+ %rhs.gep = getelementptr inbounds <4 x i32>, ptr %rhs, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %iv
+ %a = load <4 x i32>, ptr %lhs.gep, align 16
+ %b = load <4 x i32>, ptr %rhs.gep, align 16
+ %cmp = icmp sgt <4 x i32> %a, %b
+ %sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %b
+ store <4 x i32> %sel, ptr %dst.gep, align 16
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @icmp_select_v4_uniform_cond(ptr noalias %dst, ptr noalias %lhs, ptr noalias %rhs, i64 %n, <4 x i32> %conds) {
+; CHECK-LABEL: define void @icmp_select_v4_uniform_cond(
+; CHECK-SAME: ptr noalias [[DST:%.*]], ptr noalias [[LHS:%.*]], ptr noalias [[RHS:%.*]], i64 [[N:%.*]], <4 x i32> [[CONDS:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[CMP:%.*]] = icmp sgt <4 x i32> [[CONDS]], zeroinitializer
+; CHECK-NEXT: [[BROADCAST:%.*]] = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v4i1(<4 x i1> [[CMP]])
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP0]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP0]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds <4 x i32>, ptr [[LHS]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds <4 x i32>, ptr [[RHS]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP1]], align 16
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 4 x i32>, ptr [[TMP2]], align 16
+; CHECK-NEXT: [[TMP4:%.*]] = select <vscale x 4 x i1> [[BROADCAST]], <vscale x 4 x i32> [[WIDE_LOAD]], <vscale x 4 x i32> [[WIDE_LOAD1]]
+; CHECK-NEXT: store <vscale x 4 x i32> [[TMP4]], ptr [[TMP3]], align 16
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP0]]
+; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP5]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], {{!llvm.loop ![0-9]+}}
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+;
+entry:
+ %cmp = icmp sgt <4 x i32> %conds, splat (i32 0)
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %lhs.gep = getelementptr inbounds <4 x i32>, ptr %lhs, i64 %iv
+ %rhs.gep = getelementptr inbounds <4 x i32>, ptr %rhs, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %iv
+ %a = load <4 x i32>, ptr %lhs.gep, align 16
+ %b = load <4 x i32>, ptr %rhs.gep, align 16
+ %sel = select <4 x i1> %cmp, <4 x i32> %a, <4 x i32> %b
+ store <4 x i32> %sel, ptr %dst.gep, align 16
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @icmp_select_scalar_cond_uniform(ptr noalias %dst, ptr noalias %lhs, ptr noalias %rhs, i64 %n, i1 %cond) {
+; CHECK-LABEL: define void @icmp_select_scalar_cond_uniform(
+; CHECK-SAME: ptr noalias [[DST:%.*]], ptr noalias [[LHS:%.*]], ptr noalias [[RHS:%.*]], i64 [[N:%.*]], i1 [[COND:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP0]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP0]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds <4 x i32>, ptr [[LHS]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds <4 x i32>, ptr [[RHS]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP1]], align 16
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 4 x i32>, ptr [[TMP2]], align 16
+; CHECK-NEXT: [[TMP4:%.*]] = select i1 [[COND]], <vscale x 4 x i32> [[WIDE_LOAD]], <vscale x 4 x i32> [[WIDE_LOAD1]]
+; CHECK-NEXT: store <vscale x 4 x i32> [[TMP4]], ptr [[TMP3]], align 16
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP0]]
+; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP5]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], {{!llvm.loop ![0-9]+}}
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %lhs.gep = getelementptr inbounds <4 x i32>, ptr %lhs, i64 %iv
+ %rhs.gep = getelementptr inbounds <4 x i32>, ptr %rhs, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %iv
+ %a = load <4 x i32>, ptr %lhs.gep, align 16
+ %b = load <4 x i32>, ptr %rhs.gep, align 16
+ %sel = select i1 %cond, <4 x i32> %a, <4 x i32> %b
+ store <4 x i32> %sel, ptr %dst.gep, align 16
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+; TODO-REVEC: LV does not handle stretching a nxv1i1 predicate to nxv4i1
+define void @icmp_select_scalar_cond_load(ptr noalias %dst, ptr noalias %lhs, ptr noalias %rhs, ptr noalias %cond, i64 %n) {
+; CHECK-LABEL: define void @icmp_select_scalar_cond_load(
+; CHECK-SAME: ptr noalias [[DST:%.*]], ptr noalias [[LHS:%.*]], ptr noalias [[RHS:%.*]], ptr noalias [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[LHS_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[LHS]], i64 [[IV]]
+; CHECK-NEXT: [[RHS_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[RHS]], i64 [[IV]]
+; CHECK-NEXT: [[COND_GEP:%.*]] = getelementptr inbounds i8, ptr [[COND]], i64 [[IV]]
+; CHECK-NEXT: [[DST_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT: [[A:%.*]] = load <4 x i32>, ptr [[LHS_GEP]], align 16
+; CHECK-NEXT: [[B:%.*]] = load <4 x i32>, ptr [[RHS_GEP]], align 16
+; CHECK-NEXT: [[COND_I8:%.*]] = load i8, ptr [[COND_GEP]], align 1
+; CHECK-NEXT: [[COND_I1:%.*]] = icmp ne i8 [[COND_I8]], 0
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[COND_I1]], <4 x i32> [[A]], <4 x i32> [[B]]
+; CHECK-NEXT: store <4 x i32> [[SEL]], ptr [[DST_GEP]], align 16
+; CHECK-NEXT: [[IV_NEXT]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %lhs.gep = getelementptr inbounds <4 x i32>, ptr %lhs, i64 %iv
+ %rhs.gep = getelementptr inbounds <4 x i32>, ptr %rhs, i64 %iv
+ %cond.gep = getelementptr inbounds i8, ptr %cond, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %iv
+ %a = load <4 x i32>, ptr %lhs.gep, align 16
+ %b = load <4 x i32>, ptr %rhs.gep, align 16
+ %cond.i8 = load i8, ptr %cond.gep, align 1
+ %cond.i1 = icmp ne i8 %cond.i8, 0
+ %sel = select i1 %cond.i1, <4 x i32> %a, <4 x i32> %b
+ store <4 x i32> %sel, ptr %dst.gep, align 16
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+; TODO-REVEC: LV does not handle stretching a nxv1i1 predicate to nxv4i1
+define void @icmp_select_scalar_cond_iv(ptr noalias %dst, ptr noalias %lhs, ptr noalias %rhs, i64 %n) {
+; CHECK-LABEL: define void @icmp_select_scalar_cond_iv(
+; CHECK-SAME: ptr noalias [[DST:%.*]], ptr noalias [[LHS:%.*]], ptr noalias [[RHS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[LHS_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[LHS]], i64 [[IV]]
+; CHECK-NEXT: [[RHS_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[RHS]], i64 [[IV]]
+; CHECK-NEXT: [[DST_GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT: [[A:%.*]] = load <4 x i32>, ptr [[LHS_GEP]], align 16
+; CHECK-NEXT: [[B:%.*]] = load <4 x i32>, ptr [[RHS_GEP]], align 16
+; CHECK-NEXT: [[COND_I1:%.*]] = icmp ugt i64 [[IV]], 8
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[COND_I1]], <4 x i32> [[A]], <4 x i32> [[B]]
+; CHECK-NEXT: store <4 x i32> [[SEL]], ptr [[DST_GEP]], align 16
+; CHECK-NEXT: [[IV_NEXT]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %lhs.gep = getelementptr inbounds <4 x i32>, ptr %lhs, i64 %iv
+ %rhs.gep = getelementptr inbounds <4 x i32>, ptr %rhs, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %iv
+ %a = load <4 x i32>, ptr %lhs.gep, align 16
+ %b = load <4 x i32>, ptr %rhs.gep, align 16
+ %cond.i1 = icmp ugt i64 %iv, 8
+ %sel = select i1 %cond.i1, <4 x i32> %a, <4 x i32> %b
+ store <4 x i32> %sel, ptr %dst.gep, align 16
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @fcmp_select_v4(ptr noalias %dst, ptr noalias %lhs, ptr noalias %rhs, i64 %n) {
+; CHECK-LABEL: define void @fcmp_select_v4(
+; CHECK-SAME: ptr noalias [[DST:%.*]], ptr noalias [[LHS:%.*]], ptr noalias [[RHS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP0]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP0]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds <4 x float>, ptr [[LHS]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds <4 x float>, ptr [[RHS]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds <4 x float>, ptr [[DST]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x float>, ptr [[TMP1]], align 16
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 4 x float>, ptr [[TMP2]], align 16
+; CHECK-NEXT: [[TMP4:%.*]] = fcmp olt <vscale x 4 x float> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP5:%.*]] = select <vscale x 4 x i1> [[TMP4]], <vscale x 4 x float> [[WIDE_LOAD]], <vscale x 4 x float> [[WIDE_LOAD1]]
+; CHECK-NEXT: store <vscale x 4 x float> [[TMP5]], ptr [[TMP3]], align 16
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP0]]
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], {{!llvm.loop ![0-9]+}}
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %lhs.gep = getelementptr inbounds <4 x float>, ptr %lhs, i64 %iv
+ %rhs.gep = getelementptr inbounds <4 x float>, ptr %rhs, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x float>, ptr %dst, i64 %iv
+ %a = load <4 x float>, ptr %lhs.gep, align 16
+ %b = load <4 x float>, ptr %rhs.gep, align 16
+ %cmp = fcmp olt <4 x float> %a, %b
+ %sel = select <4 x i1> %cmp, <4 x float> %a, <4 x float> %b
+ store <4 x float> %sel, ptr %dst.gep, align 16
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/revec-unroll.ll b/llvm/test/Transforms/LoopVectorize/AArch64/revec-unroll.ll
new file mode 100644
index 0000000000000..feaef7eaaf1d3
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/revec-unroll.ll
@@ -0,0 +1,100 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter-out-after "scalar.ph:" --replace-value-regex "!llvm.loop ![0-9]+" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -force-vector-width=1 -force-vector-interleave=2 \
+; RUN: -vectorize-vector-loops < %s | FileCheck %s --check-prefix=VF1IC2
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve \
+; RUN: -force-vector-width="vscale x 1" -force-vector-interleave=2 \
+; RUN: -vectorize-vector-loops < %s | FileCheck %s --check-prefix=SVF1IC2
+
+; Verify that we can interleave vector loop iterations.
+
+define void @load_add_store(ptr noalias %dst, ptr noalias %lhs, ptr noalias %rhs, i64 %n) {
+; VF1IC2-LABEL: define void @load_add_store(
+; VF1IC2-SAME: ptr noalias [[DST:%.*]], ptr noalias [[LHS:%.*]], ptr noalias [[RHS:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; VF1IC2-NEXT: [[ENTRY:.*:]]
+; VF1IC2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 2
+; VF1IC2-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; VF1IC2: [[VECTOR_PH]]:
+; VF1IC2-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N]], 2
+; VF1IC2-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; VF1IC2-NEXT: br label %[[VECTOR_BODY:.*]]
+; VF1IC2: [[VECTOR_BODY]]:
+; VF1IC2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; VF1IC2-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; VF1IC2-NEXT: [[TMP1:%.*]] = getelementptr inbounds <4 x i32>, ptr [[LHS]], i64 [[INDEX]]
+; VF1IC2-NEXT: [[TMP2:%.*]] = getelementptr inbounds <4 x i32>, ptr [[LHS]], i64 [[TMP0]]
+; VF1IC2-NEXT: [[TMP3:%.*]] = getelementptr inbounds <4 x i32>, ptr [[RHS]], i64 [[INDEX]]
+; VF1IC2-NEXT: [[TMP4:%.*]] = getelementptr inbounds <4 x i32>, ptr [[RHS]], i64 [[TMP0]]
+; VF1IC2-NEXT: [[TMP5:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[INDEX]]
+; VF1IC2-NEXT: [[TMP6:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[TMP0]]
+; VF1IC2-NEXT: [[TMP7:%.*]] = load <4 x i32>, ptr [[TMP1]], align 16
+; VF1IC2-NEXT: [[TMP8:%.*]] = load <4 x i32>, ptr [[TMP2]], align 16
+; VF1IC2-NEXT: [[TMP9:%.*]] = load <4 x i32>, ptr [[TMP3]], align 16
+; VF1IC2-NEXT: [[TMP10:%.*]] = load <4 x i32>, ptr [[TMP4]], align 16
+; VF1IC2-NEXT: [[TMP11:%.*]] = add <4 x i32> [[TMP7]], [[TMP9]]
+; VF1IC2-NEXT: [[TMP12:%.*]] = add <4 x i32> [[TMP8]], [[TMP10]]
+; VF1IC2-NEXT: store <4 x i32> [[TMP11]], ptr [[TMP5]], align 16
+; VF1IC2-NEXT: store <4 x i32> [[TMP12]], ptr [[TMP6]], align 16
+; VF1IC2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 2
+; VF1IC2-NEXT: [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; VF1IC2-NEXT: br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], {{!llvm.loop ![0-9]+}}
+; VF1IC2: [[MIDDLE_BLOCK]]:
+; VF1IC2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; VF1IC2-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; VF1IC2: [[SCALAR_PH]]:
+;
+; SVF1IC2-LABEL: define void @load_add_store(
+; SVF1IC2-SAME: ptr noalias [[DST:%.*]], ptr noalias [[LHS:%.*]], ptr noalias [[RHS:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; SVF1IC2-NEXT: [[ENTRY:.*:]]
+; SVF1IC2-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; SVF1IC2-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 1
+; SVF1IC2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; SVF1IC2-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; SVF1IC2: [[VECTOR_PH]]:
+; SVF1IC2-NEXT: [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 1
+; SVF1IC2-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; SVF1IC2-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; SVF1IC2-NEXT: br label %[[VECTOR_BODY:.*]]
+; SVF1IC2: [[VECTOR_BODY]]:
+; SVF1IC2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; SVF1IC2-NEXT: [[TMP3:%.*]] = getelementptr inbounds <4 x i32>, ptr [[LHS]], i64 [[INDEX]]
+; SVF1IC2-NEXT: [[TMP4:%.*]] = getelementptr inbounds <4 x i32>, ptr [[RHS]], i64 [[INDEX]]
+; SVF1IC2-NEXT: [[TMP5:%.*]] = getelementptr inbounds <4 x i32>, ptr [[DST]], i64 [[INDEX]]
+; SVF1IC2-NEXT: [[TMP6:%.*]] = getelementptr inbounds <4 x i32>, ptr [[TMP3]], i64 [[TMP0]]
+; SVF1IC2-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP3]], align 16
+; SVF1IC2-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 4 x i32>, ptr [[TMP6]], align 16
+; SVF1IC2-NEXT: [[TMP7:%.*]] = getelementptr inbounds <4 x i32>, ptr [[TMP4]], i64 [[TMP0]]
+; SVF1IC2-NEXT: [[WIDE_LOAD2:%.*]] = load <vscale x 4 x i32>, ptr [[TMP4]], align 16
+; SVF1IC2-NEXT: [[WIDE_LOAD3:%.*]] = load <vscale x 4 x i32>, ptr [[TMP7]], align 16
+; SVF1IC2-NEXT: [[TMP8:%.*]] = add <vscale x 4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; SVF1IC2-NEXT: [[TMP9:%.*]] = add <vscale x 4 x i32> [[WIDE_LOAD1]], [[WIDE_LOAD3]]
+; SVF1IC2-NEXT: [[TMP10:%.*]] = getelementptr inbounds <4 x i32>, ptr [[TMP5]], i64 [[TMP0]]
+; SVF1IC2-NEXT: store <vscale x 4 x i32> [[TMP8]], ptr [[TMP5]], align 16
+; SVF1IC2-NEXT: store <vscale x 4 x i32> [[TMP9]], ptr [[TMP10]], align 16
+; SVF1IC2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; SVF1IC2-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; SVF1IC2-NEXT: br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], {{!llvm.loop ![0-9]+}}
+; SVF1IC2: [[MIDDLE_BLOCK]]:
+; SVF1IC2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; SVF1IC2-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; SVF1IC2: [[SCALAR_PH]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %lhs.gep = getelementptr inbounds <4 x i32>, ptr %lhs, i64 %iv
+ %rhs.gep = getelementptr inbounds <4 x i32>, ptr %rhs, i64 %iv
+ %dst.gep = getelementptr inbounds <4 x i32>, ptr %dst, i64 %iv
+ %a = load <4 x i32>, ptr %lhs.gep, align 16
+ %b = load <4 x i32>, ptr %rhs.gep, align 16
+ %add = add <4 x i32> %a, %b
+ store <4 x i32> %add, ptr %dst.gep, align 16
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, %n
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
More information about the llvm-branch-commits
mailing list