[llvm] [LV] Support masked interleaved access with gaps for scalable vector (PR #196479)
Mel Chen via llvm-commits
llvm-commits at lists.llvm.org
Thu Oct 1 06:46:07 PDT 2026
https://github.com/Mel-Chen updated https://github.com/llvm/llvm-project/pull/196479
>From b9d50336917287166ca13a96fe107851671d11b8 Mon Sep 17 00:00:00 2001
From: Mel Chen <mel.chen at sifive.com>
Date: Thu, 7 May 2026 23:20:16 -0700
Subject: [PATCH] Initial patch for scalable interleave access with gap mask
---
.../Target/RISCV/RISCVTargetTransformInfo.cpp | 27 ++++---
.../Transforms/Vectorize/LoopVectorize.cpp | 6 --
llvm/lib/Transforms/Vectorize/VPlan.h | 2 -
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 72 +++++++++++--------
.../LoopVectorize/RISCV/pointer-induction.ll | 52 ++++----------
.../RISCV/tail-folding-interleave.ll | 48 +++++--------
6 files changed, 88 insertions(+), 119 deletions(-)
diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index 69554ca2b8155..0a16811a5581f 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -1214,13 +1214,12 @@ InstructionCost RISCVTTIImpl::getInterleavedMemoryOpCost(
unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
bool UseMaskForCond, bool UseMaskForGaps) const {
-
+ auto *VTy = cast<VectorType>(VecTy);
// The interleaved memory access pass will lower (de)interleave ops combined
// with an adjacent appropriate memory to vlseg/vsseg intrinsics. vlseg/vsseg
// only support masking per-iteration (i.e. condition), not per-segment (i.e.
// gap).
if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
- auto *VTy = cast<VectorType>(VecTy);
std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(VTy);
// Need to make sure type has't been scalarized
if (LT.second.isVector()) {
@@ -1256,12 +1255,6 @@ InstructionCost RISCVTTIImpl::getInterleavedMemoryOpCost(
}
}
- // TODO: Return the cost of interleaved accesses for scalable vector when
- // unable to convert to segment accesses instructions.
- if (isa<ScalableVectorType>(VecTy))
- return InstructionCost::getInvalid();
-
- auto *FVTy = cast<FixedVectorType>(VecTy);
// When gaps are only at the tail, for interleaved load, we can emit a wide
// masked load and shufflevectors. For interleaved store, we can emit
// shufflevectors and a wide masked store. The interleaved memory access pass
@@ -1274,22 +1267,28 @@ InstructionCost RISCVTTIImpl::getInterleavedMemoryOpCost(
bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.back() + 1);
if (IsTailGapOnly &&
NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
- std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(FVTy);
+ std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(VTy);
if (LT.second.isVector() &&
- FVTy->getElementCount().isKnownMultipleOf(Factor)) {
- auto *SubVecTy = VectorType::get(
- FVTy->getElementType(),
- FVTy->getElementCount().divideCoefficientBy(Factor));
+ VTy->getElementCount().isKnownMultipleOf(Factor)) {
+ auto *SubVecTy =
+ VectorType::get(VTy->getElementType(),
+ VTy->getElementCount().divideCoefficientBy(Factor));
if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
AddressSpace, DL)) {
// The cost is proportional to the total number of element accesses.
- unsigned NumAccesses = getEstimatedVLFor(FVTy);
+ unsigned NumAccesses = getEstimatedVLFor(VTy);
return NumAccesses * TTI::TCC_Basic;
}
}
}
}
+ // TODO: Return the cost of interleaved accesses for scalable vector when
+ // unable to convert to segment accesses instructions.
+ if (isa<ScalableVectorType>(VecTy))
+ return InstructionCost::getInvalid();
+
+ auto *FVTy = cast<FixedVectorType>(VecTy);
InstructionCost MemCost =
getMemoryOpCost(Opcode, VecTy, Alignment, AddressSpace, CostKind);
unsigned VF = FVTy->getNumElements() / Factor;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 0986d8ea9e59d..dfed3d6c567e9 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -2630,12 +2630,6 @@ bool LoopVectorizationCostModel::interleavedAccessCanBeWidened(
if (Group->isReverse())
return false;
- // TODO: Support interleaved access that requires a gap mask for scalable VFs.
- bool NeedsMaskForGaps = LoadAccessWithGapsRequiresEpilogMasking ||
- StoreAccessWithGapsRequiresMasking;
- if (VF.isScalable() && NeedsMaskForGaps)
- return false;
-
return isLegalMaskedLoadOrStore(I, VF);
}
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index aa93dbaa65170..066ee6a478560 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -3178,8 +3178,6 @@ class LLVM_ABI_FOR_TEST VPInterleaveEVLRecipe final : public VPInterleaveBase {
R.getDebugLoc()) {
assert(!getInterleaveGroup()->isReverse() &&
"Reversed interleave-group with tail folding is not supported.");
- assert(!needsMaskForGaps() && "Interleaved access with gap mask is not "
- "supported for scalable vector.");
}
~VPInterleaveEVLRecipe() override = default;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 5c31590283dbb..8b6f3cd44ab59 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -4643,8 +4643,6 @@ static Value *interleaveVectors(IRBuilderBase &Builder, ArrayRef<Value *> Vals,
// <0, 4, 8, 1, 5, 9, 2, 6, 10, 3, 7, 11> ; Interleave R,G,B elements
// store <12 x i32> %interleaved.vec ; Write 4 tuples of R,G,B
void VPInterleaveRecipe::execute(VPTransformState &State) {
- assert((!needsMaskForGaps() || !State.VF.isScalable()) &&
- "Masking gaps for scalable vectors is not yet supported.");
const InterleaveGroup<Instruction> *Group = getInterleaveGroup();
Instruction *Instr = Group->getInsertPos();
@@ -4657,25 +4655,42 @@ void VPInterleaveRecipe::execute(VPTransformState &State) {
VPValue *Addr = getAddr();
Value *ResAddr = State.get(Addr, VPLane(0));
+ // Compute the mask for gaps in the interleave group, if needed.
+ Value *MaskForGaps = nullptr;
+ if (needsMaskForGaps()) {
+ if (State.VF.isScalable()) {
+ auto *MaskTy = VectorType::get(State.Builder.getInt1Ty(), State.VF);
+ SmallVector<Value *> Ops;
+ for (unsigned I = 0; I < InterleaveFactor; I++)
+ Ops.push_back(Group->getMember(I) ? Constant::getAllOnesValue(MaskTy)
+ : Constant::getNullValue(MaskTy));
+ MaskForGaps =
+ interleaveVectors(State.Builder, Ops, "interleaved.gaps.mask");
+ } else {
+ MaskForGaps =
+ createBitMaskForGaps(State.Builder, State.VF.getFixedValue(), *Group);
+ }
+ assert(MaskForGaps && "Mask for Gaps is required but it is null");
+ }
+
auto CreateGroupMask = [&BlockInMask, &State,
&InterleaveFactor](Value *MaskForGaps) -> Value * {
+ if (!BlockInMask)
+ return MaskForGaps;
+
+ auto *ResBlockInMask = State.get(BlockInMask);
+ Value *ShuffledMask;
if (State.VF.isScalable()) {
- assert(!MaskForGaps && "Interleaved groups with gaps are not supported.");
assert(InterleaveFactor <= 8 &&
"Unsupported deinterleave factor for scalable vectors");
- auto *ResBlockInMask = State.get(BlockInMask);
SmallVector<Value *> Ops(InterleaveFactor, ResBlockInMask);
- return interleaveVectors(State.Builder, Ops, "interleaved.mask");
+ ShuffledMask = interleaveVectors(State.Builder, Ops, "interleaved.mask");
+ } else {
+ ShuffledMask = State.Builder.CreateShuffleVector(
+ ResBlockInMask,
+ createReplicatedMask(InterleaveFactor, State.VF.getFixedValue()),
+ "interleaved.mask");
}
-
- if (!BlockInMask)
- return MaskForGaps;
-
- Value *ResBlockInMask = State.get(BlockInMask);
- Value *ShuffledMask = State.Builder.CreateShuffleVector(
- ResBlockInMask,
- createReplicatedMask(InterleaveFactor, State.VF.getFixedValue()),
- "interleaved.mask");
return MaskForGaps ? State.Builder.CreateBinOp(Instruction::And,
ShuffledMask, MaskForGaps)
: ShuffledMask;
@@ -4684,13 +4699,6 @@ void VPInterleaveRecipe::execute(VPTransformState &State) {
const DataLayout &DL = Instr->getDataLayout();
// Vectorize the interleaved load group.
if (isa<LoadInst>(Instr)) {
- Value *MaskForGaps = nullptr;
- if (needsMaskForGaps()) {
- MaskForGaps =
- createBitMaskForGaps(State.Builder, State.VF.getFixedValue(), *Group);
- assert(MaskForGaps && "Mask for Gaps is required but it is null");
- }
-
Instruction *NewLoad;
if (BlockInMask || MaskForGaps) {
Value *GroupMask = CreateGroupMask(MaskForGaps);
@@ -4758,12 +4766,7 @@ void VPInterleaveRecipe::execute(VPTransformState &State) {
// The sub vector type for current instruction.
auto *SubVT = VectorType::get(ScalarTy, State.VF);
-
// Vectorize the interleaved store group.
- Value *MaskForGaps =
- createBitMaskForGaps(State.Builder, State.VF.getKnownMinValue(), *Group);
- assert(((MaskForGaps != nullptr) == needsMaskForGaps()) &&
- "Mismatch between NeedsMaskForGaps and MaskForGaps");
ArrayRef<VPValue *> StoredValues = getStoredValues();
// Collect the stored vector from each member.
SmallVector<Value *, 4> StoredVecs;
@@ -4843,8 +4846,6 @@ void VPInterleaveRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
void VPInterleaveEVLRecipe::execute(VPTransformState &State) {
assert(State.VF.isScalable() &&
"Only support scalable VF for EVL tail-folding.");
- assert(!needsMaskForGaps() &&
- "Masking gaps for scalable vectors is not yet supported.");
const InterleaveGroup<Instruction> *Group = getInterleaveGroup();
Instruction *Instr = Group->getInsertPos();
@@ -4864,6 +4865,19 @@ void VPInterleaveEVLRecipe::execute(VPTransformState &State) {
/* NUW= */ true, /* NSW= */ true);
LLVMContext &Ctx = State.Builder.getContext();
+ // Compute the mask for gaps in the interleave group, if needed.
+ Value *MaskForGaps = nullptr;
+ if (needsMaskForGaps()) {
+ auto *MaskTy = VectorType::get(State.Builder.getInt1Ty(), State.VF);
+ SmallVector<Value *> Ops;
+ for (unsigned I = 0; I < InterleaveFactor; I++)
+ Ops.push_back(Group->getMember(I) ? Constant::getAllOnesValue(MaskTy)
+ : Constant::getNullValue(MaskTy));
+ MaskForGaps =
+ interleaveVectors(State.Builder, Ops, "interleaved.gaps.mask");
+ assert(MaskForGaps && "Mask for Gaps is required but it is null");
+ }
+
Value *GroupMask = nullptr;
if (VPValue *BlockInMask = getMask()) {
SmallVector<Value *> Ops(InterleaveFactor, State.get(BlockInMask));
@@ -4872,6 +4886,8 @@ void VPInterleaveEVLRecipe::execute(VPTransformState &State) {
GroupMask =
State.Builder.CreateVectorSplat(WideVF, State.Builder.getTrue());
}
+ if (MaskForGaps)
+ GroupMask = State.Builder.CreateAnd(GroupMask, MaskForGaps);
// Vectorize the interleaved load group.
if (isa<LoadInst>(Instr)) {
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll b/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll
index 667612a193dd5..0f4cbdba42e7a 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll
@@ -62,50 +62,22 @@ exit:
define i1 @scalarize_ptr_induction(ptr %start, ptr %end, ptr noalias %dst, i1 %c) #1 {
; CHECK-LABEL: define i1 @scalarize_ptr_induction(
; CHECK-SAME: ptr [[START:%.*]], ptr [[END:%.*]], ptr noalias [[DST:%.*]], i1 [[C:%.*]]) #[[ATTR1:[0-9]+]] {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[END1:%.*]] = ptrtoaddr ptr [[END]] to i64
-; CHECK-NEXT: [[TMP4:%.*]] = ptrtoaddr ptr [[START]] to i64
-; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[END1]], -12
-; CHECK-NEXT: [[TMP1:%.*]] = sub i64 [[TMP5]], [[TMP4]]
-; CHECK-NEXT: [[TMP2:%.*]] = udiv i64 [[TMP1]], 12
-; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[TMP2]], 1
-; CHECK-NEXT: br label %[[VECTOR_MEMCHECK:.*]]
-; CHECK: [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT: [[SCEVGEP6:%.*]] = getelementptr i8, ptr [[START]], i64 4
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x ptr> poison, ptr [[DST]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x ptr> [[BROADCAST_SPLATINSERT]], <vscale x 2 x ptr> poison, <vscale x 2 x i32> zeroinitializer
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT6:%.*]] = insertelement <vscale x 2 x ptr> poison, ptr [[END]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT7:%.*]] = shufflevector <vscale x 2 x ptr> [[BROADCAST_SPLATINSERT6]], <vscale x 2 x ptr> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[VECTOR_MEMCHECK:.*]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_MEMCHECK]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[POINTER_PHI:%.*]] = phi ptr [ [[START]], %[[VECTOR_MEMCHECK]] ], [ [[PTR_IND:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[AVL:%.*]] = phi i64 [ [[TMP3]], %[[VECTOR_MEMCHECK]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP13:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
-; CHECK-NEXT: [[TMP14:%.*]] = mul <vscale x 2 x i64> [[TMP13]], splat (i64 12)
-; CHECK-NEXT: [[VECTOR_GEP:%.*]] = getelementptr i8, ptr [[POINTER_PHI]], <vscale x 2 x i64> [[TMP14]]
-; CHECK-NEXT: [[TMP11:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 2, i1 true)
+; CHECK-NEXT: [[PTR_IV:%.*]] = phi ptr [ [[START]], %[[VECTOR_MEMCHECK]] ], [ [[PTR_IV_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr i8, ptr [[PTR_IV]], i64 4
+; CHECK-NEXT: [[TMP11:%.*]] = load i32, ptr [[GEP]], align 4
; CHECK-NEXT: [[TMP26:%.*]] = zext i32 [[TMP11]] to i64
-; CHECK-NEXT: [[TMP12:%.*]] = mul i64 [[INDEX]], 12
-; CHECK-NEXT: [[TMP15:%.*]] = getelementptr i8, ptr [[SCEVGEP6]], i64 [[TMP12]]
-; CHECK-NEXT: [[TMP18:%.*]] = call <vscale x 2 x i32> @llvm.experimental.vp.strided.load.nxv2i32.p0.i64(ptr align 4 [[TMP15]], i64 12, <vscale x 2 x i1> splat (i1 true), i32 [[TMP11]])
-; CHECK-NEXT: [[TMP19:%.*]] = zext <vscale x 2 x i32> [[TMP18]] to <vscale x 2 x i64>
-; CHECK-NEXT: [[TMP20:%.*]] = mul <vscale x 2 x i64> [[TMP19]], splat (i64 -7070675565921424023)
-; CHECK-NEXT: [[TMP21:%.*]] = add <vscale x 2 x i64> [[TMP20]], splat (i64 -4)
-; CHECK-NEXT: call void @llvm.vp.scatter.nxv2i64.nxv2p0(<vscale x 2 x i64> [[TMP21]], <vscale x 2 x ptr> align 1 [[BROADCAST_SPLAT]], <vscale x 2 x i1> splat (i1 true), i32 [[TMP11]])
-; CHECK-NEXT: [[CURRENT_ITERATION_NEXT]] = add i64 [[TMP26]], [[INDEX]]
-; CHECK-NEXT: [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP26]]
-; CHECK-NEXT: [[TMP27:%.*]] = mul i64 12, [[TMP26]]
-; CHECK-NEXT: [[PTR_IND]] = getelementptr i8, ptr [[POINTER_PHI]], i64 [[TMP27]]
-; CHECK-NEXT: [[TMP28:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
-; CHECK-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
-; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: [[TMP30:%.*]] = getelementptr nusw i8, <vscale x 2 x ptr> [[VECTOR_GEP]], i64 12
-; CHECK-NEXT: [[TMP17:%.*]] = icmp eq <vscale x 2 x ptr> [[TMP30]], [[BROADCAST_SPLAT7]]
-; CHECK-NEXT: [[TMP29:%.*]] = sub i64 [[TMP26]], 1
-; CHECK-NEXT: [[TMP25:%.*]] = extractelement <vscale x 2 x i1> [[TMP17]], i64 [[TMP29]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK-NEXT: [[UNUSED:%.*]] = load i32, ptr [[PTR_IV]], align 4
+; CHECK-NEXT: [[MUL1:%.*]] = mul i64 [[TMP26]], -7070675565921424023
+; CHECK-NEXT: [[MUL2:%.*]] = add i64 [[MUL1]], -4
+; CHECK-NEXT: store i64 [[MUL2]], ptr [[DST]], align 1
+; CHECK-NEXT: [[PTR_IV_NEXT]] = getelementptr nusw i8, ptr [[PTR_IV]], i64 12
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq ptr [[PTR_IV_NEXT]], [[END]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP:.*]], label %[[VECTOR_BODY]]
; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[TMP25:%.*]] = phi i1 [ [[CMP]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: ret i1 [[TMP25]]
;
entry:
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-interleave.ll b/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-interleave.ll
index a6d11e2201644..5de71c3da42ec 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-interleave.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-interleave.ll
@@ -348,23 +348,23 @@ define i32 @load_factor_4_with_tail_gap(i64 %n, ptr noalias %a) vscale_range(2,
; IF-EVL-NEXT: entry:
; IF-EVL-NEXT: br label [[VECTOR_PH:%.*]]
; IF-EVL: vector.ph:
-; IF-EVL-NEXT: [[TMP0:%.*]] = getelementptr nuw i8, ptr [[A:%.*]], i64 4
-; IF-EVL-NEXT: [[TMP1:%.*]] = getelementptr nuw i8, ptr [[A]], i64 8
; IF-EVL-NEXT: br label [[VECTOR_BODY:%.*]]
; IF-EVL: vector.body:
; IF-EVL-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], [[VECTOR_BODY]] ]
; IF-EVL-NEXT: [[VEC_PHI:%.*]] = phi <vscale x 4 x i32> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP13:%.*]], [[VECTOR_BODY]] ]
; IF-EVL-NEXT: [[AVL:%.*]] = phi i64 [ [[N:%.*]], [[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], [[VECTOR_BODY]] ]
; IF-EVL-NEXT: [[TMP2:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 4, i1 true)
-; IF-EVL-NEXT: [[TMP3:%.*]] = shl nuw i64 [[INDEX]], 4
-; IF-EVL-NEXT: [[TMP4:%.*]] = getelementptr nuw i8, ptr [[A]], i64 [[TMP3]]
-; IF-EVL-NEXT: [[TMP5:%.*]] = call <vscale x 4 x i32> @llvm.experimental.vp.strided.load.nxv4i32.p0.i64(ptr align 4 [[TMP4]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP2]])
+; IF-EVL-NEXT: [[TMP1:%.*]] = getelementptr inbounds [4 x i32], ptr [[A:%.*]], i64 [[INDEX]], i32 0
+; IF-EVL-NEXT: [[INTERLEAVE_EVL:%.*]] = mul nuw nsw i32 [[TMP2]], 4
+; IF-EVL-NEXT: [[INTERLEAVED_GAPS_MASK:%.*]] = call <vscale x 16 x i1> @llvm.vector.interleave4.nxv16i1(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> zeroinitializer)
+; IF-EVL-NEXT: [[TMP3:%.*]] = and <vscale x 16 x i1> splat (i1 true), [[INTERLEAVED_GAPS_MASK]]
+; IF-EVL-NEXT: [[WIDE_VP_LOAD:%.*]] = call <vscale x 16 x i32> @llvm.vp.load.nxv16i32.p0(ptr align 4 [[TMP1]], <vscale x 16 x i1> [[TMP3]], i32 [[INTERLEAVE_EVL]])
+; IF-EVL-NEXT: [[STRIDED_VEC:%.*]] = call { <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32> } @llvm.vector.deinterleave4.nxv16i32(<vscale x 16 x i32> [[WIDE_VP_LOAD]])
+; IF-EVL-NEXT: [[TMP5:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 0
+; IF-EVL-NEXT: [[TMP8:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 1
+; IF-EVL-NEXT: [[TMP11:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 2
; IF-EVL-NEXT: [[TMP6:%.*]] = add <vscale x 4 x i32> [[VEC_PHI]], [[TMP5]]
-; IF-EVL-NEXT: [[TMP7:%.*]] = getelementptr nuw i8, ptr [[TMP0]], i64 [[TMP3]]
-; IF-EVL-NEXT: [[TMP8:%.*]] = call <vscale x 4 x i32> @llvm.experimental.vp.strided.load.nxv4i32.p0.i64(ptr align 4 [[TMP7]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP2]])
; IF-EVL-NEXT: [[TMP9:%.*]] = add <vscale x 4 x i32> [[TMP6]], [[TMP8]]
-; IF-EVL-NEXT: [[TMP10:%.*]] = getelementptr nuw i8, ptr [[TMP1]], i64 [[TMP3]]
-; IF-EVL-NEXT: [[TMP11:%.*]] = call <vscale x 4 x i32> @llvm.experimental.vp.strided.load.nxv4i32.p0.i64(ptr align 4 [[TMP10]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP2]])
; IF-EVL-NEXT: [[TMP12:%.*]] = add <vscale x 4 x i32> [[TMP9]], [[TMP11]]
; IF-EVL-NEXT: [[TMP13]] = call <vscale x 4 x i32> @llvm.vp.merge.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> [[TMP12]], <vscale x 4 x i32> [[VEC_PHI]], i32 [[TMP2]])
; IF-EVL-NEXT: [[TMP14:%.*]] = zext i32 [[TMP2]] to i64
@@ -470,8 +470,6 @@ define void @store_factor_4_with_tail_gap(i32 %n, ptr noalias %a) vscale_range(2
; IF-EVL-NEXT: entry:
; IF-EVL-NEXT: br label [[VECTOR_PH:%.*]]
; IF-EVL: vector.ph:
-; IF-EVL-NEXT: [[TMP0:%.*]] = getelementptr nuw i8, ptr [[A:%.*]], i64 4
-; IF-EVL-NEXT: [[TMP1:%.*]] = getelementptr nuw i8, ptr [[A]], i64 8
; IF-EVL-NEXT: [[TMP2:%.*]] = call <vscale x 4 x i32> @llvm.stepvector.nxv4i32()
; IF-EVL-NEXT: br label [[VECTOR_BODY:%.*]]
; IF-EVL: vector.body:
@@ -481,14 +479,12 @@ define void @store_factor_4_with_tail_gap(i32 %n, ptr noalias %a) vscale_range(2
; IF-EVL-NEXT: [[TMP3:%.*]] = call i32 @llvm.experimental.get.vector.length.i32(i32 [[AVL]], i32 4, i1 true)
; IF-EVL-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 4 x i32> poison, i32 [[TMP3]], i64 0
; IF-EVL-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 4 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 4 x i32> poison, <vscale x 4 x i32> zeroinitializer
-; IF-EVL-NEXT: [[TMP4:%.*]] = zext i32 [[INDEX]] to i64
-; IF-EVL-NEXT: [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 4
-; IF-EVL-NEXT: [[TMP6:%.*]] = getelementptr nuw i8, ptr [[A]], i64 [[TMP5]]
-; IF-EVL-NEXT: call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP6]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP3]])
-; IF-EVL-NEXT: [[TMP7:%.*]] = getelementptr nuw i8, ptr [[TMP0]], i64 [[TMP5]]
-; IF-EVL-NEXT: call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP7]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP3]])
-; IF-EVL-NEXT: [[TMP8:%.*]] = getelementptr nuw i8, ptr [[TMP1]], i64 [[TMP5]]
-; IF-EVL-NEXT: call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP8]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP3]])
+; IF-EVL-NEXT: [[TMP4:%.*]] = getelementptr inbounds [4 x i32], ptr [[A:%.*]], i32 [[INDEX]], i32 0
+; IF-EVL-NEXT: [[INTERLEAVE_EVL:%.*]] = mul nuw nsw i32 [[TMP3]], 4
+; IF-EVL-NEXT: [[INTERLEAVED_GAPS_MASK:%.*]] = call <vscale x 16 x i1> @llvm.vector.interleave4.nxv16i1(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> zeroinitializer)
+; IF-EVL-NEXT: [[TMP5:%.*]] = and <vscale x 16 x i1> splat (i1 true), [[INTERLEAVED_GAPS_MASK]]
+; IF-EVL-NEXT: [[INTERLEAVED_VEC:%.*]] = call <vscale x 16 x i32> @llvm.vector.interleave4.nxv16i32(<vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> poison)
+; IF-EVL-NEXT: call void @llvm.vp.store.nxv16i32.p0(<vscale x 16 x i32> [[INTERLEAVED_VEC]], ptr align 4 [[TMP4]], <vscale x 16 x i1> [[TMP5]], i32 [[INTERLEAVE_EVL]])
; IF-EVL-NEXT: [[CURRENT_ITERATION_NEXT]] = add i32 [[TMP3]], [[INDEX]]
; IF-EVL-NEXT: [[AVL_NEXT]] = sub nuw i32 [[AVL]], [[TMP3]]
; IF-EVL-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <vscale x 4 x i32> [[VEC_IND]], [[BROADCAST_SPLAT]]
@@ -508,8 +504,6 @@ define void @store_factor_4_with_tail_gap(i32 %n, ptr noalias %a) vscale_range(2
; NO-VP: vector.ph:
; NO-VP-NEXT: [[N_MOD_VF:%.*]] = urem i32 [[N]], [[TMP1]]
; NO-VP-NEXT: [[N_VEC:%.*]] = sub i32 [[N]], [[N_MOD_VF]]
-; NO-VP-NEXT: [[TMP4:%.*]] = getelementptr nuw i8, ptr [[A:%.*]], i64 4
-; NO-VP-NEXT: [[TMP5:%.*]] = getelementptr nuw i8, ptr [[A]], i64 8
; NO-VP-NEXT: [[TMP6:%.*]] = call <vscale x 4 x i32> @llvm.stepvector.nxv4i32()
; NO-VP-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 4 x i32> poison, i32 [[TMP1]], i64 0
; NO-VP-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 4 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 4 x i32> poison, <vscale x 4 x i32> zeroinitializer
@@ -517,14 +511,10 @@ define void @store_factor_4_with_tail_gap(i32 %n, ptr noalias %a) vscale_range(2
; NO-VP: vector.body:
; NO-VP-NEXT: [[INDEX:%.*]] = phi i32 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
; NO-VP-NEXT: [[VEC_IND:%.*]] = phi <vscale x 4 x i32> [ [[TMP6]], [[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], [[VECTOR_BODY]] ]
-; NO-VP-NEXT: [[TMP7:%.*]] = zext i32 [[INDEX]] to i64
-; NO-VP-NEXT: [[TMP8:%.*]] = shl nuw i64 [[TMP7]], 4
-; NO-VP-NEXT: [[TMP9:%.*]] = getelementptr nuw i8, ptr [[A]], i64 [[TMP8]]
-; NO-VP-NEXT: call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP9]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP1]])
-; NO-VP-NEXT: [[TMP10:%.*]] = getelementptr nuw i8, ptr [[TMP4]], i64 [[TMP8]]
-; NO-VP-NEXT: call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP10]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP1]])
-; NO-VP-NEXT: [[TMP11:%.*]] = getelementptr nuw i8, ptr [[TMP5]], i64 [[TMP8]]
-; NO-VP-NEXT: call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP11]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP1]])
+; NO-VP-NEXT: [[TMP3:%.*]] = getelementptr inbounds [4 x i32], ptr [[A:%.*]], i32 [[INDEX]], i32 0
+; NO-VP-NEXT: [[INTERLEAVED_GAPS_MASK:%.*]] = call <vscale x 16 x i1> @llvm.vector.interleave4.nxv16i1(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> zeroinitializer)
+; NO-VP-NEXT: [[INTERLEAVED_VEC:%.*]] = call <vscale x 16 x i32> @llvm.vector.interleave4.nxv16i32(<vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> poison)
+; NO-VP-NEXT: call void @llvm.masked.store.nxv16i32.p0(<vscale x 16 x i32> [[INTERLEAVED_VEC]], ptr align 4 [[TMP3]], <vscale x 16 x i1> [[INTERLEAVED_GAPS_MASK]])
; NO-VP-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], [[TMP1]]
; NO-VP-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <vscale x 4 x i32> [[VEC_IND]], [[BROADCAST_SPLAT]]
; NO-VP-NEXT: [[TMP12:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
More information about the llvm-commits
mailing list