[llvm] [LV] Support masked interleaved access with gaps for scalable vector (PR #196479)

Mel Chen via llvm-commits llvm-commits at lists.llvm.org
Thu Oct 1 06:46:07 PDT 2026


https://github.com/Mel-Chen updated https://github.com/llvm/llvm-project/pull/196479

>From b9d50336917287166ca13a96fe107851671d11b8 Mon Sep 17 00:00:00 2001
From: Mel Chen <mel.chen at sifive.com>
Date: Thu, 7 May 2026 23:20:16 -0700
Subject: [PATCH] Initial patch for scalable interleave access with gap mask

---
 .../Target/RISCV/RISCVTargetTransformInfo.cpp | 27 ++++---
 .../Transforms/Vectorize/LoopVectorize.cpp    |  6 --
 llvm/lib/Transforms/Vectorize/VPlan.h         |  2 -
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp | 72 +++++++++++--------
 .../LoopVectorize/RISCV/pointer-induction.ll  | 52 ++++----------
 .../RISCV/tail-folding-interleave.ll          | 48 +++++--------
 6 files changed, 88 insertions(+), 119 deletions(-)

diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index 69554ca2b8155..0a16811a5581f 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -1214,13 +1214,12 @@ InstructionCost RISCVTTIImpl::getInterleavedMemoryOpCost(
     unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
     Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
     bool UseMaskForCond, bool UseMaskForGaps) const {
-
+  auto *VTy = cast<VectorType>(VecTy);
   // The interleaved memory access pass will lower (de)interleave ops combined
   // with an adjacent appropriate memory to vlseg/vsseg intrinsics. vlseg/vsseg
   // only support masking per-iteration (i.e. condition), not per-segment (i.e.
   // gap).
   if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
-    auto *VTy = cast<VectorType>(VecTy);
     std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(VTy);
     // Need to make sure type has't been scalarized
     if (LT.second.isVector()) {
@@ -1256,12 +1255,6 @@ InstructionCost RISCVTTIImpl::getInterleavedMemoryOpCost(
     }
   }
 
-  // TODO: Return the cost of interleaved accesses for scalable vector when
-  // unable to convert to segment accesses instructions.
-  if (isa<ScalableVectorType>(VecTy))
-    return InstructionCost::getInvalid();
-
-  auto *FVTy = cast<FixedVectorType>(VecTy);
   // When gaps are only at the tail, for interleaved load, we can emit a wide
   // masked load and shufflevectors. For interleaved store, we can emit
   // shufflevectors and a wide masked store. The interleaved memory access pass
@@ -1274,22 +1267,28 @@ InstructionCost RISCVTTIImpl::getInterleavedMemoryOpCost(
     bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.back() + 1);
     if (IsTailGapOnly &&
         NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
-      std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(FVTy);
+      std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(VTy);
       if (LT.second.isVector() &&
-          FVTy->getElementCount().isKnownMultipleOf(Factor)) {
-        auto *SubVecTy = VectorType::get(
-            FVTy->getElementType(),
-            FVTy->getElementCount().divideCoefficientBy(Factor));
+          VTy->getElementCount().isKnownMultipleOf(Factor)) {
+        auto *SubVecTy =
+            VectorType::get(VTy->getElementType(),
+                            VTy->getElementCount().divideCoefficientBy(Factor));
         if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
                                               AddressSpace, DL)) {
           // The cost is proportional to the total number of element accesses.
-          unsigned NumAccesses = getEstimatedVLFor(FVTy);
+          unsigned NumAccesses = getEstimatedVLFor(VTy);
           return NumAccesses * TTI::TCC_Basic;
         }
       }
     }
   }
 
+  // TODO: Return the cost of interleaved accesses for scalable vector when
+  // unable to convert to segment accesses instructions.
+  if (isa<ScalableVectorType>(VecTy))
+    return InstructionCost::getInvalid();
+
+  auto *FVTy = cast<FixedVectorType>(VecTy);
   InstructionCost MemCost =
       getMemoryOpCost(Opcode, VecTy, Alignment, AddressSpace, CostKind);
   unsigned VF = FVTy->getNumElements() / Factor;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 0986d8ea9e59d..dfed3d6c567e9 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -2630,12 +2630,6 @@ bool LoopVectorizationCostModel::interleavedAccessCanBeWidened(
   if (Group->isReverse())
     return false;
 
-  // TODO: Support interleaved access that requires a gap mask for scalable VFs.
-  bool NeedsMaskForGaps = LoadAccessWithGapsRequiresEpilogMasking ||
-                          StoreAccessWithGapsRequiresMasking;
-  if (VF.isScalable() && NeedsMaskForGaps)
-    return false;
-
   return isLegalMaskedLoadOrStore(I, VF);
 }
 
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index aa93dbaa65170..066ee6a478560 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -3178,8 +3178,6 @@ class LLVM_ABI_FOR_TEST VPInterleaveEVLRecipe final : public VPInterleaveBase {
                          R.getDebugLoc()) {
     assert(!getInterleaveGroup()->isReverse() &&
            "Reversed interleave-group with tail folding is not supported.");
-    assert(!needsMaskForGaps() && "Interleaved access with gap mask is not "
-                                  "supported for scalable vector.");
   }
 
   ~VPInterleaveEVLRecipe() override = default;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 5c31590283dbb..8b6f3cd44ab59 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -4643,8 +4643,6 @@ static Value *interleaveVectors(IRBuilderBase &Builder, ArrayRef<Value *> Vals,
 //        <0, 4, 8, 1, 5, 9, 2, 6, 10, 3, 7, 11>    ; Interleave R,G,B elements
 //   store <12 x i32> %interleaved.vec              ; Write 4 tuples of R,G,B
 void VPInterleaveRecipe::execute(VPTransformState &State) {
-  assert((!needsMaskForGaps() || !State.VF.isScalable()) &&
-         "Masking gaps for scalable vectors is not yet supported.");
   const InterleaveGroup<Instruction> *Group = getInterleaveGroup();
   Instruction *Instr = Group->getInsertPos();
 
@@ -4657,25 +4655,42 @@ void VPInterleaveRecipe::execute(VPTransformState &State) {
   VPValue *Addr = getAddr();
   Value *ResAddr = State.get(Addr, VPLane(0));
 
+  // Compute the mask for gaps in the interleave group, if needed.
+  Value *MaskForGaps = nullptr;
+  if (needsMaskForGaps()) {
+    if (State.VF.isScalable()) {
+      auto *MaskTy = VectorType::get(State.Builder.getInt1Ty(), State.VF);
+      SmallVector<Value *> Ops;
+      for (unsigned I = 0; I < InterleaveFactor; I++)
+        Ops.push_back(Group->getMember(I) ? Constant::getAllOnesValue(MaskTy)
+                                          : Constant::getNullValue(MaskTy));
+      MaskForGaps =
+          interleaveVectors(State.Builder, Ops, "interleaved.gaps.mask");
+    } else {
+      MaskForGaps =
+          createBitMaskForGaps(State.Builder, State.VF.getFixedValue(), *Group);
+    }
+    assert(MaskForGaps && "Mask for Gaps is required but it is null");
+  }
+
   auto CreateGroupMask = [&BlockInMask, &State,
                           &InterleaveFactor](Value *MaskForGaps) -> Value * {
+    if (!BlockInMask)
+      return MaskForGaps;
+
+    auto *ResBlockInMask = State.get(BlockInMask);
+    Value *ShuffledMask;
     if (State.VF.isScalable()) {
-      assert(!MaskForGaps && "Interleaved groups with gaps are not supported.");
       assert(InterleaveFactor <= 8 &&
              "Unsupported deinterleave factor for scalable vectors");
-      auto *ResBlockInMask = State.get(BlockInMask);
       SmallVector<Value *> Ops(InterleaveFactor, ResBlockInMask);
-      return interleaveVectors(State.Builder, Ops, "interleaved.mask");
+      ShuffledMask = interleaveVectors(State.Builder, Ops, "interleaved.mask");
+    } else {
+      ShuffledMask = State.Builder.CreateShuffleVector(
+          ResBlockInMask,
+          createReplicatedMask(InterleaveFactor, State.VF.getFixedValue()),
+          "interleaved.mask");
     }
-
-    if (!BlockInMask)
-      return MaskForGaps;
-
-    Value *ResBlockInMask = State.get(BlockInMask);
-    Value *ShuffledMask = State.Builder.CreateShuffleVector(
-        ResBlockInMask,
-        createReplicatedMask(InterleaveFactor, State.VF.getFixedValue()),
-        "interleaved.mask");
     return MaskForGaps ? State.Builder.CreateBinOp(Instruction::And,
                                                    ShuffledMask, MaskForGaps)
                        : ShuffledMask;
@@ -4684,13 +4699,6 @@ void VPInterleaveRecipe::execute(VPTransformState &State) {
   const DataLayout &DL = Instr->getDataLayout();
   // Vectorize the interleaved load group.
   if (isa<LoadInst>(Instr)) {
-    Value *MaskForGaps = nullptr;
-    if (needsMaskForGaps()) {
-      MaskForGaps =
-          createBitMaskForGaps(State.Builder, State.VF.getFixedValue(), *Group);
-      assert(MaskForGaps && "Mask for Gaps is required but it is null");
-    }
-
     Instruction *NewLoad;
     if (BlockInMask || MaskForGaps) {
       Value *GroupMask = CreateGroupMask(MaskForGaps);
@@ -4758,12 +4766,7 @@ void VPInterleaveRecipe::execute(VPTransformState &State) {
 
   // The sub vector type for current instruction.
   auto *SubVT = VectorType::get(ScalarTy, State.VF);
-
   // Vectorize the interleaved store group.
-  Value *MaskForGaps =
-      createBitMaskForGaps(State.Builder, State.VF.getKnownMinValue(), *Group);
-  assert(((MaskForGaps != nullptr) == needsMaskForGaps()) &&
-         "Mismatch between NeedsMaskForGaps and MaskForGaps");
   ArrayRef<VPValue *> StoredValues = getStoredValues();
   // Collect the stored vector from each member.
   SmallVector<Value *, 4> StoredVecs;
@@ -4843,8 +4846,6 @@ void VPInterleaveRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
 void VPInterleaveEVLRecipe::execute(VPTransformState &State) {
   assert(State.VF.isScalable() &&
          "Only support scalable VF for EVL tail-folding.");
-  assert(!needsMaskForGaps() &&
-         "Masking gaps for scalable vectors is not yet supported.");
   const InterleaveGroup<Instruction> *Group = getInterleaveGroup();
   Instruction *Instr = Group->getInsertPos();
 
@@ -4864,6 +4865,19 @@ void VPInterleaveEVLRecipe::execute(VPTransformState &State) {
       /* NUW= */ true, /* NSW= */ true);
   LLVMContext &Ctx = State.Builder.getContext();
 
+  // Compute the mask for gaps in the interleave group, if needed.
+  Value *MaskForGaps = nullptr;
+  if (needsMaskForGaps()) {
+    auto *MaskTy = VectorType::get(State.Builder.getInt1Ty(), State.VF);
+    SmallVector<Value *> Ops;
+    for (unsigned I = 0; I < InterleaveFactor; I++)
+      Ops.push_back(Group->getMember(I) ? Constant::getAllOnesValue(MaskTy)
+                                        : Constant::getNullValue(MaskTy));
+    MaskForGaps =
+        interleaveVectors(State.Builder, Ops, "interleaved.gaps.mask");
+    assert(MaskForGaps && "Mask for Gaps is required but it is null");
+  }
+
   Value *GroupMask = nullptr;
   if (VPValue *BlockInMask = getMask()) {
     SmallVector<Value *> Ops(InterleaveFactor, State.get(BlockInMask));
@@ -4872,6 +4886,8 @@ void VPInterleaveEVLRecipe::execute(VPTransformState &State) {
     GroupMask =
         State.Builder.CreateVectorSplat(WideVF, State.Builder.getTrue());
   }
+  if (MaskForGaps)
+    GroupMask = State.Builder.CreateAnd(GroupMask, MaskForGaps);
 
   // Vectorize the interleaved load group.
   if (isa<LoadInst>(Instr)) {
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll b/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll
index 667612a193dd5..0f4cbdba42e7a 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/pointer-induction.ll
@@ -62,50 +62,22 @@ exit:
 define i1 @scalarize_ptr_induction(ptr %start, ptr %end, ptr noalias %dst, i1 %c) #1 {
 ; CHECK-LABEL: define i1 @scalarize_ptr_induction(
 ; CHECK-SAME: ptr [[START:%.*]], ptr [[END:%.*]], ptr noalias [[DST:%.*]], i1 [[C:%.*]]) #[[ATTR1:[0-9]+]] {
-; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[END1:%.*]] = ptrtoaddr ptr [[END]] to i64
-; CHECK-NEXT:    [[TMP4:%.*]] = ptrtoaddr ptr [[START]] to i64
-; CHECK-NEXT:    [[TMP5:%.*]] = add i64 [[END1]], -12
-; CHECK-NEXT:    [[TMP1:%.*]] = sub i64 [[TMP5]], [[TMP4]]
-; CHECK-NEXT:    [[TMP2:%.*]] = udiv i64 [[TMP1]], 12
-; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[TMP2]], 1
-; CHECK-NEXT:    br label %[[VECTOR_MEMCHECK:.*]]
-; CHECK:       [[VECTOR_MEMCHECK]]:
-; CHECK-NEXT:    [[SCEVGEP6:%.*]] = getelementptr i8, ptr [[START]], i64 4
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x ptr> poison, ptr [[DST]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x ptr> [[BROADCAST_SPLATINSERT]], <vscale x 2 x ptr> poison, <vscale x 2 x i32> zeroinitializer
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT6:%.*]] = insertelement <vscale x 2 x ptr> poison, ptr [[END]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT7:%.*]] = shufflevector <vscale x 2 x ptr> [[BROADCAST_SPLATINSERT6]], <vscale x 2 x ptr> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:  [[VECTOR_MEMCHECK:.*]]:
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_MEMCHECK]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[POINTER_PHI:%.*]] = phi ptr [ [[START]], %[[VECTOR_MEMCHECK]] ], [ [[PTR_IND:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[AVL:%.*]] = phi i64 [ [[TMP3]], %[[VECTOR_MEMCHECK]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP13:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
-; CHECK-NEXT:    [[TMP14:%.*]] = mul <vscale x 2 x i64> [[TMP13]], splat (i64 12)
-; CHECK-NEXT:    [[VECTOR_GEP:%.*]] = getelementptr i8, ptr [[POINTER_PHI]], <vscale x 2 x i64> [[TMP14]]
-; CHECK-NEXT:    [[TMP11:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 2, i1 true)
+; CHECK-NEXT:    [[PTR_IV:%.*]] = phi ptr [ [[START]], %[[VECTOR_MEMCHECK]] ], [ [[PTR_IV_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr i8, ptr [[PTR_IV]], i64 4
+; CHECK-NEXT:    [[TMP11:%.*]] = load i32, ptr [[GEP]], align 4
 ; CHECK-NEXT:    [[TMP26:%.*]] = zext i32 [[TMP11]] to i64
-; CHECK-NEXT:    [[TMP12:%.*]] = mul i64 [[INDEX]], 12
-; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr i8, ptr [[SCEVGEP6]], i64 [[TMP12]]
-; CHECK-NEXT:    [[TMP18:%.*]] = call <vscale x 2 x i32> @llvm.experimental.vp.strided.load.nxv2i32.p0.i64(ptr align 4 [[TMP15]], i64 12, <vscale x 2 x i1> splat (i1 true), i32 [[TMP11]])
-; CHECK-NEXT:    [[TMP19:%.*]] = zext <vscale x 2 x i32> [[TMP18]] to <vscale x 2 x i64>
-; CHECK-NEXT:    [[TMP20:%.*]] = mul <vscale x 2 x i64> [[TMP19]], splat (i64 -7070675565921424023)
-; CHECK-NEXT:    [[TMP21:%.*]] = add <vscale x 2 x i64> [[TMP20]], splat (i64 -4)
-; CHECK-NEXT:    call void @llvm.vp.scatter.nxv2i64.nxv2p0(<vscale x 2 x i64> [[TMP21]], <vscale x 2 x ptr> align 1 [[BROADCAST_SPLAT]], <vscale x 2 x i1> splat (i1 true), i32 [[TMP11]])
-; CHECK-NEXT:    [[CURRENT_ITERATION_NEXT]] = add i64 [[TMP26]], [[INDEX]]
-; CHECK-NEXT:    [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP26]]
-; CHECK-NEXT:    [[TMP27:%.*]] = mul i64 12, [[TMP26]]
-; CHECK-NEXT:    [[PTR_IND]] = getelementptr i8, ptr [[POINTER_PHI]], i64 [[TMP27]]
-; CHECK-NEXT:    [[TMP28:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
-; CHECK-NEXT:    br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[TMP30:%.*]] = getelementptr nusw i8, <vscale x 2 x ptr> [[VECTOR_GEP]], i64 12
-; CHECK-NEXT:    [[TMP17:%.*]] = icmp eq <vscale x 2 x ptr> [[TMP30]], [[BROADCAST_SPLAT7]]
-; CHECK-NEXT:    [[TMP29:%.*]] = sub i64 [[TMP26]], 1
-; CHECK-NEXT:    [[TMP25:%.*]] = extractelement <vscale x 2 x i1> [[TMP17]], i64 [[TMP29]]
-; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK-NEXT:    [[UNUSED:%.*]] = load i32, ptr [[PTR_IV]], align 4
+; CHECK-NEXT:    [[MUL1:%.*]] = mul i64 [[TMP26]], -7070675565921424023
+; CHECK-NEXT:    [[MUL2:%.*]] = add i64 [[MUL1]], -4
+; CHECK-NEXT:    store i64 [[MUL2]], ptr [[DST]], align 1
+; CHECK-NEXT:    [[PTR_IV_NEXT]] = getelementptr nusw i8, ptr [[PTR_IV]], i64 12
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq ptr [[PTR_IV_NEXT]], [[END]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[TMP25:%.*]] = phi i1 [ [[CMP]], %[[VECTOR_BODY]] ]
 ; CHECK-NEXT:    ret i1 [[TMP25]]
 ;
 entry:
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-interleave.ll b/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-interleave.ll
index a6d11e2201644..5de71c3da42ec 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-interleave.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-interleave.ll
@@ -348,23 +348,23 @@ define i32 @load_factor_4_with_tail_gap(i64 %n, ptr noalias %a) vscale_range(2,
 ; IF-EVL-NEXT:  entry:
 ; IF-EVL-NEXT:    br label [[VECTOR_PH:%.*]]
 ; IF-EVL:       vector.ph:
-; IF-EVL-NEXT:    [[TMP0:%.*]] = getelementptr nuw i8, ptr [[A:%.*]], i64 4
-; IF-EVL-NEXT:    [[TMP1:%.*]] = getelementptr nuw i8, ptr [[A]], i64 8
 ; IF-EVL-NEXT:    br label [[VECTOR_BODY:%.*]]
 ; IF-EVL:       vector.body:
 ; IF-EVL-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], [[VECTOR_BODY]] ]
 ; IF-EVL-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 4 x i32> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP13:%.*]], [[VECTOR_BODY]] ]
 ; IF-EVL-NEXT:    [[AVL:%.*]] = phi i64 [ [[N:%.*]], [[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], [[VECTOR_BODY]] ]
 ; IF-EVL-NEXT:    [[TMP2:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 4, i1 true)
-; IF-EVL-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[INDEX]], 4
-; IF-EVL-NEXT:    [[TMP4:%.*]] = getelementptr nuw i8, ptr [[A]], i64 [[TMP3]]
-; IF-EVL-NEXT:    [[TMP5:%.*]] = call <vscale x 4 x i32> @llvm.experimental.vp.strided.load.nxv4i32.p0.i64(ptr align 4 [[TMP4]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP2]])
+; IF-EVL-NEXT:    [[TMP1:%.*]] = getelementptr inbounds [4 x i32], ptr [[A:%.*]], i64 [[INDEX]], i32 0
+; IF-EVL-NEXT:    [[INTERLEAVE_EVL:%.*]] = mul nuw nsw i32 [[TMP2]], 4
+; IF-EVL-NEXT:    [[INTERLEAVED_GAPS_MASK:%.*]] = call <vscale x 16 x i1> @llvm.vector.interleave4.nxv16i1(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> zeroinitializer)
+; IF-EVL-NEXT:    [[TMP3:%.*]] = and <vscale x 16 x i1> splat (i1 true), [[INTERLEAVED_GAPS_MASK]]
+; IF-EVL-NEXT:    [[WIDE_VP_LOAD:%.*]] = call <vscale x 16 x i32> @llvm.vp.load.nxv16i32.p0(ptr align 4 [[TMP1]], <vscale x 16 x i1> [[TMP3]], i32 [[INTERLEAVE_EVL]])
+; IF-EVL-NEXT:    [[STRIDED_VEC:%.*]] = call { <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32> } @llvm.vector.deinterleave4.nxv16i32(<vscale x 16 x i32> [[WIDE_VP_LOAD]])
+; IF-EVL-NEXT:    [[TMP5:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 0
+; IF-EVL-NEXT:    [[TMP8:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 1
+; IF-EVL-NEXT:    [[TMP11:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 2
 ; IF-EVL-NEXT:    [[TMP6:%.*]] = add <vscale x 4 x i32> [[VEC_PHI]], [[TMP5]]
-; IF-EVL-NEXT:    [[TMP7:%.*]] = getelementptr nuw i8, ptr [[TMP0]], i64 [[TMP3]]
-; IF-EVL-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i32> @llvm.experimental.vp.strided.load.nxv4i32.p0.i64(ptr align 4 [[TMP7]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP2]])
 ; IF-EVL-NEXT:    [[TMP9:%.*]] = add <vscale x 4 x i32> [[TMP6]], [[TMP8]]
-; IF-EVL-NEXT:    [[TMP10:%.*]] = getelementptr nuw i8, ptr [[TMP1]], i64 [[TMP3]]
-; IF-EVL-NEXT:    [[TMP11:%.*]] = call <vscale x 4 x i32> @llvm.experimental.vp.strided.load.nxv4i32.p0.i64(ptr align 4 [[TMP10]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP2]])
 ; IF-EVL-NEXT:    [[TMP12:%.*]] = add <vscale x 4 x i32> [[TMP9]], [[TMP11]]
 ; IF-EVL-NEXT:    [[TMP13]] = call <vscale x 4 x i32> @llvm.vp.merge.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> [[TMP12]], <vscale x 4 x i32> [[VEC_PHI]], i32 [[TMP2]])
 ; IF-EVL-NEXT:    [[TMP14:%.*]] = zext i32 [[TMP2]] to i64
@@ -470,8 +470,6 @@ define void @store_factor_4_with_tail_gap(i32 %n, ptr noalias %a) vscale_range(2
 ; IF-EVL-NEXT:  entry:
 ; IF-EVL-NEXT:    br label [[VECTOR_PH:%.*]]
 ; IF-EVL:       vector.ph:
-; IF-EVL-NEXT:    [[TMP0:%.*]] = getelementptr nuw i8, ptr [[A:%.*]], i64 4
-; IF-EVL-NEXT:    [[TMP1:%.*]] = getelementptr nuw i8, ptr [[A]], i64 8
 ; IF-EVL-NEXT:    [[TMP2:%.*]] = call <vscale x 4 x i32> @llvm.stepvector.nxv4i32()
 ; IF-EVL-NEXT:    br label [[VECTOR_BODY:%.*]]
 ; IF-EVL:       vector.body:
@@ -481,14 +479,12 @@ define void @store_factor_4_with_tail_gap(i32 %n, ptr noalias %a) vscale_range(2
 ; IF-EVL-NEXT:    [[TMP3:%.*]] = call i32 @llvm.experimental.get.vector.length.i32(i32 [[AVL]], i32 4, i1 true)
 ; IF-EVL-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 4 x i32> poison, i32 [[TMP3]], i64 0
 ; IF-EVL-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 4 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 4 x i32> poison, <vscale x 4 x i32> zeroinitializer
-; IF-EVL-NEXT:    [[TMP4:%.*]] = zext i32 [[INDEX]] to i64
-; IF-EVL-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 4
-; IF-EVL-NEXT:    [[TMP6:%.*]] = getelementptr nuw i8, ptr [[A]], i64 [[TMP5]]
-; IF-EVL-NEXT:    call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP6]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP3]])
-; IF-EVL-NEXT:    [[TMP7:%.*]] = getelementptr nuw i8, ptr [[TMP0]], i64 [[TMP5]]
-; IF-EVL-NEXT:    call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP7]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP3]])
-; IF-EVL-NEXT:    [[TMP8:%.*]] = getelementptr nuw i8, ptr [[TMP1]], i64 [[TMP5]]
-; IF-EVL-NEXT:    call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP8]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP3]])
+; IF-EVL-NEXT:    [[TMP4:%.*]] = getelementptr inbounds [4 x i32], ptr [[A:%.*]], i32 [[INDEX]], i32 0
+; IF-EVL-NEXT:    [[INTERLEAVE_EVL:%.*]] = mul nuw nsw i32 [[TMP3]], 4
+; IF-EVL-NEXT:    [[INTERLEAVED_GAPS_MASK:%.*]] = call <vscale x 16 x i1> @llvm.vector.interleave4.nxv16i1(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> zeroinitializer)
+; IF-EVL-NEXT:    [[TMP5:%.*]] = and <vscale x 16 x i1> splat (i1 true), [[INTERLEAVED_GAPS_MASK]]
+; IF-EVL-NEXT:    [[INTERLEAVED_VEC:%.*]] = call <vscale x 16 x i32> @llvm.vector.interleave4.nxv16i32(<vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> poison)
+; IF-EVL-NEXT:    call void @llvm.vp.store.nxv16i32.p0(<vscale x 16 x i32> [[INTERLEAVED_VEC]], ptr align 4 [[TMP4]], <vscale x 16 x i1> [[TMP5]], i32 [[INTERLEAVE_EVL]])
 ; IF-EVL-NEXT:    [[CURRENT_ITERATION_NEXT]] = add i32 [[TMP3]], [[INDEX]]
 ; IF-EVL-NEXT:    [[AVL_NEXT]] = sub nuw i32 [[AVL]], [[TMP3]]
 ; IF-EVL-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <vscale x 4 x i32> [[VEC_IND]], [[BROADCAST_SPLAT]]
@@ -508,8 +504,6 @@ define void @store_factor_4_with_tail_gap(i32 %n, ptr noalias %a) vscale_range(2
 ; NO-VP:       vector.ph:
 ; NO-VP-NEXT:    [[N_MOD_VF:%.*]] = urem i32 [[N]], [[TMP1]]
 ; NO-VP-NEXT:    [[N_VEC:%.*]] = sub i32 [[N]], [[N_MOD_VF]]
-; NO-VP-NEXT:    [[TMP4:%.*]] = getelementptr nuw i8, ptr [[A:%.*]], i64 4
-; NO-VP-NEXT:    [[TMP5:%.*]] = getelementptr nuw i8, ptr [[A]], i64 8
 ; NO-VP-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i32> @llvm.stepvector.nxv4i32()
 ; NO-VP-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 4 x i32> poison, i32 [[TMP1]], i64 0
 ; NO-VP-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 4 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 4 x i32> poison, <vscale x 4 x i32> zeroinitializer
@@ -517,14 +511,10 @@ define void @store_factor_4_with_tail_gap(i32 %n, ptr noalias %a) vscale_range(2
 ; NO-VP:       vector.body:
 ; NO-VP-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
 ; NO-VP-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 4 x i32> [ [[TMP6]], [[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], [[VECTOR_BODY]] ]
-; NO-VP-NEXT:    [[TMP7:%.*]] = zext i32 [[INDEX]] to i64
-; NO-VP-NEXT:    [[TMP8:%.*]] = shl nuw i64 [[TMP7]], 4
-; NO-VP-NEXT:    [[TMP9:%.*]] = getelementptr nuw i8, ptr [[A]], i64 [[TMP8]]
-; NO-VP-NEXT:    call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP9]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP1]])
-; NO-VP-NEXT:    [[TMP10:%.*]] = getelementptr nuw i8, ptr [[TMP4]], i64 [[TMP8]]
-; NO-VP-NEXT:    call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP10]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP1]])
-; NO-VP-NEXT:    [[TMP11:%.*]] = getelementptr nuw i8, ptr [[TMP5]], i64 [[TMP8]]
-; NO-VP-NEXT:    call void @llvm.experimental.vp.strided.store.nxv4i32.p0.i64(<vscale x 4 x i32> [[VEC_IND]], ptr align 4 [[TMP11]], i64 16, <vscale x 4 x i1> splat (i1 true), i32 [[TMP1]])
+; NO-VP-NEXT:    [[TMP3:%.*]] = getelementptr inbounds [4 x i32], ptr [[A:%.*]], i32 [[INDEX]], i32 0
+; NO-VP-NEXT:    [[INTERLEAVED_GAPS_MASK:%.*]] = call <vscale x 16 x i1> @llvm.vector.interleave4.nxv16i1(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> zeroinitializer)
+; NO-VP-NEXT:    [[INTERLEAVED_VEC:%.*]] = call <vscale x 16 x i32> @llvm.vector.interleave4.nxv16i32(<vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> [[VEC_IND]], <vscale x 4 x i32> poison)
+; NO-VP-NEXT:    call void @llvm.masked.store.nxv16i32.p0(<vscale x 16 x i32> [[INTERLEAVED_VEC]], ptr align 4 [[TMP3]], <vscale x 16 x i1> [[INTERLEAVED_GAPS_MASK]])
 ; NO-VP-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], [[TMP1]]
 ; NO-VP-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <vscale x 4 x i32> [[VEC_IND]], [[BROADCAST_SPLAT]]
 ; NO-VP-NEXT:    [[TMP12:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]



More information about the llvm-commits mailing list