[llvm] [LV] Add support for widening loads/stores to a UF x VF (PR #217670)

Benjamin Maxwell via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 01:59:34 PDT 2026


https://github.com/MacDue updated https://github.com/llvm/llvm-project/pull/217670

>From c09effc5664a3e05c2f5fd7873d1ca0448f93309 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Thu, 20 Aug 2026 14:39:53 +0000
Subject: [PATCH 01/20] [LV] Add support for widening loads/stores to a VF
 multiple

This patch adds support for widening loads and stores by a "VF multiple"
that must divide the UF. It is currently limited to unmasked operations,
but we plan to extend it to masked loops via the wide active lane mask.

For now, this is driven by a new TTI hook,
`getPreferredVFMultipleForMemoryOp`. A small VPlan transform uses that
hook to set the VF multiple on `VPWidenLoadRecipe` and
`VPWidenStoreRecipe`.

When VPlan unrolling encounters a load or store with a VF multiple
greater than 1, it inserts extracts for the unroll parts of widened
loads and concatenates multiple unroll parts for widened stores.

On AArch64 this is used to target multi-vector load/store instructions.
---
 .../llvm/Analysis/TargetTransformInfo.h       |   9 +
 .../llvm/Analysis/TargetTransformInfoImpl.h   |   8 +
 llvm/lib/Analysis/TargetTransformInfo.cpp     |   7 +
 .../AArch64/AArch64TargetTransformInfo.cpp    |  31 ++
 .../AArch64/AArch64TargetTransformInfo.h      |   4 +
 .../Transforms/Vectorize/LoopVectorize.cpp    |   2 +
 llvm/lib/Transforms/Vectorize/VPlan.h         |  27 +-
 .../Transforms/Vectorize/VPlanPatternMatch.h  |  37 +-
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp |  28 +-
 .../Transforms/Vectorize/VPlanTransforms.cpp  |  41 ++
 .../Transforms/Vectorize/VPlanTransforms.h    |   5 +
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp |  66 ++++
 .../AArch64/multi-vector-mem-ops.ll           | 374 ++++++++++++++++++
 .../vplan-printing-multi-vector-mem-ops.ll    | 151 +++++++
 .../VPlan/vplan-print-before-after-all.ll     |   1 +
 15 files changed, 784 insertions(+), 7 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index a33e6f62e941e..434e7bbf980d9 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1001,6 +1001,15 @@ class TargetTransformInfo {
                                 unsigned Opcode1,
                                 const SmallBitVector &OpcodeMask) const;
 
+  /// Return the preferred multiple of VF to use for a contiguous load/store.
+  /// Returning 1 leaves the operation at VF. The returned value must divide UF.
+  ///
+  /// \p Opcode must be either Instruction::Load or Instruction::Store.
+  LLVM_ABI unsigned
+  getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataType,
+                                    ElementCount VF, unsigned UF,
+                                    bool IsMasked = false) const;
+
   /// Return true if we should be enabling ordered reductions for the target.
   LLVM_ABI bool enableOrderedReductions() const;
 
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index a19c122c16f20..01006cfce4d08 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -429,6 +429,14 @@ class LLVM_ABI TargetTransformInfoImplBase {
     return false;
   }
 
+  virtual unsigned getPreferredVFMultipleForMemoryOp(unsigned Opcode,
+                                                     Type *DataType,
+                                                     ElementCount VF,
+                                                     unsigned UF,
+                                                     bool IsMasked) const {
+    return 1;
+  }
+
   virtual bool isLegalInterleavedAccessType(VectorType *VTy, unsigned Factor,
                                             Align Alignment,
                                             unsigned AddrSpace) const {
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index af73a615f25fa..33cb290209afa 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -554,6 +554,13 @@ bool TargetTransformInfo::isLegalStridedLoadStore(Type *DataType,
   return TTIImpl->isLegalStridedLoadStore(DataType, Alignment);
 }
 
+unsigned TargetTransformInfo::getPreferredVFMultipleForMemoryOp(
+    unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
+    bool IsMasked) const {
+  return TTIImpl->getPreferredVFMultipleForMemoryOp(Opcode, DataType, VF, UF,
+                                                    IsMasked);
+}
+
 bool TargetTransformInfo::isLegalInterleavedAccessType(
     VectorType *VTy, unsigned Factor, Align Alignment,
     unsigned AddrSpace) const {
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index ddea78fa4e808..93a1b3492619e 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -5959,6 +5959,37 @@ bool AArch64TTIImpl::isLegalSpeculativeLoad(Type *DataType,
          Size.getFixedValue() <= 16;
 }
 
+unsigned
+AArch64TTIImpl::getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataTy,
+                                                  ElementCount VF, unsigned UF,
+                                                  bool IsMasked) const {
+  assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
+         "expected load/store opcode");
+  if (IsMasked)
+    return 1; // TODO: Support masked multi-vector loads/stores.
+
+  if (!ST->enableSubRegLiveness())
+    return 1;
+
+  if ((Opcode != Instruction::Load && Opcode != Instruction::Store) ||
+      !ST->hasSVE2p1() || !VF.isScalable() || !isPowerOf2_32(UF))
+    return 1;
+
+  unsigned VectorWidth = VF.getKnownMinValue() * DL.getTypeSizeInBits(DataTy);
+  if (VectorWidth % 128 != 0)
+    return 1;
+
+  for (unsigned TargetWidth : {512u, 256u}) {
+    if (TargetWidth % VectorWidth == 0) {
+      unsigned Scale = TargetWidth / VectorWidth;
+      if (Scale <= UF)
+        return Scale;
+    }
+  }
+
+  return 1;
+}
+
 unsigned
 AArch64TTIImpl::getMaxInterleaveFactor(ElementCount VF,
                                        bool HasUnorderedReductions) const {
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index d090f69c1476a..a5513114bd112 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -282,6 +282,10 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
   bool isLegalSpeculativeLoad(Type *DataType,
                               unsigned AddressSpace) const override;
 
+  unsigned getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataType,
+                                             ElementCount VF, unsigned UF,
+                                             bool IsMasked) const override;
+
   void getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
                                TTI::UnrollingPreferences &UP,
                                OptimizationRemarkEmitter *ORE) const override;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index c5f2d0c2a1520..9cdae4b9c136e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5752,6 +5752,8 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
                  *PSE.getSE(), TTI, Config.CostKind, BestVF, BestUF);
   // TODO: Move to VPlan transform stage once the transition to the VPlan-based
   // cost model is complete for better cost estimates.
+  RUN_VPLAN_PASS(VPlanTransforms::scaleMemoryAccessesByUF, BestVPlan, BestVF,
+                 BestUF, CM.TTI);
   RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
   RUN_VPLAN_PASS(VPlanTransforms::materializePacksAndUnpacks, BestVPlan);
   RUN_VPLAN_PASS(VPlanTransforms::materializeBroadcasts, BestVPlan);
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index aa93dbaa65170..c1747697a9f2f 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1313,6 +1313,8 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
     WideActiveLaneMask,
     // Extracts each unrolled part of a (VF * UF) widened vector/mask.
     ExtractVectorForPart,
+    // Concatenates its unrolled part operands into one widened vector.
+    ConcatVectorParts,
     ExplicitVectorLength,
     // Represents the incoming loop-invariant alias-mask. All memory accesses
     // in the loop must stay within the active lanes.
@@ -3753,6 +3755,10 @@ class LLVM_ABI_FOR_TEST VPWidenMemoryRecipe : public VPIRMetadata {
   /// Whether the memory access is masked.
   bool IsMasked = false;
 
+  /// Multiple of VF used to widen this memory operation. The final operation
+  /// loads or stores VF * VFMultiple elements
+  unsigned VFMultiple = 1;
+
   void setMask(VPValue *Mask) {
     assert(!IsMasked && "cannot re-set mask");
     if (!Mask)
@@ -3799,6 +3805,12 @@ class LLVM_ABI_FOR_TEST VPWidenMemoryRecipe : public VPIRMetadata {
   InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const;
 
   Instruction &getIngredient() const { return Ingredient; }
+
+  /// Set the VF multiple for this memory operation.
+  void setVFMultiple(unsigned VFMultiple) { this->VFMultiple = VFMultiple; }
+
+  /// Returns the VF multiple of this memory operation.
+  unsigned getVFMultiple() const { return VFMultiple; }
 };
 
 /// A recipe for widening load operations, using the address to load from and an
@@ -3814,8 +3826,11 @@ struct LLVM_ABI_FOR_TEST VPWidenLoadRecipe final : public VPSingleDefRecipe,
   }
 
   VPWidenLoadRecipe *clone() override {
-    return new VPWidenLoadRecipe(cast<LoadInst>(Ingredient), getAddr(),
-                                 getMask(), Consecutive, *this, getDebugLoc());
+    auto *R =
+        new VPWidenLoadRecipe(cast<LoadInst>(Ingredient), getAddr(), getMask(),
+                              Consecutive, *this, getDebugLoc());
+    R->setVFMultiple(VFMultiple);
+    return R;
   }
 
   VP_CLASSOF_IMPL(VPRecipeBase::VPWidenLoadSC);
@@ -3919,9 +3934,11 @@ struct LLVM_ABI_FOR_TEST VPWidenStoreRecipe final : public VPRecipeBase,
   }
 
   VPWidenStoreRecipe *clone() override {
-    return new VPWidenStoreRecipe(cast<StoreInst>(Ingredient), getAddr(),
-                                  getStoredValue(), getMask(), Consecutive,
-                                  *this, getDebugLoc());
+    auto *R = new VPWidenStoreRecipe(cast<StoreInst>(Ingredient), getAddr(),
+                                     getStoredValue(), getMask(), Consecutive,
+                                     *this, getDebugLoc());
+    R->setVFMultiple(VFMultiple);
+    return R;
   }
 
   VP_CLASSOF_IMPL(VPRecipeBase::VPWidenStoreSC);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
index d795f9748f1ba..22f35f1b96ecd 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
@@ -292,7 +292,10 @@ struct Recipe_match {
     // Check for recipes that do not have opcodes.
     if constexpr (std::is_same_v<RecipeTy, VPScalarIVStepsRecipe> ||
                   std::is_same_v<RecipeTy, VPDerivedIVRecipe> ||
-                  std::is_same_v<RecipeTy, VPVectorEndPointerRecipe>)
+                  std::is_same_v<RecipeTy, VPVectorEndPointerRecipe> ||
+                  std::is_same_v<RecipeTy, VPVectorPointerRecipe> ||
+                  std::is_same_v<RecipeTy, VPWidenLoadRecipe> ||
+                  std::is_same_v<RecipeTy, VPWidenStoreRecipe>)
       return DefR;
     else
       return DefR && DefR->getOpcode() == Opcode;
@@ -1049,6 +1052,38 @@ m_MaskedStore(const Addr_t &Addr, const Val_t &Val, const Mask_t &Mask) {
   return Store_match<Addr_t, Val_t, Mask_t>(Addr, Val, Mask);
 }
 
+template <typename Op0_t, typename Op1_t>
+using VectorPointerRecipe_match =
+    Recipe_match<std::tuple<Op0_t, Op1_t>, 0,
+                 /*Commutative*/ false, VPVectorPointerRecipe>;
+
+template <typename Op0_t, typename Op1_t>
+VectorPointerRecipe_match<Op0_t, Op1_t> m_VecPtr(const Op0_t &Op0,
+                                                 const Op1_t &Op1) {
+  return VectorPointerRecipe_match<Op0_t, Op1_t>(Op0, Op1);
+}
+
+template <typename Op0_t>
+using VPWidenLoadRecipe_match =
+    Recipe_match<std::tuple<Op0_t>, 0,
+                 /*Commutative*/ false, VPWidenLoadRecipe>;
+
+template <typename Op0_t>
+VPWidenLoadRecipe_match<Op0_t> m_WidenLoad(const Op0_t &Op0) {
+  return VPWidenLoadRecipe_match<Op0_t>(Op0);
+}
+
+template <typename Op0_t, typename Op1_t>
+using VPWidenStoreRecipe_match =
+    Recipe_match<std::tuple<Op0_t, Op1_t>, 0,
+                 /*Commutative*/ false, VPWidenStoreRecipe>;
+
+template <typename Op0_t, typename Op1_t>
+VPWidenStoreRecipe_match<Op0_t, Op1_t> m_WidenStore(const Op0_t &Op0,
+                                                    const Op1_t &Op1) {
+  return VPWidenStoreRecipe_match<Op0_t, Op1_t>(Op0, Op1);
+}
+
 template <typename Op0_t, typename Op1_t>
 using VectorEndPointerRecipe_match =
     Recipe_match<std::tuple<Op0_t, Op1_t>, 0,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index ae9001dc8c67f..4dc978e04f343 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -564,6 +564,9 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
     return StructTy->getTypeAtIndex(
         cast<VPConstantInt>(Operands[1])->getZExtValue());
   }
+  case VPInstruction::ConcatVectorParts:
+  case VPInstruction::ExtractVectorForPart:
+    return Op0Ty;
   case VPInstruction::FirstActiveLane:
   case VPInstruction::LastActiveLane:
   case VPInstruction::NumActiveLanes:
@@ -691,6 +694,7 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
   case VPInstruction::LastActiveLane:
   case VPInstruction::ExtractLane:
   case VPInstruction::ExtractLastActive:
+  case VPInstruction::ConcatVectorParts:
     // Cannot determine the number of operands from the opcode.
     return -1u;
   }
@@ -1177,6 +1181,19 @@ Value *VPInstruction::generate(VPTransformState &State,
                                          vputils::getIntrinsicID(this), Args,
                                          /*FMFSource=*/nullptr, getName());
   }
+  case VPInstruction::ConcatVectorParts: {
+    unsigned VectorOps = getNumOperands();
+    auto *WideDataTy = VectorType::get(
+        getScalarType(), State.VF.multiplyCoefficientBy(VectorOps));
+    Value *WideData = PoisonValue::get(WideDataTy);
+
+    for (unsigned I = 0; I < VectorOps; ++I) {
+      Value *Part = State.get(getOperand(I));
+      WideData = Builder.CreateInsertVector(WideDataTy, WideData, Part,
+                                            I * State.VF.getKnownMinValue());
+    }
+    return WideData;
+  }
   default:
     llvm_unreachable("Unsupported opcode for instruction");
   }
@@ -1726,6 +1743,7 @@ bool VPInstruction::opcodeMayReadOrWriteFromMemory() const {
   case VPInstruction::ActiveLaneMask:
   case VPInstruction::WideActiveLaneMask:
   case VPInstruction::IncomingAliasMask:
+  case VPInstruction::ConcatVectorParts:
   case VPInstruction::ExitingIVValue:
   case VPInstruction::ExplicitVectorLength:
   case VPInstruction::FirstActiveLane:
@@ -1862,6 +1880,9 @@ void VPInstruction::printRecipe(raw_ostream &O, const Twine &Indent,
   case VPInstruction::IncomingAliasMask:
     O << "incoming-alias-mask";
     break;
+  case VPInstruction::ConcatVectorParts:
+    O << "concat-vector-parts";
+    break;
   case VPInstruction::ExplicitVectorLength:
     O << "EXPLICIT-VECTOR-LENGTH";
     break;
@@ -4369,7 +4390,8 @@ InstructionCost VPWidenMemoryRecipe::computeCost(ElementCount VF,
 
 void VPWidenLoadRecipe::execute(VPTransformState &State) {
   Type *ScalarDataTy = getScalarType();
-  auto *DataTy = VectorType::get(ScalarDataTy, State.VF);
+  auto *DataTy =
+      VectorType::get(ScalarDataTy, State.VF.multiplyCoefficientBy(VFMultiple));
   bool CreateGather = !isConsecutive();
 
   auto &Builder = State.Builder;
@@ -4399,6 +4421,8 @@ void VPWidenLoadRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
   O << Indent << "WIDEN ";
   printAsOperand(O, SlotTracker);
   O << " = load ";
+  if (VFMultiple > 1)
+    O << "x" << VFMultiple << ' ';
   printOperands(O, SlotTracker);
 }
 #endif
@@ -4486,6 +4510,8 @@ void VPWidenStoreRecipe::execute(VPTransformState &State) {
 void VPWidenStoreRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
                                      VPSlotTracker &SlotTracker) const {
   O << Indent << "WIDEN store ";
+  if (VFMultiple > 1)
+    O << "x" << VFMultiple << ' ';
   printOperands(O, SlotTracker);
 }
 #endif
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index cc1c9a683e404..b670c7be7e3de 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4179,6 +4179,47 @@ void VPlanTransforms::sinkPredicatedStores(VPlan &Plan,
   }
 }
 
+void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
+                                              unsigned UF,
+                                              const TargetTransformInfo &TTI) {
+  if (UF == 1)
+    return;
+
+  for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
+           vp_depth_first_deep(Plan.getVectorLoopRegion()->getEntry()))) {
+    for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
+      uint64_t Stride;
+      VPValue *StoredValue = nullptr;
+      auto m_ConstantStrideVecPtr =
+          m_VecPtr(m_VPValue(), m_ConstantInt(Stride));
+      if ((!match(&R, m_WidenLoad(m_ConstantStrideVecPtr)) &&
+           !match(&R, m_WidenStore(m_ConstantStrideVecPtr,
+                                   m_VPValue(StoredValue)))) ||
+          Stride != 1)
+        continue;
+
+      auto *MemOp = cast<VPWidenMemoryRecipe>(&R);
+      if (!MemOp->isConsecutive())
+        continue;
+
+      // TODO: Support masked loads/stores. This requires widening the header
+      // mask to the same factor as the memory operation.
+      assert(!MemOp->isMasked() && "Masked accesses are not supported yet");
+
+      Type *AccessType = StoredValue ? StoredValue->getScalarType()
+                                     : R.getVPSingleValue()->getScalarType();
+      unsigned Opcode = isa<VPWidenLoadRecipe>(MemOp->getAsRecipe())
+                            ? Instruction::Load
+                            : Instruction::Store;
+      unsigned ScaleFactor = TTI.getPreferredVFMultipleForMemoryOp(
+          Opcode, AccessType, VF, UF, /*IsMasked=*/false);
+      assert((ScaleFactor != 0 && UF % ScaleFactor == 0) &&
+             "ScaleFactor must divide UF");
+      MemOp->setVFMultiple(ScaleFactor);
+    }
+  }
+}
+
 /// Returns true if \p V is VPWidenLoadRecipe or VPInterleaveRecipe that can be
 /// converted to a narrower recipe. \p V is used by a wide recipe that feeds a
 /// store interleave group at index \p Idx, \p WideMember0 is the recipe feeding
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 30d584ab8f589..3f0634f039807 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -482,6 +482,11 @@ struct VPlanTransforms {
   static void sinkPredicatedStores(VPlan &Plan, PredicatedScalarEvolution &PSE,
                                    const Loop *L);
 
+  /// Widens memory operations by a factor of UF based on a target hook.
+  /// This allows targets to use wider memory operations when profitable.
+  static void scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF, unsigned UF,
+                                      const TargetTransformInfo &TTI);
+
   // Materialize vector trip counts for constants early if it can simply be
   // computed as (Original TC / VF * UF) * VF * UF.
   static void
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index d1e30ab0cff65..e3705bd900f8c 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -74,6 +74,9 @@ class UnrollState {
     return Plan.getConstantInt(CanIVIntTy, Part);
   }
 
+  /// Unroll a VPWidenLoadRecipe or VPWidenStoreRecipe with a VFMultiple > 1.
+  void unrollMemOpWithVFMultiple(VPRecipeBase &R, unsigned VFMultiple);
+
 public:
   UnrollState(VPlan &Plan, unsigned UF) : Plan(Plan), UF(UF) {}
 
@@ -290,6 +293,62 @@ void UnrollState::unrollHeaderPHIByUF(VPHeaderPHIRecipe *R,
   }
 }
 
+void UnrollState::unrollMemOpWithVFMultiple(VPRecipeBase &R,
+                                            unsigned VFMultiple) {
+  assert(VFMultiple > 1 && UF % VFMultiple == 0);
+  SmallVector<VPRecipeBase *, 4> Groups(UF / VFMultiple, nullptr);
+  Groups[0] = &R;
+
+  // A memory op with a VFMultiple is widened to VF * VFMultiple elements, so
+  // after unrolling by UF we materialize UF / VFMultiple such ops, each
+  // covering VFMultiple unroll parts.
+  VPBuilder Builder = VPBuilder::getToInsertAfter(&R);
+  for (unsigned Group = 1; Group < Groups.size(); ++Group) {
+    auto *Copy = Builder.insert(R.clone());
+    remapOperands(Copy, Group * VFMultiple);
+    Groups[Group] = Copy;
+  }
+
+  if (auto *Store = dyn_cast<VPWidenStoreRecipe>(&R)) {
+    VPValue *StoredValue = Store->getStoredValue();
+    for (unsigned Group = 0; Group < Groups.size(); ++Group) {
+      VPRecipeBase *Store = Groups[Group];
+      Builder.setInsertPoint(Store);
+      SmallVector<VPValue *, 4> Parts;
+      // We need to concatenate VFMultiple parts to form the stored value.
+      for (unsigned Part = 0; Part < VFMultiple; ++Part)
+        Parts.push_back(
+            getValueForPart(StoredValue, Group * VFMultiple + Part));
+      auto *Concat =
+          Builder.createNaryOp(VPInstruction::ConcatVectorParts, Parts);
+      Groups[Group]->setOperand(1, Concat);
+      ToSkip.insert(Concat);
+    }
+  } else {
+    assert(isa<VPWidenLoadRecipe>(R) && "Expected a load recipe");
+    // We need to extract each unroll part as a subvector.
+    auto ExtractPart0 = Builder.createNaryOp(
+        VPInstruction::ExtractVectorForPart,
+        {Groups[0]->getVPSingleValue(), getConstantInt(0)});
+    // First replace R with an extract of the first unroll part (ExtractPart0).
+    R.getVPSingleValue()->replaceUsesWithIf(
+        ExtractPart0, [&](VPUser &U, unsigned) { return &U != ExtractPart0; });
+    ToSkip.insert(ExtractPart0);
+
+    // Create extracts for the remaining unroll parts and remap later uses of
+    // ExtractPart0 to the correct unrolled part.
+    for (unsigned Part = 1; Part != UF; ++Part) {
+      VPRecipeBase *Group = Groups[Part / VFMultiple];
+      unsigned IndexInGroup = Part % VFMultiple;
+      auto *Extract = Builder.createNaryOp(
+          VPInstruction::ExtractVectorForPart,
+          {Group->getVPSingleValue(), getConstantInt(IndexInGroup)});
+      addRecipeForPart(ExtractPart0, Extract, Part);
+      ToSkip.insert(Extract);
+    }
+  }
+}
+
 /// Handle non-header-phi recipes.
 void UnrollState::unrollRecipeByUF(VPRecipeBase &R) {
   if (match(&R, m_CombineOr(m_BranchOnCond(), m_BranchOnCount())))
@@ -301,6 +360,13 @@ void UnrollState::unrollRecipeByUF(VPRecipeBase &R) {
       return;
     }
   }
+
+  if (auto *WidenMem = dyn_cast<VPWidenMemoryRecipe>(&R);
+      WidenMem && WidenMem->getVFMultiple() > 1) {
+    unrollMemOpWithVFMultiple(R, WidenMem->getVFMultiple());
+    return;
+  }
+
   if (auto *RepR = dyn_cast<VPReplicateRecipe>(&R)) {
     if (isa<StoreInst>(RepR->getUnderlyingValue()) &&
         RepR->getOperand(1)->isDefinedOutsideLoopRegions()) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
new file mode 100644
index 0000000000000..f20e7aff70ef7
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -0,0 +1,374 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^middle.block:" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -sve-tail-folding=disabled %s | FileCheck %s --check-prefix=UNMASKED-SVE2P1
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @mixed_i64_i32_accesses(
+; UNMASKED-SVE2P1-LABEL: define void @mixed_i64_i32_accesses(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[X:%.*]], ptr noalias [[Y:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = shl nuw nsw i64 [[TMP3]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i64, ptr [[TMP5]], i64 [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 8 x i64>, ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = add <vscale x 4 x i64> [[TMP6]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <vscale x 4 x i64> [[TMP7]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <vscale x 4 x i64> [[TMP8]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add <vscale x 4 x i64> [[TMP9]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP17]], ptr [[TMP5]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP18]], <vscale x 4 x i64> [[TMP13]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP19]], ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 16 x i32>, ptr [[TMP20]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = add <vscale x 4 x i32> [[TMP21]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = add <vscale x 4 x i32> [[TMP22]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = add <vscale x 4 x i32> [[TMP23]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = add <vscale x 4 x i32> [[TMP24]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP25]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP30]], <vscale x 4 x i32> [[TMP27]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP31]], <vscale x 4 x i32> [[TMP28]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP32]], ptr [[TMP20]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP33]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+  ptr noalias %x, ptr noalias %y, i64 %n) {
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %px = getelementptr inbounds i64, ptr %x, i64 %iv
+  %vx = load i64, ptr %px, align 8
+  %ax = add i64 %vx, 1
+  store i64 %ax, ptr %px, align 8
+  %py = getelementptr inbounds i32, ptr %y, i64 %iv
+  %vy = load i32, ptr %py, align 4
+  %ay = add i32 %vy, 1
+  store i32 %ay, ptr %py, align 4
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+define void @mixed_i32_more_frequent_than_i64(
+; UNMASKED-SVE2P1-LABEL: define void @mixed_i32_more_frequent_than_i64(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[X:%.*]], ptr noalias [[Y:%.*]], ptr noalias [[Z:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = shl nuw nsw i64 [[TMP3]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i64, ptr [[TMP5]], i64 [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 8 x i64>, ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD3]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD3]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = add <vscale x 4 x i64> [[TMP6]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <vscale x 4 x i64> [[TMP7]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <vscale x 4 x i64> [[TMP8]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add <vscale x 4 x i64> [[TMP9]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP17]], ptr [[TMP5]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP18]], <vscale x 4 x i64> [[TMP13]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP19]], ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 16 x i32>, ptr [[TMP20]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = add <vscale x 4 x i32> [[TMP21]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = add <vscale x 4 x i32> [[TMP22]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = add <vscale x 4 x i32> [[TMP23]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = add <vscale x 4 x i32> [[TMP24]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP25]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP30]], <vscale x 4 x i32> [[TMP27]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP31]], <vscale x 4 x i32> [[TMP28]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP32]], ptr [[TMP20]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = getelementptr inbounds i32, ptr [[Z]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 16 x i32>, ptr [[TMP33]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP34:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP35:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP36:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP37:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP38:%.*]] = add <vscale x 4 x i32> [[TMP34]], splat (i32 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP39:%.*]] = add <vscale x 4 x i32> [[TMP35]], splat (i32 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP40:%.*]] = add <vscale x 4 x i32> [[TMP36]], splat (i32 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP41:%.*]] = add <vscale x 4 x i32> [[TMP37]], splat (i32 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP42:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP38]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP43:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP42]], <vscale x 4 x i32> [[TMP39]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP44:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP43]], <vscale x 4 x i32> [[TMP40]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP45:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP44]], <vscale x 4 x i32> [[TMP41]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP45]], ptr [[TMP33]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP46:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP46]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+  ptr noalias %x, ptr noalias %y, ptr noalias %z, i64 %n) {
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %px = getelementptr inbounds i64, ptr %x, i64 %iv
+  %vx = load i64, ptr %px, align 8
+  %ax = add i64 %vx, 1
+  store i64 %ax, ptr %px, align 8
+  %py = getelementptr inbounds i32, ptr %y, i64 %iv
+  %vy = load i32, ptr %py, align 4
+  %ay = add i32 %vy, 1
+  store i32 %ay, ptr %py, align 4
+  %pz = getelementptr inbounds i32, ptr %z, i64 %iv
+  %vz = load i32, ptr %pz, align 4
+  %az = add i32 %vz, 2
+  store i32 %az, ptr %pz, align 4
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+define void @first_order_recurrence_i64_scaled_load_and_store(
+; UNMASKED-SVE2P1-LABEL: define void @first_order_recurrence_i64_scaled_load_and_store(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[SRC:%.*]], ptr noalias [[DST:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[FIRST:%.*]] = load i64, ptr [[SRC]], align 8
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[VECTOR_RECUR_INIT:%.*]] = insertelement <2 x i64> poison, i64 [[FIRST]], i32 1
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VECTOR_RECUR:%.*]] = phi <2 x i64> [ [[VECTOR_RECUR_INIT]], %[[VECTOR_PH]] ], [ [[WIDE_LOAD3:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = add nuw i64 1, [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 [[TMP0]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i64, ptr [[TMP1]], i64 2
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[TMP1]], i64 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i64, ptr [[TMP1]], i64 6
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <2 x i64>, ptr [[TMP1]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD1:%.*]] = load <2 x i64>, ptr [[TMP2]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <2 x i64>, ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3]] = load <2 x i64>, ptr [[TMP4]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = shufflevector <2 x i64> [[VECTOR_RECUR]], <2 x i64> [[WIDE_LOAD]], <2 x i32> <i32 1, i32 2>
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = shufflevector <2 x i64> [[WIDE_LOAD]], <2 x i64> [[WIDE_LOAD1]], <2 x i32> <i32 1, i32 2>
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = shufflevector <2 x i64> [[WIDE_LOAD1]], <2 x i64> [[WIDE_LOAD2]], <2 x i32> <i32 1, i32 2>
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = shufflevector <2 x i64> [[WIDE_LOAD2]], <2 x i64> [[WIDE_LOAD3]], <2 x i32> <i32 1, i32 2>
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = add <2 x i64> [[TMP5]], [[WIDE_LOAD]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = add <2 x i64> [[TMP6]], [[WIDE_LOAD1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <2 x i64> [[TMP7]], [[WIDE_LOAD2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <2 x i64> [[TMP8]], [[WIDE_LOAD3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i64, ptr [[DST]], i64 [[TMP0]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i64, ptr [[TMP13]], i64 2
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i64, ptr [[TMP13]], i64 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i64, ptr [[TMP13]], i64 6
+; UNMASKED-SVE2P1-NEXT:    store <2 x i64> [[TMP9]], ptr [[TMP13]], align 8
+; UNMASKED-SVE2P1-NEXT:    store <2 x i64> [[TMP10]], ptr [[TMP14]], align 8
+; UNMASKED-SVE2P1-NEXT:    store <2 x i64> [[TMP11]], ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    store <2 x i64> [[TMP12]], ptr [[TMP16]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+  ptr noalias %src, ptr noalias %dst) {
+entry:
+  %first = load i64, ptr %src, align 8
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 1, %entry ], [ %next, %loop ]
+  %prev = phi i64 [ %first, %entry ], [ %cur, %loop ]
+  %cur.ptr = getelementptr inbounds i64, ptr %src, i64 %iv
+  %cur = load i64, ptr %cur.ptr, align 8
+  %sum = add i64 %prev, %cur
+  %dst.ptr = getelementptr inbounds i64, ptr %dst, i64 %iv
+  store i64 %sum, ptr %dst.ptr, align 8
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp eq i64 %next, 1025
+  br i1 %cmp, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+define i64 @i64_sum_reduction_scaled_partial_reduce(ptr noalias %a, i64 %n) {
+; UNMASKED-SVE2P1-LABEL: define i64 @i64_sum_reduction_scaled_partial_reduce(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 3
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP11:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP12:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP4]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9]] = add <vscale x 2 x i64> [[VEC_PHI]], [[TMP5]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_PHI1]], [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP11]] = add <vscale x 2 x i64> [[VEC_PHI2]], [[TMP7]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP12]] = add <vscale x 2 x i64> [[VEC_PHI3]], [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %acc = phi i64 [ 0, %entry ], [ %sum, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  %ld = load i64, ptr %p, align 8
+  %sum = add i64 %acc, %ld
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret i64 %sum
+}
+
+define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold) {
+; UNMASKED-SVE2P1-LABEL: define i64 @find_last_i64_scaled_load(
+; UNMASKED-SVE2P1-SAME: i64 [[N:%.*]], ptr noalias [[DATA:%.*]], i64 [[THRESHOLD:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 3
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[THRESHOLD]], i64 0
+; UNMASKED-SVE2P1-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP28:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP29:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP30:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP31:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP24:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP25:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP26:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP27:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i64, ptr [[DATA]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP7]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP9]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP10]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP11]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = freeze <vscale x 2 x i1> [[TMP12]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = freeze <vscale x 2 x i1> [[TMP13]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = or <vscale x 2 x i1> [[TMP16]], [[TMP17]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = freeze <vscale x 2 x i1> [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = or <vscale x 2 x i1> [[TMP18]], [[TMP19]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = freeze <vscale x 2 x i1> [[TMP15]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = or <vscale x 2 x i1> [[TMP20]], [[TMP21]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call i1 @llvm.vector.reduce.or.nxv2i1(<vscale x 2 x i1> [[TMP22]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP24]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP12]], <vscale x 2 x i1> [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP25]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP13]], <vscale x 2 x i1> [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP26]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP14]], <vscale x 2 x i1> [[TMP5]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP27]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP15]], <vscale x 2 x i1> [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP28]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP8]], <vscale x 2 x i64> [[VEC_PHI]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP29]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP9]], <vscale x 2 x i64> [[VEC_PHI1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP30]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP10]], <vscale x 2 x i64> [[VEC_PHI2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP31]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP11]], <vscale x 2 x i64> [[VEC_PHI3]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP32]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %data.phi = phi i64 [ -1, %entry ], [ %select.data, %loop ]
+  %p = getelementptr inbounds i64, ptr %data, i64 %iv
+  %ld = load i64, ptr %p, align 8
+  %select.cmp = icmp slt i64 %threshold, %ld
+  %select.data = select i1 %select.cmp, i64 %ld, i64 %data.phi
+  %next = add nuw i64 %iv, 1
+  %exit.cmp = icmp eq i64 %next, %n
+  br i1 %exit.cmp, label %exit, label %loop
+
+exit:
+  ret i64 %select.data
+}
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
new file mode 100644
index 0000000000000..3943cb5f28103
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
@@ -0,0 +1,151 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter-out-after "middle.block:" --version 6
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-before="scaleMemoryAccessesByUF$" 2>&1 \
+; RUN:   | FileCheck --check-prefix=BEFORE-SCALE %s
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-after="scaleMemoryAccessesByUF$" 2>&1 \
+; RUN:   | FileCheck --check-prefix=AFTER-SCALE %s
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-after="unrollByUF$" 2>&1 \
+; RUN:   | FileCheck --check-prefix=AFTER-UNROLL %s
+
+define void @i64_load_store(ptr noalias %x, i64 %n) {
+; BEFORE-SCALE-LABEL: VPlan for loop in 'i64_load_store'
+; BEFORE-SCALE:  VPlan 'Initial VPlan for VF={vscale x 4},UF>=1' {
+; BEFORE-SCALE-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; BEFORE-SCALE-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; BEFORE-SCALE-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; BEFORE-SCALE-NEXT:  vp<[[VP3:%[0-9]+]]> = original trip-count
+; BEFORE-SCALE-EMPTY:
+; BEFORE-SCALE-NEXT:  ir-bb<entry>:
+; BEFORE-SCALE-NEXT:    EMIT vp<[[VP3]]> = EXPAND SCEV (1 umax %n)
+; BEFORE-SCALE-NEXT:    EMIT-SCALAR vp<[[VP4:%[0-9]+]]> = call i64 @llvm.vscale()
+; BEFORE-SCALE-NEXT:    EMIT vp<[[VP5:%[0-9]+]]> = mul nuw vp<[[VP4]]>, ir<16>
+; BEFORE-SCALE-NEXT:    EMIT vp<%min.iters.check> = icmp ult vp<[[VP3]]>, vp<[[VP5]]>
+; BEFORE-SCALE-NEXT:    EMIT branch-on-cond vp<%min.iters.check>
+; BEFORE-SCALE-NEXT:  Successor(s): scalar.ph, vector.ph
+; BEFORE-SCALE-EMPTY:
+; BEFORE-SCALE-NEXT:  vector.ph:
+; BEFORE-SCALE-NEXT:  Successor(s): vector loop
+; BEFORE-SCALE-EMPTY:
+; BEFORE-SCALE-NEXT:  <x1> vector loop: {
+; BEFORE-SCALE-NEXT:  vp<[[VP7:%[0-9]+]]> = CANONICAL-IV
+; BEFORE-SCALE-EMPTY:
+; BEFORE-SCALE-NEXT:    vector.body:
+; BEFORE-SCALE-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
+; BEFORE-SCALE-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
+; BEFORE-SCALE-NEXT:      vp<[[VP9:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; BEFORE-SCALE-NEXT:      WIDEN ir<%ld> = load vp<[[VP9]]>
+; BEFORE-SCALE-NEXT:      WIDEN ir<%add> = add ir<%ld>, ir<1>
+; BEFORE-SCALE-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; BEFORE-SCALE-NEXT:      WIDEN store vp<[[VP10]]>, ir<%add>
+; BEFORE-SCALE-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
+; BEFORE-SCALE-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; BEFORE-SCALE-NEXT:    No successors
+; BEFORE-SCALE-NEXT:  }
+; BEFORE-SCALE-NEXT:  Successor(s): middle.block
+; BEFORE-SCALE-EMPTY:
+; BEFORE-SCALE-NEXT:  middle.block:
+;
+; AFTER-SCALE-LABEL: VPlan for loop in 'i64_load_store'
+; AFTER-SCALE:  VPlan 'Initial VPlan for VF={vscale x 4},UF>=1' {
+; AFTER-SCALE-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; AFTER-SCALE-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; AFTER-SCALE-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; AFTER-SCALE-NEXT:  vp<[[VP3:%[0-9]+]]> = original trip-count
+; AFTER-SCALE-EMPTY:
+; AFTER-SCALE-NEXT:  ir-bb<entry>:
+; AFTER-SCALE-NEXT:    EMIT vp<[[VP3]]> = EXPAND SCEV (1 umax %n)
+; AFTER-SCALE-NEXT:    EMIT-SCALAR vp<[[VP4:%[0-9]+]]> = call i64 @llvm.vscale()
+; AFTER-SCALE-NEXT:    EMIT vp<[[VP5:%[0-9]+]]> = mul nuw vp<[[VP4]]>, ir<16>
+; AFTER-SCALE-NEXT:    EMIT vp<%min.iters.check> = icmp ult vp<[[VP3]]>, vp<[[VP5]]>
+; AFTER-SCALE-NEXT:    EMIT branch-on-cond vp<%min.iters.check>
+; AFTER-SCALE-NEXT:  Successor(s): scalar.ph, vector.ph
+; AFTER-SCALE-EMPTY:
+; AFTER-SCALE-NEXT:  vector.ph:
+; AFTER-SCALE-NEXT:  Successor(s): vector loop
+; AFTER-SCALE-EMPTY:
+; AFTER-SCALE-NEXT:  <x1> vector loop: {
+; AFTER-SCALE-NEXT:  vp<[[VP7:%[0-9]+]]> = CANONICAL-IV
+; AFTER-SCALE-EMPTY:
+; AFTER-SCALE-NEXT:    vector.body:
+; AFTER-SCALE-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
+; AFTER-SCALE-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
+; AFTER-SCALE-NEXT:      vp<[[VP9:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-SCALE-NEXT:      WIDEN ir<%ld> = load x2 vp<[[VP9]]>
+; AFTER-SCALE-NEXT:      WIDEN ir<%add> = add ir<%ld>, ir<1>
+; AFTER-SCALE-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-SCALE-NEXT:      WIDEN store x2 vp<[[VP10]]>, ir<%add>
+; AFTER-SCALE-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
+; AFTER-SCALE-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; AFTER-SCALE-NEXT:    No successors
+; AFTER-SCALE-NEXT:  }
+; AFTER-SCALE-NEXT:  Successor(s): middle.block
+; AFTER-SCALE-EMPTY:
+; AFTER-SCALE-NEXT:  middle.block:
+;
+; AFTER-UNROLL-LABEL: VPlan for loop in 'i64_load_store'
+; AFTER-UNROLL:  VPlan 'Initial VPlan for VF={vscale x 4},UF={4}' {
+; AFTER-UNROLL-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; AFTER-UNROLL-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; AFTER-UNROLL-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; AFTER-UNROLL-NEXT:  vp<[[VP3:%[0-9]+]]> = original trip-count
+; AFTER-UNROLL-EMPTY:
+; AFTER-UNROLL-NEXT:  ir-bb<entry>:
+; AFTER-UNROLL-NEXT:    EMIT vp<[[VP3]]> = EXPAND SCEV (1 umax %n)
+; AFTER-UNROLL-NEXT:    EMIT-SCALAR vp<[[VP4:%[0-9]+]]> = call i64 @llvm.vscale()
+; AFTER-UNROLL-NEXT:    EMIT vp<[[VP5:%[0-9]+]]> = mul nuw vp<[[VP4]]>, ir<16>
+; AFTER-UNROLL-NEXT:    EMIT vp<%min.iters.check> = icmp ult vp<[[VP3]]>, vp<[[VP5]]>
+; AFTER-UNROLL-NEXT:    EMIT branch-on-cond vp<%min.iters.check>
+; AFTER-UNROLL-NEXT:  Successor(s): scalar.ph, vector.ph
+; AFTER-UNROLL-EMPTY:
+; AFTER-UNROLL-NEXT:  vector.ph:
+; AFTER-UNROLL-NEXT:  Successor(s): vector loop
+; AFTER-UNROLL-EMPTY:
+; AFTER-UNROLL-NEXT:  <x1> vector loop: {
+; AFTER-UNROLL-NEXT:  vp<[[VP7:%[0-9]+]]> = CANONICAL-IV
+; AFTER-UNROLL-EMPTY:
+; AFTER-UNROLL-NEXT:    vector.body:
+; AFTER-UNROLL-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
+; AFTER-UNROLL-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
+; AFTER-UNROLL-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-UNROLL-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP9]]>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%ld> = load x2 vp<[[VP10]]>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%ld>.1 = load x2 vp<[[VP11]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = extract-vector-for-part ir<%ld>, ir<0>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = extract-vector-for-part ir<%ld>, ir<1>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP14:%[0-9]+]]> = extract-vector-for-part ir<%ld>.1, ir<0>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP15:%[0-9]+]]> = extract-vector-for-part ir<%ld>.1, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add> = add vp<[[VP12]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.1 = add vp<[[VP13]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.2 = add vp<[[VP14]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.3 = add vp<[[VP15]]>, ir<1>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP16:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
+; AFTER-UNROLL-NEXT:      vp<[[VP17:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-UNROLL-NEXT:      vp<[[VP18:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP16]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP19:%[0-9]+]]> = concat-vector-parts ir<%add>, ir<%add>.1
+; AFTER-UNROLL-NEXT:      WIDEN store x2 vp<[[VP17]]>, vp<[[VP19]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP20:%[0-9]+]]> = concat-vector-parts ir<%add>.2, ir<%add>.3
+; AFTER-UNROLL-NEXT:      WIDEN store x2 vp<[[VP18]]>, vp<[[VP20]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
+; AFTER-UNROLL-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; AFTER-UNROLL-NEXT:    No successors
+; AFTER-UNROLL-NEXT:  }
+; AFTER-UNROLL-NEXT:  Successor(s): middle.block
+; AFTER-UNROLL-EMPTY:
+; AFTER-UNROLL-NEXT:  middle.block:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %ptr = getelementptr inbounds i64, ptr %x, i64 %iv
+  %ld = load i64, ptr %ptr, align 8
+  %add = add i64 %ld, 1
+  store i64 %add, ptr %ptr, align 8
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
index 5fd844e186e44..cb492acc29e1e 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
@@ -71,6 +71,7 @@
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] printOptimizedVPlan
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::addMinimumIterationCheck
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::replaceWideCanonicalIVWithWideIV
+; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::scaleMemoryAccessesByUF
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::unrollByUF
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::materializePacksAndUnpacks
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::materializeBroadcasts

>From 3003f56005efaa9cc987c5e5420a7b2dc999a71b Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Thu, 20 Aug 2026 19:33:58 +0000
Subject: [PATCH 02/20] Rebase fixups

---
 .../AArch64/multi-vector-mem-ops.ll           | 20 ++++++++-----------
 1 file changed, 8 insertions(+), 12 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
index f20e7aff70ef7..416eea04db1cb 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -17,8 +17,7 @@ define void @mixed_i64_i32_accesses(
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
 ; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
-; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 2
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
@@ -57,7 +56,7 @@ define void @mixed_i64_i32_accesses(
 ; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP30]], <vscale x 4 x i32> [[TMP27]], i64 8)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP31]], <vscale x 4 x i32> [[TMP28]], i64 12)
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP32]], ptr [[TMP20]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP33]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
@@ -98,8 +97,7 @@ define void @mixed_i32_more_frequent_than_i64(
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
 ; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
-; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 2
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
@@ -153,7 +151,7 @@ define void @mixed_i32_more_frequent_than_i64(
 ; UNMASKED-SVE2P1-NEXT:    [[TMP44:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP43]], <vscale x 4 x i32> [[TMP40]], i64 8)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP45:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP44]], <vscale x 4 x i32> [[TMP41]], i64 12)
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP45]], ptr [[TMP33]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP46:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP46]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
@@ -257,8 +255,7 @@ define i64 @i64_sum_reduction_scaled_partial_reduce(ptr noalias %a, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 3
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
@@ -277,7 +274,7 @@ define i64 @i64_sum_reduction_scaled_partial_reduce(ptr noalias %a, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_PHI1]], [[TMP6]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP11]] = add <vscale x 2 x i64> [[VEC_PHI2]], [[TMP7]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12]] = add <vscale x 2 x i64> [[VEC_PHI3]], [[TMP8]]
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
@@ -308,8 +305,7 @@ define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold)
 ; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 3
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[THRESHOLD]], i64 0
 ; UNMASKED-SVE2P1-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
@@ -350,7 +346,7 @@ define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP29]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP9]], <vscale x 2 x i64> [[VEC_PHI1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP30]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP10]], <vscale x 2 x i64> [[VEC_PHI2]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP31]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP11]], <vscale x 2 x i64> [[VEC_PHI3]]
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP32]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:

>From 6dece1edd0a5a90d6ebebe4f3951a5611f4e18db Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Wed, 2 Sep 2026 10:46:03 +0000
Subject: [PATCH 03/20] Fixups

---
 llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp | 3 +--
 1 file changed, 1 insertion(+), 2 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 93a1b3492619e..e4848dc714f06 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -5971,8 +5971,7 @@ AArch64TTIImpl::getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataTy,
   if (!ST->enableSubRegLiveness())
     return 1;
 
-  if ((Opcode != Instruction::Load && Opcode != Instruction::Store) ||
-      !ST->hasSVE2p1() || !VF.isScalable() || !isPowerOf2_32(UF))
+  if (!ST->hasSVE2p1() || !VF.isScalable() || !isPowerOf2_32(UF))
     return 1;
 
   unsigned VectorWidth = VF.getKnownMinValue() * DL.getTypeSizeInBits(DataTy);

>From a04bbb2f9e183aff49e630b81988ed8aa27116bd Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Wed, 2 Sep 2026 13:43:35 +0000
Subject: [PATCH 04/20] Check for extending loads

---
 .../llvm/Analysis/TargetTransformInfo.h       | 10 +--
 .../llvm/Analysis/TargetTransformInfoImpl.h   |  8 +--
 llvm/lib/Analysis/TargetTransformInfo.cpp     |  4 +-
 .../AArch64/AArch64TargetTransformInfo.cpp    | 14 ++--
 .../AArch64/AArch64TargetTransformInfo.h      |  7 +-
 .../Transforms/Vectorize/VPlanTransforms.cpp  | 10 ++-
 .../AArch64/multi-vector-mem-ops.ll           | 70 +++++++++++++++++++
 7 files changed, 104 insertions(+), 19 deletions(-)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index 434e7bbf980d9..95c6d9507370b 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1003,12 +1003,14 @@ class TargetTransformInfo {
 
   /// Return the preferred multiple of VF to use for a contiguous load/store.
   /// Returning 1 leaves the operation at VF. The returned value must divide UF.
+  /// \p CastHint is non-null if the stored or loaded valued is produced by or
+  /// consumed a cast instruction respectively.
   ///
   /// \p Opcode must be either Instruction::Load or Instruction::Store.
-  LLVM_ABI unsigned
-  getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataType,
-                                    ElementCount VF, unsigned UF,
-                                    bool IsMasked = false) const;
+  LLVM_ABI unsigned getPreferredVFMultipleForMemoryOp(
+      unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
+      bool IsMasked = false,
+      std::optional<Instruction::CastOps> CastHint = std::nullopt) const;
 
   /// Return true if we should be enabling ordered reductions for the target.
   LLVM_ABI bool enableOrderedReductions() const;
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 01006cfce4d08..f70257b10fbac 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -429,11 +429,9 @@ class LLVM_ABI TargetTransformInfoImplBase {
     return false;
   }
 
-  virtual unsigned getPreferredVFMultipleForMemoryOp(unsigned Opcode,
-                                                     Type *DataType,
-                                                     ElementCount VF,
-                                                     unsigned UF,
-                                                     bool IsMasked) const {
+  virtual unsigned getPreferredVFMultipleForMemoryOp(
+      unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
+      bool IsMasked, std::optional<Instruction::CastOps> CastHint) const {
     return 1;
   }
 
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 33cb290209afa..ad2270fbd6e10 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -556,9 +556,9 @@ bool TargetTransformInfo::isLegalStridedLoadStore(Type *DataType,
 
 unsigned TargetTransformInfo::getPreferredVFMultipleForMemoryOp(
     unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
-    bool IsMasked) const {
+    bool IsMasked, std::optional<Instruction::CastOps> CastHint) const {
   return TTIImpl->getPreferredVFMultipleForMemoryOp(Opcode, DataType, VF, UF,
-                                                    IsMasked);
+                                                    IsMasked, CastHint);
 }
 
 bool TargetTransformInfo::isLegalInterleavedAccessType(
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index e4848dc714f06..1c71ec719a1dd 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -5959,15 +5959,21 @@ bool AArch64TTIImpl::isLegalSpeculativeLoad(Type *DataType,
          Size.getFixedValue() <= 16;
 }
 
-unsigned
-AArch64TTIImpl::getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataTy,
-                                                  ElementCount VF, unsigned UF,
-                                                  bool IsMasked) const {
+unsigned AArch64TTIImpl::getPreferredVFMultipleForMemoryOp(
+    unsigned Opcode, Type *DataTy, ElementCount VF, unsigned UF, bool IsMasked,
+    std::optional<Instruction::CastOps> CastHint) const {
   assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
          "expected load/store opcode");
   if (IsMasked)
     return 1; // TODO: Support masked multi-vector loads/stores.
 
+  // Conservatively, avoid using multi-vector loads when it's possible we could
+  // use extending loads instead. Note: We can ignore stores as we only use
+  // truncating stores when the store vector-width is < a full SVE vector.
+  if (Opcode == Instruction::Load &&
+      (CastHint == Instruction::ZExt || CastHint == Instruction::SExt))
+    return 1;
+
   if (!ST->enableSubRegLiveness())
     return 1;
 
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index a5513114bd112..aa1d059406c8c 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -282,9 +282,10 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
   bool isLegalSpeculativeLoad(Type *DataType,
                               unsigned AddressSpace) const override;
 
-  unsigned getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataType,
-                                             ElementCount VF, unsigned UF,
-                                             bool IsMasked) const override;
+  unsigned getPreferredVFMultipleForMemoryOp(
+      unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
+      bool IsMasked,
+      std::optional<Instruction::CastOps> CastHint) const override;
 
   void getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
                                TTI::UnrollingPreferences &UP,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index b670c7be7e3de..155ad1e2dc5f6 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4211,8 +4211,16 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
       unsigned Opcode = isa<VPWidenLoadRecipe>(MemOp->getAsRecipe())
                             ? Instruction::Load
                             : Instruction::Store;
+
+      std::optional<Instruction::CastOps> CastHint;
+      VPUser *MaybeCast = Opcode == Instruction::Store
+                              ? StoredValue->getDefiningRecipe()
+                              : R.getVPSingleValue()->getSingleUser();
+      if (auto *Cast = dyn_cast_if_present<VPWidenCastRecipe>(MaybeCast))
+        CastHint = Cast->getOpcode();
+
       unsigned ScaleFactor = TTI.getPreferredVFMultipleForMemoryOp(
-          Opcode, AccessType, VF, UF, /*IsMasked=*/false);
+          Opcode, AccessType, VF, UF, /*IsMasked=*/false, CastHint);
       assert((ScaleFactor != 0 && UF % ScaleFactor == 0) &&
              "ScaleFactor must divide UF");
       MemOp->setVFMultiple(ScaleFactor);
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
index 416eea04db1cb..27aa10c8d6023 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -368,3 +368,73 @@ loop:
 exit:
   ret i64 %select.data
 }
+
+; Negative test: We should not use a wide load when the result of the load is extended.
+; In this case, it's better to use SVE extending loads (rather than a multi-vector load).
+define void @extending_load(ptr noalias %dst, ptr %src, i64 %n) {
+; UNMASKED-SVE2P1-LABEL: define void @extending_load(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[DST:%.*]], ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ITER_CHECK:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw nsw i64 [[TMP2]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = mul nuw nsw i64 [[TMP2]], 3
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP3]], i64 [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP3]], i64 [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP3]], i64 [[TMP5]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP3]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 4 x i32>, ptr [[TMP6]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 4 x i32>, ptr [[TMP7]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD4:%.*]] = load <vscale x 4 x i32>, ptr [[TMP8]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = sext <vscale x 4 x i32> [[WIDE_LOAD]] to <vscale x 4 x i64>
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = sext <vscale x 4 x i32> [[WIDE_LOAD2]] to <vscale x 4 x i64>
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = sext <vscale x 4 x i32> [[WIDE_LOAD3]] to <vscale x 4 x i64>
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = sext <vscale x 4 x i32> [[WIDE_LOAD4]] to <vscale x 4 x i64>
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = mul nsw <vscale x 4 x i64> [[TMP9]], splat (i64 42)
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = mul nsw <vscale x 4 x i64> [[TMP10]], splat (i64 42)
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = mul nsw <vscale x 4 x i64> [[TMP11]], splat (i64 42)
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = mul nsw <vscale x 4 x i64> [[TMP12]], splat (i64 42)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = getelementptr inbounds nuw i64, ptr [[DST]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = getelementptr inbounds nuw i64, ptr [[TMP17]], i64 [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP13]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP19]], <vscale x 4 x i64> [[TMP14]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP20]], ptr [[TMP17]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP15]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP21]], <vscale x 4 x i64> [[TMP16]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP22]], ptr [[TMP18]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP23]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP14:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %src.ptr = getelementptr inbounds nuw i32, ptr %src, i64 %iv
+  %src.val = load i32, ptr %src.ptr, align 4
+  %conv = sext i32 %src.val to i64
+  %mul = mul nsw i64 %conv, 42
+  %dst.ptr = getelementptr inbounds nuw i64, ptr %dst, i64 %iv
+  store i64 %mul, ptr %dst.ptr, align 8
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exit.cmp = icmp eq i64 %iv.next, %n
+  br i1 %exit.cmp, label %exit, label %loop
+
+exit:
+  ret void
+}

>From dbabd2d32717fe45747736b496ccdfbe61df964e Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Thu, 3 Sep 2026 16:04:44 +0000
Subject: [PATCH 05/20] Fixups

---
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp   |  2 +-
 llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h | 12 ------------
 llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp |  9 ++-------
 3 files changed, 3 insertions(+), 20 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 9cdae4b9c136e..98a8bc0ec1392 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5753,7 +5753,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
   // TODO: Move to VPlan transform stage once the transition to the VPlan-based
   // cost model is complete for better cost estimates.
   RUN_VPLAN_PASS(VPlanTransforms::scaleMemoryAccessesByUF, BestVPlan, BestVF,
-                 BestUF, CM.TTI);
+                 BestUF, TTI);
   RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
   RUN_VPLAN_PASS(VPlanTransforms::materializePacksAndUnpacks, BestVPlan);
   RUN_VPLAN_PASS(VPlanTransforms::materializeBroadcasts, BestVPlan);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
index 22f35f1b96ecd..16addfbf1b160 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
@@ -293,7 +293,6 @@ struct Recipe_match {
     if constexpr (std::is_same_v<RecipeTy, VPScalarIVStepsRecipe> ||
                   std::is_same_v<RecipeTy, VPDerivedIVRecipe> ||
                   std::is_same_v<RecipeTy, VPVectorEndPointerRecipe> ||
-                  std::is_same_v<RecipeTy, VPVectorPointerRecipe> ||
                   std::is_same_v<RecipeTy, VPWidenLoadRecipe> ||
                   std::is_same_v<RecipeTy, VPWidenStoreRecipe>)
       return DefR;
@@ -1052,17 +1051,6 @@ m_MaskedStore(const Addr_t &Addr, const Val_t &Val, const Mask_t &Mask) {
   return Store_match<Addr_t, Val_t, Mask_t>(Addr, Val, Mask);
 }
 
-template <typename Op0_t, typename Op1_t>
-using VectorPointerRecipe_match =
-    Recipe_match<std::tuple<Op0_t, Op1_t>, 0,
-                 /*Commutative*/ false, VPVectorPointerRecipe>;
-
-template <typename Op0_t, typename Op1_t>
-VectorPointerRecipe_match<Op0_t, Op1_t> m_VecPtr(const Op0_t &Op0,
-                                                 const Op1_t &Op1) {
-  return VectorPointerRecipe_match<Op0_t, Op1_t>(Op0, Op1);
-}
-
 template <typename Op0_t>
 using VPWidenLoadRecipe_match =
     Recipe_match<std::tuple<Op0_t>, 0,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 155ad1e2dc5f6..90ad196509be6 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4188,14 +4188,9 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
   for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
            vp_depth_first_deep(Plan.getVectorLoopRegion()->getEntry()))) {
     for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
-      uint64_t Stride;
       VPValue *StoredValue = nullptr;
-      auto m_ConstantStrideVecPtr =
-          m_VecPtr(m_VPValue(), m_ConstantInt(Stride));
-      if ((!match(&R, m_WidenLoad(m_ConstantStrideVecPtr)) &&
-           !match(&R, m_WidenStore(m_ConstantStrideVecPtr,
-                                   m_VPValue(StoredValue)))) ||
-          Stride != 1)
+      if (!match(&R, m_WidenLoad(m_VPValue())) &&
+          !match(&R, m_WidenStore(m_VPValue(), m_VPValue(StoredValue))))
         continue;
 
       auto *MemOp = cast<VPWidenMemoryRecipe>(&R);

>From e916c7597845bf88f8e3f56eb0eea9554c88e154 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 7 Sep 2026 11:20:54 +0000
Subject: [PATCH 06/20] Fixups

---
 .../Transforms/Vectorize/VPlanPatternMatch.h  |  12 +
 .../Transforms/Vectorize/VPlanTransforms.cpp  |  22 +-
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp |  12 +-
 ...-vector-mem-ops-non-power-of-two-unroll.ll |  74 +++
 .../AArch64/multi-vector-mem-ops.ll           | 445 ++++++++++++------
 5 files changed, 411 insertions(+), 154 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll

diff --git a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
index 16addfbf1b160..22f35f1b96ecd 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
@@ -293,6 +293,7 @@ struct Recipe_match {
     if constexpr (std::is_same_v<RecipeTy, VPScalarIVStepsRecipe> ||
                   std::is_same_v<RecipeTy, VPDerivedIVRecipe> ||
                   std::is_same_v<RecipeTy, VPVectorEndPointerRecipe> ||
+                  std::is_same_v<RecipeTy, VPVectorPointerRecipe> ||
                   std::is_same_v<RecipeTy, VPWidenLoadRecipe> ||
                   std::is_same_v<RecipeTy, VPWidenStoreRecipe>)
       return DefR;
@@ -1051,6 +1052,17 @@ m_MaskedStore(const Addr_t &Addr, const Val_t &Val, const Mask_t &Mask) {
   return Store_match<Addr_t, Val_t, Mask_t>(Addr, Val, Mask);
 }
 
+template <typename Op0_t, typename Op1_t>
+using VectorPointerRecipe_match =
+    Recipe_match<std::tuple<Op0_t, Op1_t>, 0,
+                 /*Commutative*/ false, VPVectorPointerRecipe>;
+
+template <typename Op0_t, typename Op1_t>
+VectorPointerRecipe_match<Op0_t, Op1_t> m_VecPtr(const Op0_t &Op0,
+                                                 const Op1_t &Op1) {
+  return VectorPointerRecipe_match<Op0_t, Op1_t>(Op0, Op1);
+}
+
 template <typename Op0_t>
 using VPWidenLoadRecipe_match =
     Recipe_match<std::tuple<Op0_t>, 0,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 90ad196509be6..d3627ac3aa362 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4186,16 +4186,20 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
     return;
 
   for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
-           vp_depth_first_deep(Plan.getVectorLoopRegion()->getEntry()))) {
+           vp_depth_first_shallow(Plan.getVectorLoopRegion()->getEntry()))) {
     for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
+      uint64_t Stride;
       VPValue *StoredValue = nullptr;
-      if (!match(&R, m_WidenLoad(m_VPValue())) &&
-          !match(&R, m_WidenStore(m_VPValue(), m_VPValue(StoredValue))))
+      auto m_ConstantStrideVecPtr =
+          m_VecPtr(m_VPValue(), m_ConstantInt(Stride));
+      if ((!match(&R, m_WidenLoad(m_ConstantStrideVecPtr)) &&
+           !match(&R, m_WidenStore(m_ConstantStrideVecPtr,
+                                   m_VPValue(StoredValue)))) ||
+          Stride != 1)
         continue;
 
       auto *MemOp = cast<VPWidenMemoryRecipe>(&R);
-      if (!MemOp->isConsecutive())
-        continue;
+      assert(MemOp->isConsecutive() && "Expected consecutive load/store");
 
       // TODO: Support masked loads/stores. This requires widening the header
       // mask to the same factor as the memory operation.
@@ -4214,11 +4218,11 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
       if (auto *Cast = dyn_cast_if_present<VPWidenCastRecipe>(MaybeCast))
         CastHint = Cast->getOpcode();
 
-      unsigned ScaleFactor = TTI.getPreferredVFMultipleForMemoryOp(
+      unsigned VFMultiple = TTI.getPreferredVFMultipleForMemoryOp(
           Opcode, AccessType, VF, UF, /*IsMasked=*/false, CastHint);
-      assert((ScaleFactor != 0 && UF % ScaleFactor == 0) &&
-             "ScaleFactor must divide UF");
-      MemOp->setVFMultiple(ScaleFactor);
+      assert((VFMultiple != 0 && UF % VFMultiple == 0) &&
+             "VFMultiple must divide UF");
+      MemOp->setVFMultiple(VFMultiple);
     }
   }
 }
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index e3705bd900f8c..5ab9c6a2e7dfd 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -295,7 +295,8 @@ void UnrollState::unrollHeaderPHIByUF(VPHeaderPHIRecipe *R,
 
 void UnrollState::unrollMemOpWithVFMultiple(VPRecipeBase &R,
                                             unsigned VFMultiple) {
-  assert(VFMultiple > 1 && UF % VFMultiple == 0);
+  assert(VFMultiple > 1 && UF % VFMultiple == 0 &&
+         "expected VFMultiple to divide UF");
   SmallVector<VPRecipeBase *, 4> Groups(UF / VFMultiple, nullptr);
   Groups[0] = &R;
 
@@ -361,10 +362,11 @@ void UnrollState::unrollRecipeByUF(VPRecipeBase &R) {
     }
   }
 
-  if (auto *WidenMem = dyn_cast<VPWidenMemoryRecipe>(&R);
-      WidenMem && WidenMem->getVFMultiple() > 1) {
-    unrollMemOpWithVFMultiple(R, WidenMem->getVFMultiple());
-    return;
+  if (auto *WidenMem = dyn_cast<VPWidenMemoryRecipe>(&R)) {
+    if (WidenMem && WidenMem->getVFMultiple() > 1) {
+      unrollMemOpWithVFMultiple(R, WidenMem->getVFMultiple());
+      return;
+    }
   }
 
   if (auto *RepR = dyn_cast<VPReplicateRecipe>(&R)) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll
new file mode 100644
index 0000000000000..ac9e55117364e
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll
@@ -0,0 +1,74 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^middle.block:" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-interleave=3 -aarch64-enable-subreg-liveness-tracking -sve-tail-folding=disabled %s | FileCheck %s --check-prefix=CHECK
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; Tests `scaleMemoryAccessesByUF` with a non-power-of-two unroll factor.
+; On AArch64, this should be rejected (which results in `vscale x 4` loads/stores).
+
+define void @mixed_i64_i32_accesses(ptr noalias %x, ptr noalias %y, i64 %n) {
+; CHECK-LABEL: define void @mixed_i64_i32_accesses(
+; CHECK-SAME: ptr noalias [[X:%.*]], ptr noalias [[Y:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP2:%.*]] = mul nuw i64 [[TMP1]], 12
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP5:%.*]] = shl nuw nsw i64 [[TMP3]], 1
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i64, ptr [[TMP4]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i64, ptr [[TMP4]], i64 [[TMP5]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 4 x i64>, ptr [[TMP4]], align 8
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 4 x i64>, ptr [[TMP6]], align 8
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 4 x i64>, ptr [[TMP7]], align 8
+; CHECK-NEXT:    [[TMP8:%.*]] = add <vscale x 4 x i64> [[WIDE_LOAD]], splat (i64 1)
+; CHECK-NEXT:    [[TMP9:%.*]] = add <vscale x 4 x i64> [[WIDE_LOAD1]], splat (i64 1)
+; CHECK-NEXT:    [[TMP10:%.*]] = add <vscale x 4 x i64> [[WIDE_LOAD2]], splat (i64 1)
+; CHECK-NEXT:    store <vscale x 4 x i64> [[TMP8]], ptr [[TMP4]], align 8
+; CHECK-NEXT:    store <vscale x 4 x i64> [[TMP9]], ptr [[TMP6]], align 8
+; CHECK-NEXT:    store <vscale x 4 x i64> [[TMP10]], ptr [[TMP7]], align 8
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[TMP11]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[TMP11]], i64 [[TMP5]]
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 4 x i32>, ptr [[TMP11]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD4:%.*]] = load <vscale x 4 x i32>, ptr [[TMP12]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD5:%.*]] = load <vscale x 4 x i32>, ptr [[TMP13]], align 4
+; CHECK-NEXT:    [[TMP14:%.*]] = add <vscale x 4 x i32> [[WIDE_LOAD3]], splat (i32 1)
+; CHECK-NEXT:    [[TMP15:%.*]] = add <vscale x 4 x i32> [[WIDE_LOAD4]], splat (i32 1)
+; CHECK-NEXT:    [[TMP16:%.*]] = add <vscale x 4 x i32> [[WIDE_LOAD5]], splat (i32 1)
+; CHECK-NEXT:    store <vscale x 4 x i32> [[TMP14]], ptr [[TMP11]], align 4
+; CHECK-NEXT:    store <vscale x 4 x i32> [[TMP15]], ptr [[TMP12]], align 4
+; CHECK-NEXT:    store <vscale x 4 x i32> [[TMP16]], ptr [[TMP13]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %px = getelementptr inbounds i64, ptr %x, i64 %iv
+  %vx = load i64, ptr %px, align 8
+  %ax = add i64 %vx, 1
+  store i64 %ax, ptr %px, align 8
+  %py = getelementptr inbounds i32, ptr %y, i64 %iv
+  %vy = load i32, ptr %py, align 4
+  %ay = add i32 %vy, 1
+  store i32 %ay, ptr %py, align 4
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
index 27aa10c8d6023..b90e29519e24e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -3,30 +3,30 @@
 
 target triple = "aarch64-unknown-linux-gnu"
 
-define void @mixed_i64_i32_accesses(
+define void @mixed_i64_i32_accesses(ptr noalias %x, ptr noalias %y, i64 %n) {
 ; UNMASKED-SVE2P1-LABEL: define void @mixed_i64_i32_accesses(
 ; UNMASKED-SVE2P1-SAME: ptr noalias [[X:%.*]], ptr noalias [[Y:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
-; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
-; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 4
-; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1-NEXT:  [[ITER_CHECK:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[UMAX]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
-; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 4
-; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[UMAX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[UMAX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[UMAX]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = shl nuw nsw i64 [[TMP3]], 1
-; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i64, ptr [[TMP5]], i64 [[TMP14]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 8 x i64>, ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw nsw i64 [[TMP2]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[TMP3]], i64 [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
 ; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 0)
@@ -35,33 +35,32 @@ define void @mixed_i64_i32_accesses(
 ; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <vscale x 4 x i64> [[TMP7]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <vscale x 4 x i64> [[TMP8]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add <vscale x 4 x i64> [[TMP9]], splat (i64 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP14]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP15]], ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP13]], i64 4)
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP17]], ptr [[TMP5]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP18]], <vscale x 4 x i64> [[TMP13]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP19]], ptr [[TMP15]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 16 x i32>, ptr [[TMP20]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = add <vscale x 4 x i32> [[TMP19]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = add <vscale x 4 x i32> [[TMP20]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = add <vscale x 4 x i32> [[TMP21]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = add <vscale x 4 x i32> [[TMP22]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = add <vscale x 4 x i32> [[TMP23]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = add <vscale x 4 x i32> [[TMP24]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP25]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP30]], <vscale x 4 x i32> [[TMP27]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP31]], <vscale x 4 x i32> [[TMP28]], i64 12)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP32]], ptr [[TMP20]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP33]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP23]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP27]], <vscale x 4 x i32> [[TMP24]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP28]], <vscale x 4 x i32> [[TMP25]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP30]], ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
-  ptr noalias %x, ptr noalias %y, i64 %n) {
 entry:
   br label %loop
 
@@ -83,80 +82,79 @@ exit:
   ret void
 }
 
-define void @mixed_i32_more_frequent_than_i64(
+define void @mixed_i32_more_frequent_than_i64(ptr noalias %x, ptr noalias %y, ptr noalias %z, i64 %n) {
 ; UNMASKED-SVE2P1-LABEL: define void @mixed_i32_more_frequent_than_i64(
 ; UNMASKED-SVE2P1-SAME: ptr noalias [[X:%.*]], ptr noalias [[Y:%.*]], ptr noalias [[Z:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
-; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 4
-; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1-NEXT:  [[ITER_CHECK:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[UMAX]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
-; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 4
-; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[UMAX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[UMAX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[UMAX]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = shl nuw nsw i64 [[TMP3]], 1
-; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i64, ptr [[TMP5]], i64 [[TMP14]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 8 x i64>, ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw nsw i64 [[TMP2]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[TMP3]], i64 [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
 ; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD3]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD3]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 4)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = add <vscale x 4 x i64> [[TMP6]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <vscale x 4 x i64> [[TMP7]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <vscale x 4 x i64> [[TMP8]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add <vscale x 4 x i64> [[TMP9]], splat (i64 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP14]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP15]], ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP13]], i64 4)
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP17]], ptr [[TMP5]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP18]], <vscale x 4 x i64> [[TMP13]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP19]], ptr [[TMP15]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 16 x i32>, ptr [[TMP20]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = add <vscale x 4 x i32> [[TMP19]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = add <vscale x 4 x i32> [[TMP20]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = add <vscale x 4 x i32> [[TMP21]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = add <vscale x 4 x i32> [[TMP22]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = add <vscale x 4 x i32> [[TMP23]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = add <vscale x 4 x i32> [[TMP24]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP25]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP30]], <vscale x 4 x i32> [[TMP27]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP31]], <vscale x 4 x i32> [[TMP28]], i64 12)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP32]], ptr [[TMP20]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = getelementptr inbounds i32, ptr [[Z]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 16 x i32>, ptr [[TMP33]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[TMP34:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP35:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP36:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP37:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP23]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP27]], <vscale x 4 x i32> [[TMP24]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP28]], <vscale x 4 x i32> [[TMP25]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP30]], ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = getelementptr inbounds i32, ptr [[Z]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD4:%.*]] = load <vscale x 16 x i32>, ptr [[TMP31]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP34:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP35:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP36:%.*]] = add <vscale x 4 x i32> [[TMP32]], splat (i32 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP37:%.*]] = add <vscale x 4 x i32> [[TMP33]], splat (i32 2)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP38:%.*]] = add <vscale x 4 x i32> [[TMP34]], splat (i32 2)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP39:%.*]] = add <vscale x 4 x i32> [[TMP35]], splat (i32 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP40:%.*]] = add <vscale x 4 x i32> [[TMP36]], splat (i32 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP41:%.*]] = add <vscale x 4 x i32> [[TMP37]], splat (i32 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP42:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP38]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP43:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP42]], <vscale x 4 x i32> [[TMP39]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP44:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP43]], <vscale x 4 x i32> [[TMP40]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP45:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP44]], <vscale x 4 x i32> [[TMP41]], i64 12)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP45]], ptr [[TMP33]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP46:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP46]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP40:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP36]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP41:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP40]], <vscale x 4 x i32> [[TMP37]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP42:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP41]], <vscale x 4 x i32> [[TMP38]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP43:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP42]], <vscale x 4 x i32> [[TMP39]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP43]], ptr [[TMP31]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP44:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP44]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
-  ptr noalias %x, ptr noalias %y, ptr noalias %z, i64 %n) {
 entry:
   br label %loop
 
@@ -182,7 +180,7 @@ exit:
   ret void
 }
 
-define void @first_order_recurrence_i64_scaled_load_and_store(
+define void @first_order_recurrence_i64_scaled_load_and_store(ptr noalias %src, ptr noalias %dst) {
 ; UNMASKED-SVE2P1-LABEL: define void @first_order_recurrence_i64_scaled_load_and_store(
 ; UNMASKED-SVE2P1-SAME: ptr noalias [[SRC:%.*]], ptr noalias [[DST:%.*]]) #[[ATTR0]] {
 ; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
@@ -224,7 +222,6 @@ define void @first_order_recurrence_i64_scaled_load_and_store(
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
-  ptr noalias %src, ptr noalias %dst) {
 entry:
   %first = load i64, ptr %src, align 8
   br label %loop
@@ -260,23 +257,23 @@ define i64 @i64_sum_reduction_scaled_partial_reduce(ptr noalias %a, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP11:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP12:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP4]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
-; UNMASKED-SVE2P1-NEXT:    [[TMP9]] = add <vscale x 2 x i64> [[VEC_PHI]], [[TMP5]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_PHI1]], [[TMP6]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP11]] = add <vscale x 2 x i64> [[VEC_PHI2]], [[TMP7]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP12]] = add <vscale x 2 x i64> [[VEC_PHI3]], [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP11:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8]] = add <vscale x 2 x i64> [[VEC_PHI]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP9]] = add <vscale x 2 x i64> [[VEC_PHI1]], [[TMP5]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_PHI2]], [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP11]] = add <vscale x 2 x i64> [[VEC_PHI3]], [[TMP7]]
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -312,43 +309,43 @@ define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold)
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP28:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP29:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP30:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP31:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP27:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP28:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP29:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP30:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP23:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP24:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP25:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP26:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP27:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i64, ptr [[DATA]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP7]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i64, ptr [[DATA]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP6]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP7]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP8]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP9]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP10]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP11]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = freeze <vscale x 2 x i1> [[TMP11]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = freeze <vscale x 2 x i1> [[TMP12]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = freeze <vscale x 2 x i1> [[TMP13]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = or <vscale x 2 x i1> [[TMP16]], [[TMP17]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = freeze <vscale x 2 x i1> [[TMP14]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = or <vscale x 2 x i1> [[TMP18]], [[TMP19]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = freeze <vscale x 2 x i1> [[TMP15]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = or <vscale x 2 x i1> [[TMP20]], [[TMP21]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call i1 @llvm.vector.reduce.or.nxv2i1(<vscale x 2 x i1> [[TMP22]])
-; UNMASKED-SVE2P1-NEXT:    [[TMP24]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP12]], <vscale x 2 x i1> [[TMP3]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP25]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP13]], <vscale x 2 x i1> [[TMP4]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP26]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP14]], <vscale x 2 x i1> [[TMP5]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP27]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP15]], <vscale x 2 x i1> [[TMP6]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP28]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP8]], <vscale x 2 x i64> [[VEC_PHI]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP29]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP9]], <vscale x 2 x i64> [[VEC_PHI1]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP30]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP10]], <vscale x 2 x i64> [[VEC_PHI2]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP31]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP11]], <vscale x 2 x i64> [[VEC_PHI3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = or <vscale x 2 x i1> [[TMP15]], [[TMP16]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = freeze <vscale x 2 x i1> [[TMP13]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = or <vscale x 2 x i1> [[TMP17]], [[TMP18]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = freeze <vscale x 2 x i1> [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = or <vscale x 2 x i1> [[TMP19]], [[TMP20]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call i1 @llvm.vector.reduce.or.nxv2i1(<vscale x 2 x i1> [[TMP21]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP23]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP11]], <vscale x 2 x i1> [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP24]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP12]], <vscale x 2 x i1> [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP25]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP13]], <vscale x 2 x i1> [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP26]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP14]], <vscale x 2 x i1> [[TMP5]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP27]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP7]], <vscale x 2 x i64> [[VEC_PHI]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP28]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP8]], <vscale x 2 x i64> [[VEC_PHI1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP29]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP9]], <vscale x 2 x i64> [[VEC_PHI2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP30]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP10]], <vscale x 2 x i64> [[VEC_PHI3]]
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP32]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -438,3 +435,171 @@ loop:
 exit:
   ret void
 }
+
+; TODO: Support multi-vector operations with negative strides.
+define void @reverse_stride(ptr noalias %c, ptr noalias  %a, ptr noalias %b, i64 %n) {
+; UNMASKED-SVE2P1-LABEL: define void @reverse_stride(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[C:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ITER_CHECK:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; UNMASKED-SVE2P1-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[N]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = sub i64 [[TMP0]], [[SMIN]]
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP1]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP2]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP1]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP2]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP1]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP1]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = sub i64 [[N]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = sub i64 [[N]], [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i32, ptr [[A]], i64 [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = sub nuw nsw i64 [[TMP4]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = sub i64 0, [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 [[TMP9]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = sub i64 [[TMP9]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 [[TMP27]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = mul i64 -2, [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = sub i64 [[TMP29]], [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP34:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 [[TMP33]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP35:%.*]] = mul i64 -3, [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP36:%.*]] = sub i64 [[TMP35]], [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP37:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 [[TMP36]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = load <vscale x 4 x i32>, ptr [[TMP10]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = load <vscale x 4 x i32>, ptr [[TMP28]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = load <vscale x 4 x i32>, ptr [[TMP34]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = load <vscale x 4 x i32>, ptr [[TMP37]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds nuw i32, ptr [[B]], i64 [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[TMP15]], i64 [[TMP9]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP38:%.*]] = getelementptr inbounds i32, ptr [[TMP15]], i64 [[TMP27]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP39:%.*]] = getelementptr inbounds i32, ptr [[TMP15]], i64 [[TMP33]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP40:%.*]] = getelementptr inbounds i32, ptr [[TMP15]], i64 [[TMP36]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = load <vscale x 4 x i32>, ptr [[TMP16]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = load <vscale x 4 x i32>, ptr [[TMP38]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = load <vscale x 4 x i32>, ptr [[TMP39]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = load <vscale x 4 x i32>, ptr [[TMP40]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = add nsw <vscale x 4 x i32> [[TMP17]], [[TMP11]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = add nsw <vscale x 4 x i32> [[TMP18]], [[TMP12]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = add nsw <vscale x 4 x i32> [[TMP19]], [[TMP13]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = add nsw <vscale x 4 x i32> [[TMP20]], [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = getelementptr inbounds nuw i32, ptr [[C]], i64 [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = getelementptr inbounds i32, ptr [[TMP25]], i64 [[TMP9]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = getelementptr inbounds i32, ptr [[TMP25]], i64 [[TMP27]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP41:%.*]] = getelementptr inbounds i32, ptr [[TMP25]], i64 [[TMP33]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = getelementptr inbounds i32, ptr [[TMP25]], i64 [[TMP36]]
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 4 x i32> [[TMP21]], ptr [[TMP26]], align 4
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 4 x i32> [[TMP22]], ptr [[TMP30]], align 4
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 4 x i32> [[TMP23]], ptr [[TMP41]], align 4
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 4 x i32> [[TMP24]], ptr [[TMP32]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP17:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ %dec, %for.body ], [ %n, %entry ]
+  %a.ptr = getelementptr inbounds nuw i32, ptr %a, i64 %iv
+  %a.val = load i32, ptr %a.ptr, align 4
+  %b.ptr = getelementptr inbounds nuw i32, ptr %b, i64 %iv
+  %b.val = load i32, ptr %b.ptr, align 4
+  %add = add nsw i32 %b.val, %a.val
+  %c.ptr = getelementptr inbounds nuw i32, ptr %c, i64 %iv
+  store i32 %add, ptr %c.ptr, align 4
+  %dec = add nsw i64 %iv, -1
+  %cmp = icmp sgt i64 %iv, 0
+  br i1 %cmp, label %for.body, label %exit
+
+exit:
+  ret void
+}
+
+; Negative test (load): A strided load (stride = 2). Only the store should be widened to a VF multiple.
+define void @gather_nxv4i32_stride2(ptr noalias %a, ptr noalias %b, i64 %n) {
+; UNMASKED-SVE2P1-LABEL: define void @gather_nxv4i32_stride2(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ITER_CHECK:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ule i64 [[N]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ule i64 [[N]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[N_MOD_VF]], 0
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = select i1 [[TMP3]], i64 [[TMP1]], i64 [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = add i64 [[TMP2]], 0
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = mul i64 [[TMP5]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = add i64 [[INDEX]], [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = shl i64 [[TMP2]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = add i64 [[TMP8]], 0
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = mul i64 [[TMP9]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add i64 [[INDEX]], [[TMP10]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = mul i64 [[TMP2]], 3
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add i64 [[TMP12]], 0
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = mul i64 [[TMP13]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = add i64 [[INDEX]], [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = shl i64 [[INDEX]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = shl i64 [[TMP7]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = shl i64 [[TMP11]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = shl i64 [[TMP15]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[TMP16]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[TMP17]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[TMP18]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[TMP19]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_VEC:%.*]] = load <vscale x 8 x float>, ptr [[TMP20]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[STRIDED_VEC:%.*]] = call { <vscale x 4 x float>, <vscale x 4 x float> } @llvm.vector.deinterleave2.nxv8f32(<vscale x 8 x float> [[WIDE_VEC]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = extractvalue { <vscale x 4 x float>, <vscale x 4 x float> } [[STRIDED_VEC]], 0
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_VEC2:%.*]] = load <vscale x 8 x float>, ptr [[TMP21]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[STRIDED_VEC3:%.*]] = call { <vscale x 4 x float>, <vscale x 4 x float> } @llvm.vector.deinterleave2.nxv8f32(<vscale x 8 x float> [[WIDE_VEC2]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = extractvalue { <vscale x 4 x float>, <vscale x 4 x float> } [[STRIDED_VEC3]], 0
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_VEC4:%.*]] = load <vscale x 8 x float>, ptr [[TMP22]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[STRIDED_VEC5:%.*]] = call { <vscale x 4 x float>, <vscale x 4 x float> } @llvm.vector.deinterleave2.nxv8f32(<vscale x 8 x float> [[WIDE_VEC4]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = extractvalue { <vscale x 4 x float>, <vscale x 4 x float> } [[STRIDED_VEC5]], 0
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_VEC6:%.*]] = load <vscale x 8 x float>, ptr [[TMP23]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[STRIDED_VEC7:%.*]] = call { <vscale x 4 x float>, <vscale x 4 x float> } @llvm.vector.deinterleave2.nxv8f32(<vscale x 8 x float> [[WIDE_VEC6]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = extractvalue { <vscale x 4 x float>, <vscale x 4 x float> } [[STRIDED_VEC7]], 0
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> poison, <vscale x 4 x float> [[TMP24]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP29]], <vscale x 4 x float> [[TMP25]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP30]], <vscale x 4 x float> [[TMP26]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP31]], <vscale x 4 x float> [[TMP27]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x float> [[TMP32]], ptr [[TMP28]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP33]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ %iv.next, %for.body ], [ 0, %entry ]
+  %iv.stride2 = mul i64 %iv, 2
+  %b.ptr = getelementptr inbounds float, ptr %b, i64 %iv.stride2
+  %b.val = load float, ptr %b.ptr, align 4
+  %a.ptr = getelementptr inbounds float, ptr %a, i64 %iv
+  store float %b.val, ptr %a.ptr, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond.not = icmp eq i64 %iv.next, %n
+  br i1 %exitcond.not, label %exit, label %for.body
+
+exit:
+  ret void
+}

>From 988b5f9918961c09977e68caca5911a786c61553 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 7 Sep 2026 13:14:38 +0000
Subject: [PATCH 07/20] Rework opcodes

---
 llvm/lib/Transforms/Vectorize/VPlan.h         | 38 +++++------
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp | 55 ++++++++++------
 .../Transforms/Vectorize/VPlanTransforms.cpp  | 27 +++++++-
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp | 65 ++++++++++---------
 .../vplan-printing-multi-vector-mem-ops.ll    | 40 ++++++------
 5 files changed, 131 insertions(+), 94 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index c1747697a9f2f..f6b8d13e287b2 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1311,10 +1311,18 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
     // WideActiveLaneMask is used for control flow and is unrolled by widening,
     // with one extract vector created per unroll part.
     WideActiveLaneMask,
+    // Signature: (VFMultiple, Address, Alignment) -> Wide Vector
+    // Loads a single wide vector of `VFMultiple * VF` elements. VFMultiple must
+    // divide UF. After unrolling, each section of VF elements in the wide
+    // vector corresponds to an unroll part.
+    VFMultipleLoad,
+    // Signature: (VFMultiple, Address, Alignment, Vectors...)
+    // Concatenates VFMultiple vector operands into a single wide vector of
+    // `VFMultiple * VF` elements and stores it. After unrolling, each vector
+    // operand corresponds to an unroll part.
+    VFMultipleStore,
     // Extracts each unrolled part of a (VF * UF) widened vector/mask.
     ExtractVectorForPart,
-    // Concatenates its unrolled part operands into one widened vector.
-    ConcatVectorParts,
     ExplicitVectorLength,
     // Represents the incoming loop-invariant alias-mask. All memory accesses
     // in the loop must stay within the active lanes.
@@ -1517,6 +1525,7 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
     case VPInstruction::BranchOnCond:
     case VPInstruction::BranchOnTwoConds:
     case VPInstruction::BranchOnCount:
+    case VPInstruction::VFMultipleStore:
       return false;
     default:
       return true;
@@ -3755,10 +3764,6 @@ class LLVM_ABI_FOR_TEST VPWidenMemoryRecipe : public VPIRMetadata {
   /// Whether the memory access is masked.
   bool IsMasked = false;
 
-  /// Multiple of VF used to widen this memory operation. The final operation
-  /// loads or stores VF * VFMultiple elements
-  unsigned VFMultiple = 1;
-
   void setMask(VPValue *Mask) {
     assert(!IsMasked && "cannot re-set mask");
     if (!Mask)
@@ -3805,12 +3810,6 @@ class LLVM_ABI_FOR_TEST VPWidenMemoryRecipe : public VPIRMetadata {
   InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const;
 
   Instruction &getIngredient() const { return Ingredient; }
-
-  /// Set the VF multiple for this memory operation.
-  void setVFMultiple(unsigned VFMultiple) { this->VFMultiple = VFMultiple; }
-
-  /// Returns the VF multiple of this memory operation.
-  unsigned getVFMultiple() const { return VFMultiple; }
 };
 
 /// A recipe for widening load operations, using the address to load from and an
@@ -3826,11 +3825,8 @@ struct LLVM_ABI_FOR_TEST VPWidenLoadRecipe final : public VPSingleDefRecipe,
   }
 
   VPWidenLoadRecipe *clone() override {
-    auto *R =
-        new VPWidenLoadRecipe(cast<LoadInst>(Ingredient), getAddr(), getMask(),
-                              Consecutive, *this, getDebugLoc());
-    R->setVFMultiple(VFMultiple);
-    return R;
+    return new VPWidenLoadRecipe(cast<LoadInst>(Ingredient), getAddr(),
+                                 getMask(), Consecutive, *this, getDebugLoc());
   }
 
   VP_CLASSOF_IMPL(VPRecipeBase::VPWidenLoadSC);
@@ -3934,11 +3930,9 @@ struct LLVM_ABI_FOR_TEST VPWidenStoreRecipe final : public VPRecipeBase,
   }
 
   VPWidenStoreRecipe *clone() override {
-    auto *R = new VPWidenStoreRecipe(cast<StoreInst>(Ingredient), getAddr(),
-                                     getStoredValue(), getMask(), Consecutive,
-                                     *this, getDebugLoc());
-    R->setVFMultiple(VFMultiple);
-    return R;
+    return new VPWidenStoreRecipe(cast<StoreInst>(Ingredient), getAddr(),
+                                  getStoredValue(), getMask(), Consecutive,
+                                  *this, getDebugLoc());
   }
 
   VP_CLASSOF_IMPL(VPRecipeBase::VPWidenStoreSC);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 4dc978e04f343..c4f750315cd06 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -491,6 +491,7 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
     for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
       AssertOperandType(Idx, Op0Ty);
     return Type::getVoidTy(Ctx);
+  case VPInstruction::VFMultipleStore:
   case Instruction::Store:
     return Type::getVoidTy(Ctx);
   case Instruction::ICmp:
@@ -564,9 +565,9 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
     return StructTy->getTypeAtIndex(
         cast<VPConstantInt>(Operands[1])->getZExtValue());
   }
-  case VPInstruction::ConcatVectorParts:
   case VPInstruction::ExtractVectorForPart:
     return Op0Ty;
+  case VPInstruction::VFMultipleLoad:
   case VPInstruction::FirstActiveLane:
   case VPInstruction::LastActiveLane:
   case VPInstruction::NumActiveLanes:
@@ -675,6 +676,7 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
   case Instruction::Select:
   case VPInstruction::WideActiveLaneMask:
   case VPInstruction::ReductionStartVector:
+  case VPInstruction::VFMultipleLoad:
     return 3;
   case Instruction::Call:
     return getCalledFnOperandIndex(operands()) + 1;
@@ -694,7 +696,7 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
   case VPInstruction::LastActiveLane:
   case VPInstruction::ExtractLane:
   case VPInstruction::ExtractLastActive:
-  case VPInstruction::ConcatVectorParts:
+  case VPInstruction::VFMultipleStore:
     // Cannot determine the number of operands from the opcode.
     return -1u;
   }
@@ -1181,18 +1183,32 @@ Value *VPInstruction::generate(VPTransformState &State,
                                          vputils::getIntrinsicID(this), Args,
                                          /*FMFSource=*/nullptr, getName());
   }
-  case VPInstruction::ConcatVectorParts: {
-    unsigned VectorOps = getNumOperands();
+  case VPInstruction::VFMultipleLoad: {
+    unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
     auto *WideDataTy = VectorType::get(
-        getScalarType(), State.VF.multiplyCoefficientBy(VectorOps));
-    Value *WideData = PoisonValue::get(WideDataTy);
+        getScalarType(), State.VF.multiplyCoefficientBy(VFMultiple));
+
+    Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
+    Align Alignment = Align(cast<VPConstantInt>(getOperand(2))->getZExtValue());
+    return Builder.CreateAlignedLoad(WideDataTy, Addr, Alignment,
+                                     "vf.multiple.load");
+  }
+  case VPInstruction::VFMultipleStore: {
+    unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
+    Type *ScalarStoreTy = getOperand(3)->getScalarType();
+    auto *WideDataTy = VectorType::get(
+        ScalarStoreTy, State.VF.multiplyCoefficientBy(VFMultiple));
 
-    for (unsigned I = 0; I < VectorOps; ++I) {
-      Value *Part = State.get(getOperand(I));
+    Value *WideData = PoisonValue::get(WideDataTy);
+    for (unsigned I = 0; I < VFMultiple; ++I) {
+      Value *Part = State.get(getOperand(I + 3));
       WideData = Builder.CreateInsertVector(WideDataTy, WideData, Part,
                                             I * State.VF.getKnownMinValue());
     }
-    return WideData;
+
+    Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
+    Align Alignment = Align(cast<VPConstantInt>(getOperand(2))->getZExtValue());
+    return Builder.CreateAlignedStore(WideData, Addr, Alignment);
   }
   default:
     llvm_unreachable("Unsupported opcode for instruction");
@@ -1677,6 +1693,10 @@ void VPInstruction::addOperand(VPValue *Op) {
            "matching operand 1's type and i1, respectively");
     break;
   }
+  case VPInstruction::VFMultipleStore:
+    assert(Ty == getOperand(3)->getScalarType() &&
+           "appended operand must match operand 3's scalar type");
+    break;
   default:
     llvm_unreachable("opcode does not support growing the operand list "
                      "outside of construction");
@@ -1743,7 +1763,6 @@ bool VPInstruction::opcodeMayReadOrWriteFromMemory() const {
   case VPInstruction::ActiveLaneMask:
   case VPInstruction::WideActiveLaneMask:
   case VPInstruction::IncomingAliasMask:
-  case VPInstruction::ConcatVectorParts:
   case VPInstruction::ExitingIVValue:
   case VPInstruction::ExplicitVectorLength:
   case VPInstruction::FirstActiveLane:
@@ -1877,12 +1896,15 @@ void VPInstruction::printRecipe(raw_ostream &O, const Twine &Indent,
   case VPInstruction::WideActiveLaneMask:
     O << "wide active lane mask";
     break;
+  case VPInstruction::VFMultipleLoad:
+    O << "vf-multiple load";
+    break;
+  case VPInstruction::VFMultipleStore:
+    O << "vf-multiple store";
+    break;
   case VPInstruction::IncomingAliasMask:
     O << "incoming-alias-mask";
     break;
-  case VPInstruction::ConcatVectorParts:
-    O << "concat-vector-parts";
-    break;
   case VPInstruction::ExplicitVectorLength:
     O << "EXPLICIT-VECTOR-LENGTH";
     break;
@@ -4390,8 +4412,7 @@ InstructionCost VPWidenMemoryRecipe::computeCost(ElementCount VF,
 
 void VPWidenLoadRecipe::execute(VPTransformState &State) {
   Type *ScalarDataTy = getScalarType();
-  auto *DataTy =
-      VectorType::get(ScalarDataTy, State.VF.multiplyCoefficientBy(VFMultiple));
+  auto *DataTy = VectorType::get(ScalarDataTy, State.VF);
   bool CreateGather = !isConsecutive();
 
   auto &Builder = State.Builder;
@@ -4421,8 +4442,6 @@ void VPWidenLoadRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
   O << Indent << "WIDEN ";
   printAsOperand(O, SlotTracker);
   O << " = load ";
-  if (VFMultiple > 1)
-    O << "x" << VFMultiple << ' ';
   printOperands(O, SlotTracker);
 }
 #endif
@@ -4510,8 +4529,6 @@ void VPWidenStoreRecipe::execute(VPTransformState &State) {
 void VPWidenStoreRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
                                      VPSlotTracker &SlotTracker) const {
   O << Indent << "WIDEN store ";
-  if (VFMultiple > 1)
-    O << "x" << VFMultiple << ' ';
   printOperands(O, SlotTracker);
 }
 #endif
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index d3627ac3aa362..91af5680ec398 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4185,6 +4185,8 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
   if (UF == 1)
     return;
 
+  Type *IVTy = Plan.getVectorLoopRegion()->getCanonicalIVType();
+
   for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
            vp_depth_first_shallow(Plan.getVectorLoopRegion()->getEntry()))) {
     for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
@@ -4222,7 +4224,30 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
           Opcode, AccessType, VF, UF, /*IsMasked=*/false, CastHint);
       assert((VFMultiple != 0 && UF % VFMultiple == 0) &&
              "VFMultiple must divide UF");
-      MemOp->setVFMultiple(VFMultiple);
+
+      if (VFMultiple == 1)
+        continue;
+
+      VPBuilder Builder(VPBB, R.getIterator());
+
+      VPValue *Ptr = MemOp->getAddr();
+      VPValue *VFMultipleVPV = Plan.getConstantInt(IVTy, VFMultiple);
+      VPValue *Align = Plan.getConstantInt(IVTy, MemOp->getAlign().value());
+
+      if (Opcode == Instruction::Load) {
+        VPValue *OldLoad = R.getVPSingleValue();
+        VPValue *Load = Builder.createNaryOp(
+            VPInstruction::VFMultipleLoad, {VFMultipleVPV, Ptr, Align}, nullptr,
+            {}, {}, DebugLoc::getUnknown(), "", OldLoad->getScalarType());
+        OldLoad->replaceAllUsesWith(Load);
+      } else {
+        assert(Opcode == Instruction::Store);
+        Builder.createNaryOp(VPInstruction::VFMultipleStore,
+                             {VFMultipleVPV, Ptr, Align, StoredValue},
+                             R.getDebugLoc());
+      }
+
+      R.eraseFromParent();
     }
   }
 }
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 5ab9c6a2e7dfd..95f373889d065 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -74,8 +74,8 @@ class UnrollState {
     return Plan.getConstantInt(CanIVIntTy, Part);
   }
 
-  /// Unroll a VPWidenLoadRecipe or VPWidenStoreRecipe with a VFMultiple > 1.
-  void unrollMemOpWithVFMultiple(VPRecipeBase &R, unsigned VFMultiple);
+  /// Unroll a VFMultipleLoad or VFMultipleStore VPInstruction.
+  void unrollMemOpWithVFMultiple(VPInstruction *VPI);
 
 public:
   UnrollState(VPlan &Plan, unsigned UF) : Plan(Plan), UF(UF) {}
@@ -293,53 +293,57 @@ void UnrollState::unrollHeaderPHIByUF(VPHeaderPHIRecipe *R,
   }
 }
 
-void UnrollState::unrollMemOpWithVFMultiple(VPRecipeBase &R,
-                                            unsigned VFMultiple) {
+void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
+  assert(VPI->getOpcode() == VPInstruction::VFMultipleLoad ||
+         VPI->getOpcode() == VPInstruction::VFMultipleStore);
+
+  unsigned VFMultiple = cast<VPConstantInt>(VPI->getOperand(0))->getZExtValue();
   assert(VFMultiple > 1 && UF % VFMultiple == 0 &&
          "expected VFMultiple to divide UF");
-  SmallVector<VPRecipeBase *, 4> Groups(UF / VFMultiple, nullptr);
-  Groups[0] = &R;
+
+  SmallVector<VPInstruction *, 4> Groups(UF / VFMultiple, nullptr);
+  Groups[0] = VPI;
 
   // A memory op with a VFMultiple is widened to VF * VFMultiple elements, so
   // after unrolling by UF we materialize UF / VFMultiple such ops, each
   // covering VFMultiple unroll parts.
-  VPBuilder Builder = VPBuilder::getToInsertAfter(&R);
+  VPBuilder Builder = VPBuilder::getToInsertAfter(VPI);
   for (unsigned Group = 1; Group < Groups.size(); ++Group) {
-    auto *Copy = Builder.insert(R.clone());
+    auto *Copy = Builder.insert(VPI->clone());
     remapOperands(Copy, Group * VFMultiple);
     Groups[Group] = Copy;
   }
 
-  if (auto *Store = dyn_cast<VPWidenStoreRecipe>(&R)) {
-    VPValue *StoredValue = Store->getStoredValue();
+  if (VPI->getOpcode() == VPInstruction::VFMultipleStore) {
+    VPValue *StoredValue = VPI->getOperand(3);
     for (unsigned Group = 0; Group < Groups.size(); ++Group) {
-      VPRecipeBase *Store = Groups[Group];
-      Builder.setInsertPoint(Store);
-      SmallVector<VPValue *, 4> Parts;
-      // We need to concatenate VFMultiple parts to form the stored value.
-      for (unsigned Part = 0; Part < VFMultiple; ++Part)
-        Parts.push_back(
-            getValueForPart(StoredValue, Group * VFMultiple + Part));
-      auto *Concat =
-          Builder.createNaryOp(VPInstruction::ConcatVectorParts, Parts);
-      Groups[Group]->setOperand(1, Concat);
-      ToSkip.insert(Concat);
+      VPInstruction *Store = Groups[Group];
+      // Add the value to store for each unroll part in this group.
+      for (unsigned Part = 0; Part < VFMultiple; ++Part) {
+        VPValue *UnrollPart =
+            getValueForPart(StoredValue, Group * VFMultiple + Part);
+        if (Part == 0)
+          Store->setOperand(3, UnrollPart);
+        else
+          Store->addOperand(UnrollPart);
+      }
     }
   } else {
-    assert(isa<VPWidenLoadRecipe>(R) && "Expected a load recipe");
+    assert(VPI->getOpcode() == VPInstruction::VFMultipleLoad &&
+           "Expected a load recipe");
     // We need to extract each unroll part as a subvector.
     auto ExtractPart0 = Builder.createNaryOp(
         VPInstruction::ExtractVectorForPart,
         {Groups[0]->getVPSingleValue(), getConstantInt(0)});
-    // First replace R with an extract of the first unroll part (ExtractPart0).
-    R.getVPSingleValue()->replaceUsesWithIf(
+    // First VPI with an extract of the first unroll part (ExtractPart0).
+    VPI->getVPSingleValue()->replaceUsesWithIf(
         ExtractPart0, [&](VPUser &U, unsigned) { return &U != ExtractPart0; });
     ToSkip.insert(ExtractPart0);
 
     // Create extracts for the remaining unroll parts and remap later uses of
     // ExtractPart0 to the correct unrolled part.
     for (unsigned Part = 1; Part != UF; ++Part) {
-      VPRecipeBase *Group = Groups[Part / VFMultiple];
+      VPInstruction *Group = Groups[Part / VFMultiple];
       unsigned IndexInGroup = Part % VFMultiple;
       auto *Extract = Builder.createNaryOp(
           VPInstruction::ExtractVectorForPart,
@@ -356,15 +360,14 @@ void UnrollState::unrollRecipeByUF(VPRecipeBase &R) {
     return;
 
   if (auto *VPI = dyn_cast<VPInstruction>(&R)) {
-    if (vputils::onlyFirstPartUsed(VPI)) {
-      addUniformForAllParts(VPI);
+    if (VPI->getOpcode() == VPInstruction::VFMultipleLoad ||
+        VPI->getOpcode() == VPInstruction::VFMultipleStore) {
+      unrollMemOpWithVFMultiple(VPI);
       return;
     }
-  }
 
-  if (auto *WidenMem = dyn_cast<VPWidenMemoryRecipe>(&R)) {
-    if (WidenMem && WidenMem->getVFMultiple() > 1) {
-      unrollMemOpWithVFMultiple(R, WidenMem->getVFMultiple());
+    if (vputils::onlyFirstPartUsed(VPI)) {
+      addUniformForAllParts(VPI);
       return;
     }
   }
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
index 3943cb5f28103..15dc08babf853 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
@@ -69,10 +69,10 @@ define void @i64_load_store(ptr noalias %x, i64 %n) {
 ; AFTER-SCALE-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
 ; AFTER-SCALE-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
 ; AFTER-SCALE-NEXT:      vp<[[VP9:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-SCALE-NEXT:      WIDEN ir<%ld> = load x2 vp<[[VP9]]>
-; AFTER-SCALE-NEXT:      WIDEN ir<%add> = add ir<%ld>, ir<1>
-; AFTER-SCALE-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-SCALE-NEXT:      WIDEN store x2 vp<[[VP10]]>, ir<%add>
+; AFTER-SCALE-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP9]]>, ir<8>
+; AFTER-SCALE-NEXT:      WIDEN ir<%add> = add vp<[[VP10]]>, ir<1>
+; AFTER-SCALE-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-SCALE-NEXT:      EMIT vf-multiple store ir<2>, vp<[[VP11]]>, ir<8>, ir<%add>
 ; AFTER-SCALE-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
 ; AFTER-SCALE-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; AFTER-SCALE-NEXT:    No successors
@@ -108,23 +108,21 @@ define void @i64_load_store(ptr noalias %x, i64 %n) {
 ; AFTER-UNROLL-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
 ; AFTER-UNROLL-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
 ; AFTER-UNROLL-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP9]]>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%ld> = load x2 vp<[[VP10]]>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%ld>.1 = load x2 vp<[[VP11]]>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = extract-vector-for-part ir<%ld>, ir<0>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = extract-vector-for-part ir<%ld>, ir<1>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP14:%[0-9]+]]> = extract-vector-for-part ir<%ld>.1, ir<0>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP15:%[0-9]+]]> = extract-vector-for-part ir<%ld>.1, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add> = add vp<[[VP12]]>, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.1 = add vp<[[VP13]]>, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.2 = add vp<[[VP14]]>, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.3 = add vp<[[VP15]]>, ir<1>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP16:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
-; AFTER-UNROLL-NEXT:      vp<[[VP17:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-UNROLL-NEXT:      vp<[[VP18:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP16]]>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP19:%[0-9]+]]> = concat-vector-parts ir<%add>, ir<%add>.1
-; AFTER-UNROLL-NEXT:      WIDEN store x2 vp<[[VP17]]>, vp<[[VP19]]>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP20:%[0-9]+]]> = concat-vector-parts ir<%add>.2, ir<%add>.3
-; AFTER-UNROLL-NEXT:      WIDEN store x2 vp<[[VP18]]>, vp<[[VP20]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP10]]>, ir<8>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP11]]>, ir<8>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP14:%[0-9]+]]> = extract-vector-for-part vp<[[VP12]]>, ir<0>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP15:%[0-9]+]]> = extract-vector-for-part vp<[[VP12]]>, ir<1>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP16:%[0-9]+]]> = extract-vector-for-part vp<[[VP13]]>, ir<0>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP17:%[0-9]+]]> = extract-vector-for-part vp<[[VP13]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add> = add vp<[[VP14]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.1 = add vp<[[VP15]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.2 = add vp<[[VP16]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.3 = add vp<[[VP17]]>, ir<1>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP18:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
+; AFTER-UNROLL-NEXT:      vp<[[VP19:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-UNROLL-NEXT:      vp<[[VP20:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP18]]>
+; AFTER-UNROLL-NEXT:      EMIT vf-multiple store ir<2>, vp<[[VP19]]>, ir<8>, ir<%add>, ir<%add>.1
+; AFTER-UNROLL-NEXT:      EMIT vf-multiple store ir<2>, vp<[[VP20]]>, ir<8>, ir<%add>.2, ir<%add>.3
 ; AFTER-UNROLL-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
 ; AFTER-UNROLL-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; AFTER-UNROLL-NEXT:    No successors

>From e6c9f79bd5a3d599317e55a3c55e6f9b2a8d98c8 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 7 Sep 2026 14:09:53 +0000
Subject: [PATCH 08/20] Fixups

---
 llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp    | 12 +++++++++---
 llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp |  3 +--
 2 files changed, 10 insertions(+), 5 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index c4f750315cd06..f0851ceeb0ff9 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -59,7 +59,8 @@ bool VPRecipeBase::mayWriteToMemory() const {
   case VPInstructionSC: {
     auto *VPI = cast<VPInstruction>(this);
     // Loads read from memory but don't write to memory.
-    if (VPI->getOpcode() == Instruction::Load)
+    if (VPI->getOpcode() == Instruction::Load ||
+        VPI->getOpcode() == VPInstruction::VFMultipleLoad)
       return false;
     return VPI->opcodeMayReadOrWriteFromMemory();
   }
@@ -118,8 +119,13 @@ bool VPRecipeBase::mayReadFromMemory() const {
   switch (getVPRecipeID()) {
   case VPExpressionSC:
     return cast<VPExpressionRecipe>(this)->mayReadOrWriteMemory();
-  case VPInstructionSC:
-    return cast<VPInstruction>(this)->opcodeMayReadOrWriteFromMemory();
+  case VPInstructionSC: {
+    auto *VPI = cast<VPInstruction>(this);
+    // Stores write to memory but don't read from memory.
+    if (VPI->getOpcode() == VPInstruction::VFMultipleStore)
+      return false;
+    return VPI->opcodeMayReadOrWriteFromMemory();
+  }
   case VPWidenLoadEVLSC:
   case VPWidenLoadSC:
     return true;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 91af5680ec398..a49fc26d8cc3e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4228,12 +4228,11 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
       if (VFMultiple == 1)
         continue;
 
-      VPBuilder Builder(VPBB, R.getIterator());
-
       VPValue *Ptr = MemOp->getAddr();
       VPValue *VFMultipleVPV = Plan.getConstantInt(IVTy, VFMultiple);
       VPValue *Align = Plan.getConstantInt(IVTy, MemOp->getAlign().value());
 
+      VPBuilder Builder(VPBB, R.getIterator());
       if (Opcode == Instruction::Load) {
         VPValue *OldLoad = R.getVPSingleValue();
         VPValue *Load = Builder.createNaryOp(

>From 1ce8fb56e99e245d4886b92a5b757ea6aa9295ee Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 7 Sep 2026 16:33:45 +0000
Subject: [PATCH 09/20] Fixups

---
 llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp | 3 +++
 1 file changed, 3 insertions(+)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index f0851ceeb0ff9..8d4006e9d9e03 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1849,6 +1849,9 @@ bool VPInstruction::usesFirstLaneOnly(const VPValue *Op) const {
   case VPInstruction::WidePtrAdd:
     // WidePtrAdd supports scalar and vector base addresses.
     return false;
+  case VPInstruction::VFMultipleLoad:
+  case VPInstruction::VFMultipleStore:
+    return Op == getOperand(0) || Op == getOperand(1) || Op == getOperand(2);
   case VPInstruction::ExitingIVValue:
   case VPInstruction::ExtractLane:
     return Op == getOperand(0);

>From e8cb43f73c19ed83f8403f7f40da05d67c11b427 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Fri, 11 Sep 2026 09:59:38 +0000
Subject: [PATCH 10/20] Rebase fixups

---
 llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp | 6 ++----
 1 file changed, 2 insertions(+), 4 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 8d4006e9d9e03..29cadafeb2c3f 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1191,8 +1191,7 @@ Value *VPInstruction::generate(VPTransformState &State,
   }
   case VPInstruction::VFMultipleLoad: {
     unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
-    auto *WideDataTy = VectorType::get(
-        getScalarType(), State.VF.multiplyCoefficientBy(VFMultiple));
+    auto *WideDataTy = VectorType::get(getScalarType(), State.VF * VFMultiple);
 
     Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
     Align Alignment = Align(cast<VPConstantInt>(getOperand(2))->getZExtValue());
@@ -1202,8 +1201,7 @@ Value *VPInstruction::generate(VPTransformState &State,
   case VPInstruction::VFMultipleStore: {
     unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
     Type *ScalarStoreTy = getOperand(3)->getScalarType();
-    auto *WideDataTy = VectorType::get(
-        ScalarStoreTy, State.VF.multiplyCoefficientBy(VFMultiple));
+    auto *WideDataTy = VectorType::get(ScalarStoreTy, State.VF * VFMultiple);
 
     Value *WideData = PoisonValue::get(WideDataTy);
     for (unsigned I = 0; I < VFMultiple; ++I) {

>From 3006084d2e954c1502943d3aa1c88b7cc6183681 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Fri, 11 Sep 2026 10:24:13 +0000
Subject: [PATCH 11/20] Fixups

---
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp | 45 ++++++++++---------
 1 file changed, 23 insertions(+), 22 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 95f373889d065..7d887d868c22c 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -328,29 +328,30 @@ void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
           Store->addOperand(UnrollPart);
       }
     }
-  } else {
-    assert(VPI->getOpcode() == VPInstruction::VFMultipleLoad &&
-           "Expected a load recipe");
-    // We need to extract each unroll part as a subvector.
-    auto ExtractPart0 = Builder.createNaryOp(
+    return;
+  }
+
+  assert(VPI->getOpcode() == VPInstruction::VFMultipleLoad &&
+         "Expected a VFMultipleLoad instruction");
+  // We need to extract each unroll part as a subvector.
+  auto *ExtractPart0 =
+      Builder.createNaryOp(VPInstruction::ExtractVectorForPart,
+                           {Groups[0]->getVPSingleValue(), getConstantInt(0)});
+  // First VPI with an extract of the first unroll part (ExtractPart0).
+  VPI->getVPSingleValue()->replaceUsesWithIf(
+      ExtractPart0, [&](VPUser &U, unsigned) { return &U != ExtractPart0; });
+  ToSkip.insert(ExtractPart0);
+
+  // Create extracts for the remaining unroll parts and remap later uses of
+  // ExtractPart0 to the correct unrolled part.
+  for (unsigned Part = 1; Part != UF; ++Part) {
+    VPInstruction *Group = Groups[Part / VFMultiple];
+    unsigned IndexInGroup = Part % VFMultiple;
+    auto *Extract = Builder.createNaryOp(
         VPInstruction::ExtractVectorForPart,
-        {Groups[0]->getVPSingleValue(), getConstantInt(0)});
-    // First VPI with an extract of the first unroll part (ExtractPart0).
-    VPI->getVPSingleValue()->replaceUsesWithIf(
-        ExtractPart0, [&](VPUser &U, unsigned) { return &U != ExtractPart0; });
-    ToSkip.insert(ExtractPart0);
-
-    // Create extracts for the remaining unroll parts and remap later uses of
-    // ExtractPart0 to the correct unrolled part.
-    for (unsigned Part = 1; Part != UF; ++Part) {
-      VPInstruction *Group = Groups[Part / VFMultiple];
-      unsigned IndexInGroup = Part % VFMultiple;
-      auto *Extract = Builder.createNaryOp(
-          VPInstruction::ExtractVectorForPart,
-          {Group->getVPSingleValue(), getConstantInt(IndexInGroup)});
-      addRecipeForPart(ExtractPart0, Extract, Part);
-      ToSkip.insert(Extract);
-    }
+        {Group->getVPSingleValue(), getConstantInt(IndexInGroup)});
+    addRecipeForPart(ExtractPart0, Extract, Part);
+    ToSkip.insert(Extract);
   }
 }
 

>From 806635324769f8103b7f08c6af3df76988987e96 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Tue, 15 Sep 2026 10:45:35 +0000
Subject: [PATCH 12/20] Fixups

---
 .../llvm/Analysis/TargetTransformInfo.h       |  9 +++-
 .../llvm/Analysis/TargetTransformInfoImpl.h   |  5 ++
 llvm/lib/Analysis/TargetTransformInfo.cpp     |  6 +++
 .../AArch64/AArch64TargetTransformInfo.cpp    | 18 ++++---
 .../AArch64/AArch64TargetTransformInfo.h      |  3 ++
 .../Transforms/Vectorize/LoopVectorize.cpp    |  5 +-
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp |  2 +-
 .../Transforms/Vectorize/VPlanTransforms.cpp  |  4 +-
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp | 17 +++---
 .../AArch64/multi-vector-mem-ops.ll           | 54 +++++++++++++++++++
 .../VPlan/vplan-print-before-after-all.ll     |  1 -
 11 files changed, 98 insertions(+), 26 deletions(-)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index 95c6d9507370b..b6cfebaef50d0 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1001,10 +1001,15 @@ class TargetTransformInfo {
                                 unsigned Opcode1,
                                 const SmallBitVector &OpcodeMask) const;
 
+  /// Returns the maximum VF multiple that can be used for a contiguous
+  /// load/store for the given \p VF and \p UF.
+  LLVM_ABI unsigned getMaximumVFMultipleForMemoryOp(ElementCount VF,
+                                                    unsigned UF) const;
+
   /// Return the preferred multiple of VF to use for a contiguous load/store.
   /// Returning 1 leaves the operation at VF. The returned value must divide UF.
-  /// \p CastHint is non-null if the stored or loaded valued is produced by or
-  /// consumed a cast instruction respectively.
+  /// \p CastHint is non-null if the stored value is produced by a cast
+  /// instruction or the loaded value is consumed by one.
   ///
   /// \p Opcode must be either Instruction::Load or Instruction::Store.
   LLVM_ABI unsigned getPreferredVFMultipleForMemoryOp(
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index f70257b10fbac..57073b8d87a10 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -429,6 +429,11 @@ class LLVM_ABI TargetTransformInfoImplBase {
     return false;
   }
 
+  virtual unsigned getMaximumVFMultipleForMemoryOp(ElementCount VF,
+                                                   unsigned UF) const {
+    return 1;
+  }
+
   virtual unsigned getPreferredVFMultipleForMemoryOp(
       unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
       bool IsMasked, std::optional<Instruction::CastOps> CastHint) const {
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index ad2270fbd6e10..16477e2d58282 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -554,6 +554,12 @@ bool TargetTransformInfo::isLegalStridedLoadStore(Type *DataType,
   return TTIImpl->isLegalStridedLoadStore(DataType, Alignment);
 }
 
+unsigned
+TargetTransformInfo::getMaximumVFMultipleForMemoryOp(ElementCount VF,
+                                                     unsigned UF) const {
+  return TTIImpl->getMaximumVFMultipleForMemoryOp(VF, UF);
+}
+
 unsigned TargetTransformInfo::getPreferredVFMultipleForMemoryOp(
     unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
     bool IsMasked, std::optional<Instruction::CastOps> CastHint) const {
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 1c71ec719a1dd..6125860e7b565 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -5959,6 +5959,18 @@ bool AArch64TTIImpl::isLegalSpeculativeLoad(Type *DataType,
          Size.getFixedValue() <= 16;
 }
 
+unsigned AArch64TTIImpl::getMaximumVFMultipleForMemoryOp(ElementCount VF,
+                                                         unsigned UF) const {
+  if (!ST->enableSubRegLiveness())
+    return 1;
+
+  if (!ST->hasSVE2p1() || !VF.isScalable() || !isPowerOf2_32(UF))
+    return 1;
+
+  // +sve2p1 multi-vector loads/stores can handle up to four vectors.
+  return std::min(4U, UF);
+}
+
 unsigned AArch64TTIImpl::getPreferredVFMultipleForMemoryOp(
     unsigned Opcode, Type *DataTy, ElementCount VF, unsigned UF, bool IsMasked,
     std::optional<Instruction::CastOps> CastHint) const {
@@ -5974,12 +5986,6 @@ unsigned AArch64TTIImpl::getPreferredVFMultipleForMemoryOp(
       (CastHint == Instruction::ZExt || CastHint == Instruction::SExt))
     return 1;
 
-  if (!ST->enableSubRegLiveness())
-    return 1;
-
-  if (!ST->hasSVE2p1() || !VF.isScalable() || !isPowerOf2_32(UF))
-    return 1;
-
   unsigned VectorWidth = VF.getKnownMinValue() * DL.getTypeSizeInBits(DataTy);
   if (VectorWidth % 128 != 0)
     return 1;
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index aa1d059406c8c..042774c539a44 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -282,6 +282,9 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
   bool isLegalSpeculativeLoad(Type *DataType,
                               unsigned AddressSpace) const override;
 
+  unsigned getMaximumVFMultipleForMemoryOp(ElementCount VF,
+                                           unsigned UF) const override;
+
   unsigned getPreferredVFMultipleForMemoryOp(
       unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
       bool IsMasked,
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 98a8bc0ec1392..9a61d5d0aae9e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5752,8 +5752,9 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
                  *PSE.getSE(), TTI, Config.CostKind, BestVF, BestUF);
   // TODO: Move to VPlan transform stage once the transition to the VPlan-based
   // cost model is complete for better cost estimates.
-  RUN_VPLAN_PASS(VPlanTransforms::scaleMemoryAccessesByUF, BestVPlan, BestVF,
-                 BestUF, TTI);
+  if (TTI.getMaximumVFMultipleForMemoryOp(BestVF, BestUF) > 1)
+    RUN_VPLAN_PASS(VPlanTransforms::scaleMemoryAccessesByUF, BestVPlan, BestVF,
+                   BestUF, TTI);
   RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
   RUN_VPLAN_PASS(VPlanTransforms::materializePacksAndUnpacks, BestVPlan);
   RUN_VPLAN_PASS(VPlanTransforms::materializeBroadcasts, BestVPlan);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 29cadafeb2c3f..1e08311386deb 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1835,6 +1835,7 @@ bool VPInstruction::usesFirstLaneOnly(const VPValue *Op) const {
   case VPInstruction::Intrinsic:
   case VPInstruction::ReductionStartVector:
   case VPInstruction::ResumeForEpilogue:
+  case VPInstruction::VFMultipleLoad:
     return true;
   case VPInstruction::BuildStructVector:
   case VPInstruction::BuildVector:
@@ -1847,7 +1848,6 @@ bool VPInstruction::usesFirstLaneOnly(const VPValue *Op) const {
   case VPInstruction::WidePtrAdd:
     // WidePtrAdd supports scalar and vector base addresses.
     return false;
-  case VPInstruction::VFMultipleLoad:
   case VPInstruction::VFMultipleStore:
     return Op == getOperand(0) || Op == getOperand(1) || Op == getOperand(2);
   case VPInstruction::ExitingIVValue:
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index a49fc26d8cc3e..545990deb08d6 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4182,11 +4182,9 @@ void VPlanTransforms::sinkPredicatedStores(VPlan &Plan,
 void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
                                               unsigned UF,
                                               const TargetTransformInfo &TTI) {
-  if (UF == 1)
-    return;
+  assert(UF > 1 && "Expected plan to have an UF > 1");
 
   Type *IVTy = Plan.getVectorLoopRegion()->getCanonicalIVType();
-
   for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
            vp_depth_first_shallow(Plan.getVectorLoopRegion()->getEntry()))) {
     for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 7d887d868c22c..2f298c02c9c66 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -294,8 +294,9 @@ void UnrollState::unrollHeaderPHIByUF(VPHeaderPHIRecipe *R,
 }
 
 void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
-  assert(VPI->getOpcode() == VPInstruction::VFMultipleLoad ||
-         VPI->getOpcode() == VPInstruction::VFMultipleStore);
+  assert((VPI->getOpcode() == VPInstruction::VFMultipleLoad ||
+          VPI->getOpcode() == VPInstruction::VFMultipleStore) &&
+         "expected a vf-multiple load/store");
 
   unsigned VFMultiple = cast<VPConstantInt>(VPI->getOperand(0))->getZExtValue();
   assert(VFMultiple > 1 && UF % VFMultiple == 0 &&
@@ -331,16 +332,12 @@ void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
     return;
   }
 
-  assert(VPI->getOpcode() == VPInstruction::VFMultipleLoad &&
-         "Expected a VFMultipleLoad instruction");
   // We need to extract each unroll part as a subvector.
-  auto *ExtractPart0 =
-      Builder.createNaryOp(VPInstruction::ExtractVectorForPart,
-                           {Groups[0]->getVPSingleValue(), getConstantInt(0)});
+  auto *ExtractPart0 = Builder.createNaryOp(VPInstruction::ExtractVectorForPart,
+                                            {Groups[0], getConstantInt(0)});
   // First VPI with an extract of the first unroll part (ExtractPart0).
-  VPI->getVPSingleValue()->replaceUsesWithIf(
+  VPI->replaceUsesWithIf(
       ExtractPart0, [&](VPUser &U, unsigned) { return &U != ExtractPart0; });
-  ToSkip.insert(ExtractPart0);
 
   // Create extracts for the remaining unroll parts and remap later uses of
   // ExtractPart0 to the correct unrolled part.
@@ -351,7 +348,6 @@ void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
         VPInstruction::ExtractVectorForPart,
         {Group->getVPSingleValue(), getConstantInt(IndexInGroup)});
     addRecipeForPart(ExtractPart0, Extract, Part);
-    ToSkip.insert(Extract);
   }
 }
 
@@ -372,7 +368,6 @@ void UnrollState::unrollRecipeByUF(VPRecipeBase &R) {
       return;
     }
   }
-
   if (auto *RepR = dyn_cast<VPReplicateRecipe>(&R)) {
     if (isa<StoreInst>(RepR->getUnderlyingValue()) &&
         RepR->getOperand(1)->isDefinedOutsideLoopRegions()) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
index b90e29519e24e..398e2e0613e25 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -603,3 +603,57 @@ for.body:
 exit:
   ret void
 }
+
+define void @store_address(ptr noalias %x, ptr noalias %y, i64 %n) {
+; UNMASKED-SVE2P1-LABEL: define void @store_address(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[X:%.*]], ptr noalias [[Y:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 1
+; UNMASKED-SVE2P1-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[TMP3]], i64 0
+; UNMASKED-SVE2P1-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP4]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[STEP_ADD:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; UNMASKED-SVE2P1-NEXT:    [[STEP_ADD_2:%.*]] = add nuw <vscale x 2 x i64> [[STEP_ADD]], [[BROADCAST_SPLAT]]
+; UNMASKED-SVE2P1-NEXT:    [[STEP_ADD_3:%.*]] = add nuw <vscale x 2 x i64> [[STEP_ADD_2]], [[BROADCAST_SPLAT]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[VEC_IND]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = extractelement <vscale x 2 x ptr> [[WIDE_GEP]], i64 0
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP1:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[STEP_ADD]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP2:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[STEP_ADD_2]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP3:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[STEP_ADD_3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> poison, <vscale x 2 x ptr> [[WIDE_GEP]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> [[TMP6]], <vscale x 2 x ptr> [[WIDE_GEP1]], i64 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> [[TMP7]], <vscale x 2 x ptr> [[WIDE_GEP2]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> [[TMP8]], <vscale x 2 x ptr> [[WIDE_GEP3]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x ptr> [[TMP9]], ptr [[TMP5]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_IND_NEXT]] = add nuw <vscale x 2 x i64> [[STEP_ADD_3]], [[BROADCAST_SPLAT]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP23:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %px = getelementptr inbounds i64, ptr %x, i64 %iv
+  store ptr %px, ptr %px, align 8
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
index cb492acc29e1e..5fd844e186e44 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
@@ -71,7 +71,6 @@
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] printOptimizedVPlan
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::addMinimumIterationCheck
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::replaceWideCanonicalIVWithWideIV
-; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::scaleMemoryAccessesByUF
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::unrollByUF
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::materializePacksAndUnpacks
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::materializeBroadcasts

>From 380f249cedbf8f1779dd3e73dac18266e1a37b3f Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 28 Sep 2026 15:16:30 +0000
Subject: [PATCH 13/20] Fixups

---
 .../Transforms/Vectorize/LoopVectorize.cpp    |  4 +--
 .../Transforms/Vectorize/VPlanTransforms.cpp  | 28 ++++++++-----------
 .../Transforms/Vectorize/VPlanTransforms.h    |  5 ++--
 ...-vector-mem-ops-non-power-of-two-unroll.ll |  2 +-
 .../vplan-printing-multi-vector-mem-ops.ll    |  4 +--
 5 files changed, 20 insertions(+), 23 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 9a61d5d0aae9e..4ecef9d2e02bc 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5753,8 +5753,8 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
   // TODO: Move to VPlan transform stage once the transition to the VPlan-based
   // cost model is complete for better cost estimates.
   if (TTI.getMaximumVFMultipleForMemoryOp(BestVF, BestUF) > 1)
-    RUN_VPLAN_PASS(VPlanTransforms::scaleMemoryAccessesByUF, BestVPlan, BestVF,
-                   BestUF, TTI);
+    RUN_VPLAN_PASS(VPlanTransforms::widenMemoryAccessesToVFMultiple, BestVPlan,
+                   BestVF, BestUF, TTI);
   RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
   RUN_VPLAN_PASS(VPlanTransforms::materializePacksAndUnpacks, BestVPlan);
   RUN_VPLAN_PASS(VPlanTransforms::materializeBroadcasts, BestVPlan);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 545990deb08d6..8ea35befd5c7c 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4179,23 +4179,19 @@ void VPlanTransforms::sinkPredicatedStores(VPlan &Plan,
   }
 }
 
-void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
-                                              unsigned UF,
-                                              const TargetTransformInfo &TTI) {
+void VPlanTransforms::widenMemoryAccessesToVFMultiple(
+    VPlan &Plan, ElementCount VF, unsigned UF, const TargetTransformInfo &TTI) {
   assert(UF > 1 && "Expected plan to have an UF > 1");
 
-  Type *IVTy = Plan.getVectorLoopRegion()->getCanonicalIVType();
+  Type *I64Ty = IntegerType::getInt64Ty(Plan.getContext());
   for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
            vp_depth_first_shallow(Plan.getVectorLoopRegion()->getEntry()))) {
     for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
-      uint64_t Stride;
       VPValue *StoredValue = nullptr;
-      auto m_ConstantStrideVecPtr =
-          m_VecPtr(m_VPValue(), m_ConstantInt(Stride));
-      if ((!match(&R, m_WidenLoad(m_ConstantStrideVecPtr)) &&
-           !match(&R, m_WidenStore(m_ConstantStrideVecPtr,
-                                   m_VPValue(StoredValue)))) ||
-          Stride != 1)
+      auto m_ContiguousVecPtr = m_VecPtr(m_VPValue(), m_One());
+      if ((!match(&R, m_WidenLoad(m_ContiguousVecPtr)) &&
+           !match(&R,
+                  m_WidenStore(m_ContiguousVecPtr, m_VPValue(StoredValue)))))
         continue;
 
       auto *MemOp = cast<VPWidenMemoryRecipe>(&R);
@@ -4227,21 +4223,21 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
         continue;
 
       VPValue *Ptr = MemOp->getAddr();
-      VPValue *VFMultipleVPV = Plan.getConstantInt(IVTy, VFMultiple);
-      VPValue *Align = Plan.getConstantInt(IVTy, MemOp->getAlign().value());
+      VPValue *VFMultipleVPV = Plan.getConstantInt(I64Ty, VFMultiple);
+      VPValue *Align = Plan.getConstantInt(I64Ty, MemOp->getAlign().value());
 
       VPBuilder Builder(VPBB, R.getIterator());
       if (Opcode == Instruction::Load) {
         VPValue *OldLoad = R.getVPSingleValue();
         VPValue *Load = Builder.createNaryOp(
             VPInstruction::VFMultipleLoad, {VFMultipleVPV, Ptr, Align}, nullptr,
-            {}, {}, DebugLoc::getUnknown(), "", OldLoad->getScalarType());
+            {}, *MemOp, R.getDebugLoc(), "", OldLoad->getScalarType());
         OldLoad->replaceAllUsesWith(Load);
       } else {
         assert(Opcode == Instruction::Store);
         Builder.createNaryOp(VPInstruction::VFMultipleStore,
-                             {VFMultipleVPV, Ptr, Align, StoredValue},
-                             R.getDebugLoc());
+                             {VFMultipleVPV, Ptr, Align, StoredValue}, nullptr,
+                             {}, *MemOp, R.getDebugLoc());
       }
 
       R.eraseFromParent();
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 3f0634f039807..9683947802304 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -484,8 +484,9 @@ struct VPlanTransforms {
 
   /// Widens memory operations by a factor of UF based on a target hook.
   /// This allows targets to use wider memory operations when profitable.
-  static void scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF, unsigned UF,
-                                      const TargetTransformInfo &TTI);
+  static void widenMemoryAccessesToVFMultiple(VPlan &Plan, ElementCount VF,
+                                              unsigned UF,
+                                              const TargetTransformInfo &TTI);
 
   // Materialize vector trip counts for constants early if it can simply be
   // computed as (Original TC / VF * UF) * VF * UF.
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll
index ac9e55117364e..529909d353e17 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll
@@ -3,7 +3,7 @@
 
 target triple = "aarch64-unknown-linux-gnu"
 
-; Tests `scaleMemoryAccessesByUF` with a non-power-of-two unroll factor.
+; Tests `widenMemoryAccessesToVFMultiple` with a non-power-of-two unroll factor.
 ; On AArch64, this should be rejected (which results in `vscale x 4` loads/stores).
 
 define void @mixed_i64_i32_accesses(ptr noalias %x, ptr noalias %y, i64 %n) {
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
index 15dc08babf853..82b1201c22629 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
@@ -1,7 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter-out-after "middle.block:" --version 6
-; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-before="scaleMemoryAccessesByUF$" 2>&1 \
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-before="widenMemoryAccessesToVFMultiple$" 2>&1 \
 ; RUN:   | FileCheck --check-prefix=BEFORE-SCALE %s
-; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-after="scaleMemoryAccessesByUF$" 2>&1 \
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-after="widenMemoryAccessesToVFMultiple$" 2>&1 \
 ; RUN:   | FileCheck --check-prefix=AFTER-SCALE %s
 ; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-after="unrollByUF$" 2>&1 \
 ; RUN:   | FileCheck --check-prefix=AFTER-UNROLL %s

>From 5addf07ea52129ead6b9e4571a6b7d2f429f4520 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 28 Sep 2026 15:39:43 +0000
Subject: [PATCH 14/20] Fixups

---
 .../AArch64/multi-vector-mem-ops.ll           | 93 +++++++++++--------
 1 file changed, 53 insertions(+), 40 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
index 398e2e0613e25..3b41923deada9 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -58,7 +58,7 @@ define void @mixed_i64_i32_accesses(ptr noalias %x, ptr noalias %y, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP30]], ptr [[TMP18]], align 4
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -152,7 +152,7 @@ define void @mixed_i32_more_frequent_than_i64(ptr noalias %x, ptr noalias %y, pt
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP43]], ptr [[TMP31]], align 4
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP44:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP44]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP44]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -219,7 +219,7 @@ define void @first_order_recurrence_i64_scaled_load_and_store(ptr noalias %src,
 ; UNMASKED-SVE2P1-NEXT:    store <2 x i64> [[TMP12]], ptr [[TMP16]], align 8
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
 ; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -273,7 +273,7 @@ define i64 @i64_sum_reduction_scaled_partial_reduce(ptr noalias %a, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    [[TMP11]] = add <vscale x 2 x i64> [[VEC_PHI3]], [[TMP7]]
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -329,10 +329,10 @@ define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP10]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = freeze <vscale x 2 x i1> [[TMP11]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = freeze <vscale x 2 x i1> [[TMP12]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = or <vscale x 2 x i1> [[TMP15]], [[TMP16]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = freeze <vscale x 2 x i1> [[TMP13]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = or <vscale x 2 x i1> [[TMP17]], [[TMP18]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = freeze <vscale x 2 x i1> [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = or <vscale x 2 x i1> [[TMP15]], [[TMP16]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = or <vscale x 2 x i1> [[TMP32]], [[TMP18]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = or <vscale x 2 x i1> [[TMP19]], [[TMP20]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call i1 @llvm.vector.reduce.or.nxv2i1(<vscale x 2 x i1> [[TMP21]])
 ; UNMASKED-SVE2P1-NEXT:    [[TMP23]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP11]], <vscale x 2 x i1> [[TMP2]]
@@ -345,7 +345,7 @@ define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP30]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP10]], <vscale x 2 x i64> [[VEC_PHI3]]
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -414,7 +414,7 @@ define void @extending_load(ptr noalias %dst, ptr %src, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP22]], ptr [[TMP18]], align 8
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP23]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP14:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP23]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP17:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -500,7 +500,7 @@ define void @reverse_stride(ptr noalias %c, ptr noalias  %a, ptr noalias %b, i64
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 4 x i32> [[TMP24]], ptr [[TMP32]], align 4
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP17:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -544,16 +544,13 @@ define void @gather_nxv4i32_stride2(ptr noalias %a, ptr noalias %b, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = add i64 [[TMP2]], 0
-; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = mul i64 [[TMP5]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = mul i64 [[TMP2]], 1
 ; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = add i64 [[INDEX]], [[TMP6]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = shl i64 [[TMP2]], 1
-; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = add i64 [[TMP8]], 0
-; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = mul i64 [[TMP9]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = mul i64 [[TMP8]], 1
 ; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add i64 [[INDEX]], [[TMP10]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = mul i64 [[TMP2]], 3
-; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add i64 [[TMP12]], 0
-; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = mul i64 [[TMP13]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = mul i64 [[TMP12]], 1
 ; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = add i64 [[INDEX]], [[TMP14]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = shl i64 [[INDEX]], 1
 ; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = shl i64 [[TMP7]], 1
@@ -583,7 +580,7 @@ define void @gather_nxv4i32_stride2(ptr noalias %a, ptr noalias %b, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x float> [[TMP32]], ptr [[TMP28]], align 4
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP33]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP33]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP23:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -622,38 +619,54 @@ define void @store_address(ptr noalias %x, ptr noalias %y, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
-; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP4]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[STEP_ADD:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
-; UNMASKED-SVE2P1-NEXT:    [[STEP_ADD_2:%.*]] = add nuw <vscale x 2 x i64> [[STEP_ADD]], [[BROADCAST_SPLAT]]
-; UNMASKED-SVE2P1-NEXT:    [[STEP_ADD_3:%.*]] = add nuw <vscale x 2 x i64> [[STEP_ADD_2]], [[BROADCAST_SPLAT]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[VEC_IND]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ], !dbg [[DBG26:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 2 x i64> [ [[TMP4]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ], !dbg [[DBG26]]
+; UNMASKED-SVE2P1-NEXT:    [[STEP_ADD:%.*]] = add nuw <vscale x 2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]], !dbg [[DBG26]]
+; UNMASKED-SVE2P1-NEXT:    [[STEP_ADD_2:%.*]] = add nuw <vscale x 2 x i64> [[STEP_ADD]], [[BROADCAST_SPLAT]], !dbg [[DBG26]]
+; UNMASKED-SVE2P1-NEXT:    [[STEP_ADD_3:%.*]] = add nuw <vscale x 2 x i64> [[STEP_ADD_2]], [[BROADCAST_SPLAT]], !dbg [[DBG26]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[VEC_IND]], !dbg [[DBG30:![0-9]+]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = extractelement <vscale x 2 x ptr> [[WIDE_GEP]], i64 0
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP1:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[STEP_ADD]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP2:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[STEP_ADD_2]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP3:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[STEP_ADD_3]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> poison, <vscale x 2 x ptr> [[WIDE_GEP]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> [[TMP6]], <vscale x 2 x ptr> [[WIDE_GEP1]], i64 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> [[TMP7]], <vscale x 2 x ptr> [[WIDE_GEP2]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> [[TMP8]], <vscale x 2 x ptr> [[WIDE_GEP3]], i64 6)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x ptr> [[TMP9]], ptr [[TMP5]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_IND_NEXT]] = add nuw <vscale x 2 x i64> [[STEP_ADD_3]], [[BROADCAST_SPLAT]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP23:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP1:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[STEP_ADD]], !dbg [[DBG30]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP2:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[STEP_ADD_2]], !dbg [[DBG30]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_GEP3:%.*]] = getelementptr inbounds i64, ptr [[X]], <vscale x 2 x i64> [[STEP_ADD_3]], !dbg [[DBG30]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> poison, <vscale x 2 x ptr> [[WIDE_GEP]], i64 0), !dbg [[DBG31:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> [[TMP6]], <vscale x 2 x ptr> [[WIDE_GEP1]], i64 2), !dbg [[DBG31]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> [[TMP7]], <vscale x 2 x ptr> [[WIDE_GEP2]], i64 4), !dbg [[DBG31]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 8 x ptr> @llvm.vector.insert.nxv8p0.nxv2p0(<vscale x 8 x ptr> [[TMP8]], <vscale x 2 x ptr> [[WIDE_GEP3]], i64 6), !dbg [[DBG31]]
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x ptr> [[TMP9]], ptr [[TMP5]], align 8, !dbg [[DBG31]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]], !dbg [[DBG26]]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_IND_NEXT]] = add nuw <vscale x 2 x i64> [[STEP_ADD_3]], [[BROADCAST_SPLAT]], !dbg [[DBG26]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]], !dbg [[DBG32:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !dbg [[DBG32]], !llvm.loop [[LOOP33:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
   br label %loop
 
 loop:
-  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
-  %px = getelementptr inbounds i64, ptr %x, i64 %iv
-  store ptr %px, ptr %px, align 8
-  %next = add nuw i64 %iv, 1
-  %cmp = icmp ult i64 %next, %n
-  br i1 %cmp, label %loop, label %exit
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ], !dbg !3
+  %px = getelementptr inbounds i64, ptr %x, i64 %iv, !dbg !4
+  store ptr %px, ptr %px, align 8, !dbg !5
+  %next = add nuw i64 %iv, 1, !dbg !6
+  %cmp = icmp ult i64 %next, %n, !dbg !7
+  br i1 %cmp, label %loop, label %exit, !dbg !8
 
 exit:
   ret void
 }
+
+!llvm.dbg.cu = !{!0}
+!llvm.module.flags = !{!2}
+
+!0 = distinct !DICompileUnit(language: DW_LANG_C_plus_plus_14, file: !1, producer: "clang version 24.0.0git")
+!1 = !DIFile(filename: "test.cpp", directory: "/")
+!2 = !{i32 2, !"Debug Info Version", i32 3}
+!3 = !DILocation(line: 4, scope: !9)
+!4 = !DILocation(line: 5, scope: !9)
+!5 = !DILocation(line: 6, scope: !9)
+!6 = !DILocation(line: 7, scope: !9)
+!7 = !DILocation(line: 8, scope: !9)
+!8 = !DILocation(line: 9, scope: !9)
+!9 = distinct !DISubprogram(name: "foo", scope: !1, file: !1, line: 11, type: !10, unit: !0, retainedNodes: !11)
+!10 = distinct !DISubroutineType(types: !11)
+!11 = !{}

>From 88e88966e448af624b5bd8b48ef2171e8980532f Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Wed, 30 Sep 2026 16:48:48 +0000
Subject: [PATCH 15/20] Tweak opcodes

---
 llvm/lib/Transforms/Vectorize/VPlan.h              | 14 ++++++--------
 llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp     | 11 ++++++-----
 llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp  | 12 +++++++-----
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp      | 13 ++++++++++---
 .../AArch64/vplan-printing-multi-vector-mem-ops.ll |  6 +++---
 5 files changed, 32 insertions(+), 24 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index f6b8d13e287b2..7b8a126551237 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1311,15 +1311,13 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
     // WideActiveLaneMask is used for control flow and is unrolled by widening,
     // with one extract vector created per unroll part.
     WideActiveLaneMask,
-    // Signature: (VFMultiple, Address, Alignment) -> Wide Vector
-    // Loads a single wide vector of `VFMultiple * VF` elements. VFMultiple must
-    // divide UF. After unrolling, each section of VF elements in the wide
-    // vector corresponds to an unroll part.
+    // (PreferredVFMultiple, Address, Alignment, VFMultiple) -> Wide Vector
+    // Loads a single wide vector of `VFMultiple * VF` elements. The
+    // `PreferredVFMultiple` is used as a hint to guide unrolling.
     VFMultipleLoad,
-    // Signature: (VFMultiple, Address, Alignment, Vectors...)
-    // Concatenates VFMultiple vector operands into a single wide vector of
-    // `VFMultiple * VF` elements and stores it. After unrolling, each vector
-    // operand corresponds to an unroll part.
+    // (PreferredVFMultiple, Address, Alignment, Vectors...)
+    // Concatenates all vector operands into a single wide vector and stores the
+    // result. The `PreferredVFMultiple` is used as a hint to guide unrolling.
     VFMultipleStore,
     // Extracts each unrolled part of a (VF * UF) widened vector/mask.
     ExtractVectorForPart,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 1e08311386deb..ead3fb8df636a 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -682,8 +682,9 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
   case Instruction::Select:
   case VPInstruction::WideActiveLaneMask:
   case VPInstruction::ReductionStartVector:
-  case VPInstruction::VFMultipleLoad:
     return 3;
+  case VPInstruction::VFMultipleLoad:
+    return 4;
   case Instruction::Call:
     return getCalledFnOperandIndex(operands()) + 1;
   case Instruction::GetElementPtr:
@@ -1190,7 +1191,7 @@ Value *VPInstruction::generate(VPTransformState &State,
                                          /*FMFSource=*/nullptr, getName());
   }
   case VPInstruction::VFMultipleLoad: {
-    unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
+    unsigned VFMultiple = cast<VPConstantInt>(getOperand(3))->getZExtValue();
     auto *WideDataTy = VectorType::get(getScalarType(), State.VF * VFMultiple);
 
     Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
@@ -1199,12 +1200,12 @@ Value *VPInstruction::generate(VPTransformState &State,
                                      "vf.multiple.load");
   }
   case VPInstruction::VFMultipleStore: {
-    unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
+    unsigned NumVectorOps = getNumOperands() - 3;
     Type *ScalarStoreTy = getOperand(3)->getScalarType();
-    auto *WideDataTy = VectorType::get(ScalarStoreTy, State.VF * VFMultiple);
+    auto *WideDataTy = VectorType::get(ScalarStoreTy, State.VF * NumVectorOps);
 
     Value *WideData = PoisonValue::get(WideDataTy);
-    for (unsigned I = 0; I < VFMultiple; ++I) {
+    for (unsigned I = 0; I < NumVectorOps; ++I) {
       Value *Part = State.get(getOperand(I + 3));
       WideData = Builder.CreateInsertVector(WideDataTy, WideData, Part,
                                             I * State.VF.getKnownMinValue());
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 8ea35befd5c7c..ee9db78ca220e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4223,21 +4223,23 @@ void VPlanTransforms::widenMemoryAccessesToVFMultiple(
         continue;
 
       VPValue *Ptr = MemOp->getAddr();
-      VPValue *VFMultipleVPV = Plan.getConstantInt(I64Ty, VFMultiple);
+      VPValue *PreferredVFMultipleVPV = Plan.getConstantInt(I64Ty, VFMultiple);
       VPValue *Align = Plan.getConstantInt(I64Ty, MemOp->getAlign().value());
 
       VPBuilder Builder(VPBB, R.getIterator());
       if (Opcode == Instruction::Load) {
         VPValue *OldLoad = R.getVPSingleValue();
         VPValue *Load = Builder.createNaryOp(
-            VPInstruction::VFMultipleLoad, {VFMultipleVPV, Ptr, Align}, nullptr,
-            {}, *MemOp, R.getDebugLoc(), "", OldLoad->getScalarType());
+            VPInstruction::VFMultipleLoad,
+            {PreferredVFMultipleVPV, Ptr, Align,
+             /*VFMultiple=*/Plan.getConstantInt(I64Ty, 1)},
+            nullptr, {}, *MemOp, R.getDebugLoc(), "", OldLoad->getScalarType());
         OldLoad->replaceAllUsesWith(Load);
       } else {
         assert(Opcode == Instruction::Store);
         Builder.createNaryOp(VPInstruction::VFMultipleStore,
-                             {VFMultipleVPV, Ptr, Align, StoredValue}, nullptr,
-                             {}, *MemOp, R.getDebugLoc());
+                             {PreferredVFMultipleVPV, Ptr, Align, StoredValue},
+                             nullptr, {}, *MemOp, R.getDebugLoc());
       }
 
       R.eraseFromParent();
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 2f298c02c9c66..652c4ece36295 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -298,10 +298,13 @@ void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
           VPI->getOpcode() == VPInstruction::VFMultipleStore) &&
          "expected a vf-multiple load/store");
 
-  unsigned VFMultiple = cast<VPConstantInt>(VPI->getOperand(0))->getZExtValue();
-  assert(VFMultiple > 1 && UF % VFMultiple == 0 &&
-         "expected VFMultiple to divide UF");
+  unsigned PreferredVFMultiple =
+      cast<VPConstantInt>(VPI->getOperand(0))->getZExtValue();
+  assert(PreferredVFMultiple > 1 && UF % PreferredVFMultiple == 0 &&
+         "expected PreferredVFMultiple to divide UF");
 
+  // For now simply follow the preferred VFMultiple.
+  unsigned VFMultiple = PreferredVFMultiple;
   SmallVector<VPInstruction *, 4> Groups(UF / VFMultiple, nullptr);
   Groups[0] = VPI;
 
@@ -332,6 +335,10 @@ void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
     return;
   }
 
+  // Set the VFMultiple to match the PreferredVFMultiple.
+  for (VPInstruction *Load : Groups)
+    Load->setOperand(3, VPI->getOperand(0));
+
   // We need to extract each unroll part as a subvector.
   auto *ExtractPart0 = Builder.createNaryOp(VPInstruction::ExtractVectorForPart,
                                             {Groups[0], getConstantInt(0)});
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
index 82b1201c22629..c56f9e7bc7277 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
@@ -69,7 +69,7 @@ define void @i64_load_store(ptr noalias %x, i64 %n) {
 ; AFTER-SCALE-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
 ; AFTER-SCALE-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
 ; AFTER-SCALE-NEXT:      vp<[[VP9:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-SCALE-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP9]]>, ir<8>
+; AFTER-SCALE-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP9]]>, ir<8>, ir<1>
 ; AFTER-SCALE-NEXT:      WIDEN ir<%add> = add vp<[[VP10]]>, ir<1>
 ; AFTER-SCALE-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
 ; AFTER-SCALE-NEXT:      EMIT vf-multiple store ir<2>, vp<[[VP11]]>, ir<8>, ir<%add>
@@ -108,8 +108,8 @@ define void @i64_load_store(ptr noalias %x, i64 %n) {
 ; AFTER-UNROLL-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
 ; AFTER-UNROLL-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
 ; AFTER-UNROLL-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP9]]>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP10]]>, ir<8>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP11]]>, ir<8>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP10]]>, ir<8>, ir<2>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP11]]>, ir<8>, ir<2>
 ; AFTER-UNROLL-NEXT:      EMIT vp<[[VP14:%[0-9]+]]> = extract-vector-for-part vp<[[VP12]]>, ir<0>
 ; AFTER-UNROLL-NEXT:      EMIT vp<[[VP15:%[0-9]+]]> = extract-vector-for-part vp<[[VP12]]>, ir<1>
 ; AFTER-UNROLL-NEXT:      EMIT vp<[[VP16:%[0-9]+]]> = extract-vector-for-part vp<[[VP13]]>, ir<0>

>From 912791bf5da5f7903069c21dfaef31ff5a9e98b3 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Wed, 30 Sep 2026 17:45:38 +0000
Subject: [PATCH 16/20] Revert "Tweak opcodes"

This reverts commit cad40b17be4157e28b6edb5c6258d5eb8eb3049e.
---
 llvm/lib/Transforms/Vectorize/VPlan.h              | 14 ++++++++------
 llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp     | 11 +++++------
 llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp  | 12 +++++-------
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp      | 13 +++----------
 .../AArch64/vplan-printing-multi-vector-mem-ops.ll |  6 +++---
 5 files changed, 24 insertions(+), 32 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 7b8a126551237..f6b8d13e287b2 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1311,13 +1311,15 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
     // WideActiveLaneMask is used for control flow and is unrolled by widening,
     // with one extract vector created per unroll part.
     WideActiveLaneMask,
-    // (PreferredVFMultiple, Address, Alignment, VFMultiple) -> Wide Vector
-    // Loads a single wide vector of `VFMultiple * VF` elements. The
-    // `PreferredVFMultiple` is used as a hint to guide unrolling.
+    // Signature: (VFMultiple, Address, Alignment) -> Wide Vector
+    // Loads a single wide vector of `VFMultiple * VF` elements. VFMultiple must
+    // divide UF. After unrolling, each section of VF elements in the wide
+    // vector corresponds to an unroll part.
     VFMultipleLoad,
-    // (PreferredVFMultiple, Address, Alignment, Vectors...)
-    // Concatenates all vector operands into a single wide vector and stores the
-    // result. The `PreferredVFMultiple` is used as a hint to guide unrolling.
+    // Signature: (VFMultiple, Address, Alignment, Vectors...)
+    // Concatenates VFMultiple vector operands into a single wide vector of
+    // `VFMultiple * VF` elements and stores it. After unrolling, each vector
+    // operand corresponds to an unroll part.
     VFMultipleStore,
     // Extracts each unrolled part of a (VF * UF) widened vector/mask.
     ExtractVectorForPart,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index ead3fb8df636a..1e08311386deb 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -682,9 +682,8 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
   case Instruction::Select:
   case VPInstruction::WideActiveLaneMask:
   case VPInstruction::ReductionStartVector:
-    return 3;
   case VPInstruction::VFMultipleLoad:
-    return 4;
+    return 3;
   case Instruction::Call:
     return getCalledFnOperandIndex(operands()) + 1;
   case Instruction::GetElementPtr:
@@ -1191,7 +1190,7 @@ Value *VPInstruction::generate(VPTransformState &State,
                                          /*FMFSource=*/nullptr, getName());
   }
   case VPInstruction::VFMultipleLoad: {
-    unsigned VFMultiple = cast<VPConstantInt>(getOperand(3))->getZExtValue();
+    unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
     auto *WideDataTy = VectorType::get(getScalarType(), State.VF * VFMultiple);
 
     Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
@@ -1200,12 +1199,12 @@ Value *VPInstruction::generate(VPTransformState &State,
                                      "vf.multiple.load");
   }
   case VPInstruction::VFMultipleStore: {
-    unsigned NumVectorOps = getNumOperands() - 3;
+    unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
     Type *ScalarStoreTy = getOperand(3)->getScalarType();
-    auto *WideDataTy = VectorType::get(ScalarStoreTy, State.VF * NumVectorOps);
+    auto *WideDataTy = VectorType::get(ScalarStoreTy, State.VF * VFMultiple);
 
     Value *WideData = PoisonValue::get(WideDataTy);
-    for (unsigned I = 0; I < NumVectorOps; ++I) {
+    for (unsigned I = 0; I < VFMultiple; ++I) {
       Value *Part = State.get(getOperand(I + 3));
       WideData = Builder.CreateInsertVector(WideDataTy, WideData, Part,
                                             I * State.VF.getKnownMinValue());
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index ee9db78ca220e..8ea35befd5c7c 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4223,23 +4223,21 @@ void VPlanTransforms::widenMemoryAccessesToVFMultiple(
         continue;
 
       VPValue *Ptr = MemOp->getAddr();
-      VPValue *PreferredVFMultipleVPV = Plan.getConstantInt(I64Ty, VFMultiple);
+      VPValue *VFMultipleVPV = Plan.getConstantInt(I64Ty, VFMultiple);
       VPValue *Align = Plan.getConstantInt(I64Ty, MemOp->getAlign().value());
 
       VPBuilder Builder(VPBB, R.getIterator());
       if (Opcode == Instruction::Load) {
         VPValue *OldLoad = R.getVPSingleValue();
         VPValue *Load = Builder.createNaryOp(
-            VPInstruction::VFMultipleLoad,
-            {PreferredVFMultipleVPV, Ptr, Align,
-             /*VFMultiple=*/Plan.getConstantInt(I64Ty, 1)},
-            nullptr, {}, *MemOp, R.getDebugLoc(), "", OldLoad->getScalarType());
+            VPInstruction::VFMultipleLoad, {VFMultipleVPV, Ptr, Align}, nullptr,
+            {}, *MemOp, R.getDebugLoc(), "", OldLoad->getScalarType());
         OldLoad->replaceAllUsesWith(Load);
       } else {
         assert(Opcode == Instruction::Store);
         Builder.createNaryOp(VPInstruction::VFMultipleStore,
-                             {PreferredVFMultipleVPV, Ptr, Align, StoredValue},
-                             nullptr, {}, *MemOp, R.getDebugLoc());
+                             {VFMultipleVPV, Ptr, Align, StoredValue}, nullptr,
+                             {}, *MemOp, R.getDebugLoc());
       }
 
       R.eraseFromParent();
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 652c4ece36295..2f298c02c9c66 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -298,13 +298,10 @@ void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
           VPI->getOpcode() == VPInstruction::VFMultipleStore) &&
          "expected a vf-multiple load/store");
 
-  unsigned PreferredVFMultiple =
-      cast<VPConstantInt>(VPI->getOperand(0))->getZExtValue();
-  assert(PreferredVFMultiple > 1 && UF % PreferredVFMultiple == 0 &&
-         "expected PreferredVFMultiple to divide UF");
+  unsigned VFMultiple = cast<VPConstantInt>(VPI->getOperand(0))->getZExtValue();
+  assert(VFMultiple > 1 && UF % VFMultiple == 0 &&
+         "expected VFMultiple to divide UF");
 
-  // For now simply follow the preferred VFMultiple.
-  unsigned VFMultiple = PreferredVFMultiple;
   SmallVector<VPInstruction *, 4> Groups(UF / VFMultiple, nullptr);
   Groups[0] = VPI;
 
@@ -335,10 +332,6 @@ void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
     return;
   }
 
-  // Set the VFMultiple to match the PreferredVFMultiple.
-  for (VPInstruction *Load : Groups)
-    Load->setOperand(3, VPI->getOperand(0));
-
   // We need to extract each unroll part as a subvector.
   auto *ExtractPart0 = Builder.createNaryOp(VPInstruction::ExtractVectorForPart,
                                             {Groups[0], getConstantInt(0)});
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
index c56f9e7bc7277..82b1201c22629 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
@@ -69,7 +69,7 @@ define void @i64_load_store(ptr noalias %x, i64 %n) {
 ; AFTER-SCALE-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
 ; AFTER-SCALE-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
 ; AFTER-SCALE-NEXT:      vp<[[VP9:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-SCALE-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP9]]>, ir<8>, ir<1>
+; AFTER-SCALE-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP9]]>, ir<8>
 ; AFTER-SCALE-NEXT:      WIDEN ir<%add> = add vp<[[VP10]]>, ir<1>
 ; AFTER-SCALE-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
 ; AFTER-SCALE-NEXT:      EMIT vf-multiple store ir<2>, vp<[[VP11]]>, ir<8>, ir<%add>
@@ -108,8 +108,8 @@ define void @i64_load_store(ptr noalias %x, i64 %n) {
 ; AFTER-UNROLL-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
 ; AFTER-UNROLL-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
 ; AFTER-UNROLL-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP9]]>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP10]]>, ir<8>, ir<2>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP11]]>, ir<8>, ir<2>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP10]]>, ir<8>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP11]]>, ir<8>
 ; AFTER-UNROLL-NEXT:      EMIT vp<[[VP14:%[0-9]+]]> = extract-vector-for-part vp<[[VP12]]>, ir<0>
 ; AFTER-UNROLL-NEXT:      EMIT vp<[[VP15:%[0-9]+]]> = extract-vector-for-part vp<[[VP12]]>, ir<1>
 ; AFTER-UNROLL-NEXT:      EMIT vp<[[VP16:%[0-9]+]]> = extract-vector-for-part vp<[[VP13]]>, ir<0>

>From 81906ae5308701ae5990616bd5701d05d9a644e4 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Thu, 1 Oct 2026 10:07:02 +0000
Subject: [PATCH 17/20] Simplify operations

---
 .../llvm/Analysis/TargetTransformInfo.h       |  31 ++--
 .../llvm/Analysis/TargetTransformInfoImpl.h   |  14 +-
 llvm/lib/Analysis/TargetTransformInfo.cpp     |  16 +-
 .../AArch64/AArch64TargetTransformInfo.cpp    |  52 +++----
 .../AArch64/AArch64TargetTransformInfo.h      |   9 +-
 .../Transforms/Vectorize/LoopVectorize.cpp    |   6 +-
 llvm/lib/Transforms/Vectorize/VPlan.h         |  21 ++-
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp |  76 +++++-----
 .../Transforms/Vectorize/VPlanTransforms.cpp  |  57 ++++---
 .../Transforms/Vectorize/VPlanTransforms.h    |   5 +-
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp |  76 ++--------
 ...-vector-mem-ops-non-power-of-two-unroll.ll |   2 +-
 .../AArch64/multi-vector-mem-ops.ll           | 142 ++++++++----------
 .../vplan-printing-multi-vector-mem-ops.ll    |  45 +++---
 14 files changed, 232 insertions(+), 320 deletions(-)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index b6cfebaef50d0..a15237f31b5f7 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1001,20 +1001,25 @@ class TargetTransformInfo {
                                 unsigned Opcode1,
                                 const SmallBitVector &OpcodeMask) const;
 
-  /// Returns the maximum VF multiple that can be used for a contiguous
-  /// load/store for the given \p VF and \p UF.
-  LLVM_ABI unsigned getMaximumVFMultipleForMemoryOp(ElementCount VF,
-                                                    unsigned UF) const;
-
-  /// Return the preferred multiple of VF to use for a contiguous load/store.
-  /// Returning 1 leaves the operation at VF. The returned value must divide UF.
-  /// \p CastHint is non-null if the stored value is produced by a cast
-  /// instruction or the loaded value is consumed by one.
+  enum MaskSource {
+    /// The operation is unmasked.
+    MS_None,
+    /// The operation is masked with an arbitrary predicate.
+    MS_Masked,
+    /// The operation is masked with a contiguous active lane mask.
+    MS_ActiveLaneMask,
+  };
+
+  /// Return true if the target supports loading or storing \p NumVectors
+  /// contiguous vectors of type \p VectorTy as a single operation. A null
+  /// \p VectorTy queries whether the target supports this operation in
+  /// general.
   ///
-  /// \p Opcode must be either Instruction::Load or Instruction::Store.
-  LLVM_ABI unsigned getPreferredVFMultipleForMemoryOp(
-      unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
-      bool IsMasked = false,
+  /// \p CastHint is non-null if a stored value is produced by a cast or a
+  /// loaded value is consumed by one.
+  LLVM_ABI bool hasMultipleVectorLoadStore(
+      unsigned NumVectors, VectorType *VectorTy = nullptr, bool IsStore = false,
+      MaskSource Mask = MS_None,
       std::optional<Instruction::CastOps> CastHint = std::nullopt) const;
 
   /// Return true if we should be enabling ordered reductions for the target.
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 57073b8d87a10..cd4acc0731c7d 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -429,15 +429,11 @@ class LLVM_ABI TargetTransformInfoImplBase {
     return false;
   }
 
-  virtual unsigned getMaximumVFMultipleForMemoryOp(ElementCount VF,
-                                                   unsigned UF) const {
-    return 1;
-  }
-
-  virtual unsigned getPreferredVFMultipleForMemoryOp(
-      unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
-      bool IsMasked, std::optional<Instruction::CastOps> CastHint) const {
-    return 1;
+  virtual bool hasMultipleVectorLoadStore(
+      unsigned NumVectors, VectorType *VectorTy, bool IsStore,
+      TTI::MaskSource Mask,
+      std::optional<Instruction::CastOps> CastHint) const {
+    return false;
   }
 
   virtual bool isLegalInterleavedAccessType(VectorType *VTy, unsigned Factor,
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 16477e2d58282..99a6202fc2d7a 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -554,17 +554,11 @@ bool TargetTransformInfo::isLegalStridedLoadStore(Type *DataType,
   return TTIImpl->isLegalStridedLoadStore(DataType, Alignment);
 }
 
-unsigned
-TargetTransformInfo::getMaximumVFMultipleForMemoryOp(ElementCount VF,
-                                                     unsigned UF) const {
-  return TTIImpl->getMaximumVFMultipleForMemoryOp(VF, UF);
-}
-
-unsigned TargetTransformInfo::getPreferredVFMultipleForMemoryOp(
-    unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
-    bool IsMasked, std::optional<Instruction::CastOps> CastHint) const {
-  return TTIImpl->getPreferredVFMultipleForMemoryOp(Opcode, DataType, VF, UF,
-                                                    IsMasked, CastHint);
+bool TargetTransformInfo::hasMultipleVectorLoadStore(
+    unsigned NumVectors, VectorType *VectorTy, bool IsStore,
+    TTI::MaskSource Mask, std::optional<Instruction::CastOps> CastHint) const {
+  return TTIImpl->hasMultipleVectorLoadStore(NumVectors, VectorTy, IsStore,
+                                             Mask, CastHint);
 }
 
 bool TargetTransformInfo::isLegalInterleavedAccessType(
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 6125860e7b565..12945052e1244 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -5959,46 +5959,36 @@ bool AArch64TTIImpl::isLegalSpeculativeLoad(Type *DataType,
          Size.getFixedValue() <= 16;
 }
 
-unsigned AArch64TTIImpl::getMaximumVFMultipleForMemoryOp(ElementCount VF,
-                                                         unsigned UF) const {
-  if (!ST->enableSubRegLiveness())
-    return 1;
+bool AArch64TTIImpl::hasMultipleVectorLoadStore(
+    unsigned NumVectors, VectorType *VectorTy, bool IsStore,
+    TTI::MaskSource Mask, std::optional<Instruction::CastOps> CastHint) const {
+  if (NumVectors <= 1 || !ST->enableSubRegLiveness() || !ST->hasSVE2p1())
+    return false;
 
-  if (!ST->hasSVE2p1() || !VF.isScalable() || !isPowerOf2_32(UF))
-    return 1;
+  // TODO: Support masked multi-vector loads/stores.
+  if (Mask != TTI::MS_None)
+    return false;
 
-  // +sve2p1 multi-vector loads/stores can handle up to four vectors.
-  return std::min(4U, UF);
-}
+  // A null vector type queries whether the target supports multi-vector memory
+  // operations in general.
+  if (!VectorTy)
+    return true;
 
-unsigned AArch64TTIImpl::getPreferredVFMultipleForMemoryOp(
-    unsigned Opcode, Type *DataTy, ElementCount VF, unsigned UF, bool IsMasked,
-    std::optional<Instruction::CastOps> CastHint) const {
-  assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
-         "expected load/store opcode");
-  if (IsMasked)
-    return 1; // TODO: Support masked multi-vector loads/stores.
+  if (!isa<ScalableVectorType>(VectorTy))
+    return false;
 
   // Conservatively, avoid using multi-vector loads when it's possible we could
   // use extending loads instead. Note: We can ignore stores as we only use
   // truncating stores when the store vector-width is < a full SVE vector.
-  if (Opcode == Instruction::Load &&
+  if (!IsStore &&
       (CastHint == Instruction::ZExt || CastHint == Instruction::SExt))
-    return 1;
-
-  unsigned VectorWidth = VF.getKnownMinValue() * DL.getTypeSizeInBits(DataTy);
-  if (VectorWidth % 128 != 0)
-    return 1;
-
-  for (unsigned TargetWidth : {512u, 256u}) {
-    if (TargetWidth % VectorWidth == 0) {
-      unsigned Scale = TargetWidth / VectorWidth;
-      if (Scale <= UF)
-        return Scale;
-    }
-  }
+    return false;
 
-  return 1;
+  // For unpredicated loads/stores allow any pow-of-two multiple of a vector >=
+  // to a single z-register. We can split operations wider than a single
+  // multi-vector load/store during ISEL.
+  unsigned VectorWidth = DL.getTypeSizeInBits(VectorTy).getKnownMinValue();
+  return VectorWidth % 128 == 0 && isPowerOf2_32(NumVectors);
 }
 
 unsigned
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index 042774c539a44..010b1f9142edd 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -282,12 +282,9 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
   bool isLegalSpeculativeLoad(Type *DataType,
                               unsigned AddressSpace) const override;
 
-  unsigned getMaximumVFMultipleForMemoryOp(ElementCount VF,
-                                           unsigned UF) const override;
-
-  unsigned getPreferredVFMultipleForMemoryOp(
-      unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
-      bool IsMasked,
+  bool hasMultipleVectorLoadStore(
+      unsigned NumVectors, VectorType *VectorTy, bool IsStore,
+      TTI::MaskSource Mask,
       std::optional<Instruction::CastOps> CastHint) const override;
 
   void getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 4ecef9d2e02bc..0fcbdabeb256e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5752,9 +5752,9 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
                  *PSE.getSE(), TTI, Config.CostKind, BestVF, BestUF);
   // TODO: Move to VPlan transform stage once the transition to the VPlan-based
   // cost model is complete for better cost estimates.
-  if (TTI.getMaximumVFMultipleForMemoryOp(BestVF, BestUF) > 1)
-    RUN_VPLAN_PASS(VPlanTransforms::widenMemoryAccessesToVFMultiple, BestVPlan,
-                   BestVF, BestUF, TTI);
+  if (TTI.hasMultipleVectorLoadStore(BestUF))
+    RUN_VPLAN_PASS(VPlanTransforms::widenMemoryAccessesByUF, BestVPlan, BestVF,
+                   BestUF, TTI);
   RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
   RUN_VPLAN_PASS(VPlanTransforms::materializePacksAndUnpacks, BestVPlan);
   RUN_VPLAN_PASS(VPlanTransforms::materializeBroadcasts, BestVPlan);
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index f6b8d13e287b2..17c3cf9b2c9c6 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1311,16 +1311,15 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
     // WideActiveLaneMask is used for control flow and is unrolled by widening,
     // with one extract vector created per unroll part.
     WideActiveLaneMask,
-    // Signature: (VFMultiple, Address, Alignment) -> Wide Vector
-    // Loads a single wide vector of `VFMultiple * VF` elements. VFMultiple must
-    // divide UF. After unrolling, each section of VF elements in the wide
-    // vector corresponds to an unroll part.
-    VFMultipleLoad,
-    // Signature: (VFMultiple, Address, Alignment, Vectors...)
-    // Concatenates VFMultiple vector operands into a single wide vector of
-    // `VFMultiple * VF` elements and stores it. After unrolling, each vector
-    // operand corresponds to an unroll part.
-    VFMultipleStore,
+    // Signature: Vectors... -> WideVector
+    // Concatenates all vector operands to a single wide vector.
+    ConcatVectors,
+    // Signature: (Multiplier, Address, Align) -> Vector
+    // Loads a single wide vector of `Multiplier * VF` elements.
+    WideVectorLoad,
+    // Signature: (Multiplier, Address, Alignment, Vector)
+    // Stores a single wide vector of `Multiplier * VF` elements.
+    WideVectorStore,
     // Extracts each unrolled part of a (VF * UF) widened vector/mask.
     ExtractVectorForPart,
     ExplicitVectorLength,
@@ -1525,7 +1524,7 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
     case VPInstruction::BranchOnCond:
     case VPInstruction::BranchOnTwoConds:
     case VPInstruction::BranchOnCount:
-    case VPInstruction::VFMultipleStore:
+    case VPInstruction::WideVectorStore:
       return false;
     default:
       return true;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 1e08311386deb..e7d7bc200db37 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -60,7 +60,7 @@ bool VPRecipeBase::mayWriteToMemory() const {
     auto *VPI = cast<VPInstruction>(this);
     // Loads read from memory but don't write to memory.
     if (VPI->getOpcode() == Instruction::Load ||
-        VPI->getOpcode() == VPInstruction::VFMultipleLoad)
+        VPI->getOpcode() == VPInstruction::WideVectorLoad)
       return false;
     return VPI->opcodeMayReadOrWriteFromMemory();
   }
@@ -122,7 +122,7 @@ bool VPRecipeBase::mayReadFromMemory() const {
   case VPInstructionSC: {
     auto *VPI = cast<VPInstruction>(this);
     // Stores write to memory but don't read from memory.
-    if (VPI->getOpcode() == VPInstruction::VFMultipleStore)
+    if (VPI->getOpcode() == VPInstruction::WideVectorStore)
       return false;
     return VPI->opcodeMayReadOrWriteFromMemory();
   }
@@ -497,7 +497,7 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
     for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
       AssertOperandType(Idx, Op0Ty);
     return Type::getVoidTy(Ctx);
-  case VPInstruction::VFMultipleStore:
+  case VPInstruction::WideVectorStore:
   case Instruction::Store:
     return Type::getVoidTy(Ctx);
   case Instruction::ICmp:
@@ -573,7 +573,7 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
   }
   case VPInstruction::ExtractVectorForPart:
     return Op0Ty;
-  case VPInstruction::VFMultipleLoad:
+  case VPInstruction::WideVectorLoad:
   case VPInstruction::FirstActiveLane:
   case VPInstruction::LastActiveLane:
   case VPInstruction::NumActiveLanes:
@@ -595,7 +595,8 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
       Instruction::isBinaryOp(Opcode) ||
       is_contained({VPInstruction::FirstOrderRecurrenceSplice,
                     VPInstruction::BuildVector,
-                    VPInstruction::BuildStructVector},
+                    VPInstruction::BuildStructVector,
+                    VPInstruction::ConcatVectors},
                    Opcode);
   if (AllOperandsSameType)
     for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
@@ -682,8 +683,10 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
   case Instruction::Select:
   case VPInstruction::WideActiveLaneMask:
   case VPInstruction::ReductionStartVector:
-  case VPInstruction::VFMultipleLoad:
+  case VPInstruction::WideVectorLoad:
     return 3;
+  case VPInstruction::WideVectorStore:
+    return 4;
   case Instruction::Call:
     return getCalledFnOperandIndex(operands()) + 1;
   case Instruction::GetElementPtr:
@@ -702,7 +705,7 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
   case VPInstruction::LastActiveLane:
   case VPInstruction::ExtractLane:
   case VPInstruction::ExtractLastActive:
-  case VPInstruction::VFMultipleStore:
+  case VPInstruction::ConcatVectors:
     // Cannot determine the number of operands from the opcode.
     return -1u;
   }
@@ -960,6 +963,15 @@ Value *VPInstruction::generate(VPTransformState &State,
                                         Builder.getInt64(Idx));
     return Res;
   }
+  case VPInstruction::ConcatVectors: {
+    Type *ScalarTy = getScalarType();
+    auto *WideTy = VectorType::get(ScalarTy, State.VF * getNumOperands());
+    Value *Res = PoisonValue::get(WideTy);
+    for (const auto &[Idx, Op] : enumerate(operands()))
+      Res = Builder.CreateInsertVector(WideTy, Res, State.get(Op),
+                                       Idx * State.VF.getKnownMinValue());
+    return Res;
+  }
   case VPInstruction::ReductionStartVector: {
     if (State.VF.isScalar())
       return State.get(getOperand(0), true);
@@ -1189,26 +1201,21 @@ Value *VPInstruction::generate(VPTransformState &State,
                                          vputils::getIntrinsicID(this), Args,
                                          /*FMFSource=*/nullptr, getName());
   }
-  case VPInstruction::VFMultipleLoad: {
-    unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
-    auto *WideDataTy = VectorType::get(getScalarType(), State.VF * VFMultiple);
+  case VPInstruction::WideVectorLoad: {
+    unsigned Multiplier = cast<VPConstantInt>(getOperand(0))->getZExtValue();
+    auto *WideDataTy = VectorType::get(getScalarType(), State.VF * Multiplier);
 
     Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
     Align Alignment = Align(cast<VPConstantInt>(getOperand(2))->getZExtValue());
-    return Builder.CreateAlignedLoad(WideDataTy, Addr, Alignment,
-                                     "vf.multiple.load");
-  }
-  case VPInstruction::VFMultipleStore: {
-    unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
-    Type *ScalarStoreTy = getOperand(3)->getScalarType();
-    auto *WideDataTy = VectorType::get(ScalarStoreTy, State.VF * VFMultiple);
-
-    Value *WideData = PoisonValue::get(WideDataTy);
-    for (unsigned I = 0; I < VFMultiple; ++I) {
-      Value *Part = State.get(getOperand(I + 3));
-      WideData = Builder.CreateInsertVector(WideDataTy, WideData, Part,
-                                            I * State.VF.getKnownMinValue());
-    }
+    return Builder.CreateAlignedLoad(WideDataTy, Addr, Alignment);
+  }
+  case VPInstruction::WideVectorStore: {
+    unsigned Multiplier = cast<VPConstantInt>(getOperand(0))->getZExtValue();
+    auto *WideDataTy =
+        VectorType::get(getOperand(3)->getScalarType(), State.VF * Multiplier);
+    Value *WideData = State.get(getOperand(3));
+    assert(WideData->getType() == WideDataTy &&
+           "stored value does not match wide vector type");
 
     Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
     Align Alignment = Align(cast<VPConstantInt>(getOperand(2))->getZExtValue());
@@ -1677,6 +1684,7 @@ void VPInstruction::addOperand(VPValue *Op) {
   case VPInstruction::ComputeReductionResult:
   case VPInstruction::BuildVector:
   case VPInstruction::BuildStructVector:
+  case VPInstruction::ConcatVectors:
     assert(Ty == getOperand(0)->getScalarType() &&
            "appended operand must match operand 0's scalar type");
     break;
@@ -1697,10 +1705,6 @@ void VPInstruction::addOperand(VPValue *Op) {
            "matching operand 1's type and i1, respectively");
     break;
   }
-  case VPInstruction::VFMultipleStore:
-    assert(Ty == getOperand(3)->getScalarType() &&
-           "appended operand must match operand 3's scalar type");
-    break;
   default:
     llvm_unreachable("opcode does not support growing the operand list "
                      "outside of construction");
@@ -1758,6 +1762,7 @@ bool VPInstruction::opcodeMayReadOrWriteFromMemory() const {
   case VPInstruction::Broadcast:
   case VPInstruction::BuildStructVector:
   case VPInstruction::BuildVector:
+  case VPInstruction::ConcatVectors:
   case VPInstruction::CanonicalIVIncrementForPart:
   case VPInstruction::ComputeReductionResult:
   case VPInstruction::ExtractLane:
@@ -1835,7 +1840,7 @@ bool VPInstruction::usesFirstLaneOnly(const VPValue *Op) const {
   case VPInstruction::Intrinsic:
   case VPInstruction::ReductionStartVector:
   case VPInstruction::ResumeForEpilogue:
-  case VPInstruction::VFMultipleLoad:
+  case VPInstruction::WideVectorLoad:
     return true;
   case VPInstruction::BuildStructVector:
   case VPInstruction::BuildVector:
@@ -1848,7 +1853,7 @@ bool VPInstruction::usesFirstLaneOnly(const VPValue *Op) const {
   case VPInstruction::WidePtrAdd:
     // WidePtrAdd supports scalar and vector base addresses.
     return false;
-  case VPInstruction::VFMultipleStore:
+  case VPInstruction::WideVectorStore:
     return Op == getOperand(0) || Op == getOperand(1) || Op == getOperand(2);
   case VPInstruction::ExitingIVValue:
   case VPInstruction::ExtractLane:
@@ -1903,11 +1908,14 @@ void VPInstruction::printRecipe(raw_ostream &O, const Twine &Indent,
   case VPInstruction::WideActiveLaneMask:
     O << "wide active lane mask";
     break;
-  case VPInstruction::VFMultipleLoad:
-    O << "vf-multiple load";
+  case VPInstruction::WideVectorLoad:
+    O << "wide vector load";
+    break;
+  case VPInstruction::WideVectorStore:
+    O << "wide vector store";
     break;
-  case VPInstruction::VFMultipleStore:
-    O << "vf-multiple store";
+  case VPInstruction::ConcatVectors:
+    O << "concat-vectors";
     break;
   case VPInstruction::IncomingAliasMask:
     O << "incoming-alias-mask";
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 8ea35befd5c7c..d98f885073360 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4179,19 +4179,17 @@ void VPlanTransforms::sinkPredicatedStores(VPlan &Plan,
   }
 }
 
-void VPlanTransforms::widenMemoryAccessesToVFMultiple(
-    VPlan &Plan, ElementCount VF, unsigned UF, const TargetTransformInfo &TTI) {
+void VPlanTransforms::widenMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
+                                              unsigned UF,
+                                              const TargetTransformInfo &TTI) {
   assert(UF > 1 && "Expected plan to have an UF > 1");
-
-  Type *I64Ty = IntegerType::getInt64Ty(Plan.getContext());
   for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
            vp_depth_first_shallow(Plan.getVectorLoopRegion()->getEntry()))) {
     for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
       VPValue *StoredValue = nullptr;
       auto m_ContiguousVecPtr = m_VecPtr(m_VPValue(), m_One());
-      if ((!match(&R, m_WidenLoad(m_ContiguousVecPtr)) &&
-           !match(&R,
-                  m_WidenStore(m_ContiguousVecPtr, m_VPValue(StoredValue)))))
+      if (!match(&R, m_WidenLoad(m_ContiguousVecPtr)) &&
+          !match(&R, m_WidenStore(m_ContiguousVecPtr, m_VPValue(StoredValue))))
         continue;
 
       auto *MemOp = cast<VPWidenMemoryRecipe>(&R);
@@ -4203,41 +4201,40 @@ void VPlanTransforms::widenMemoryAccessesToVFMultiple(
 
       Type *AccessType = StoredValue ? StoredValue->getScalarType()
                                      : R.getVPSingleValue()->getScalarType();
-      unsigned Opcode = isa<VPWidenLoadRecipe>(MemOp->getAsRecipe())
-                            ? Instruction::Load
-                            : Instruction::Store;
-
+      unsigned IsStore = isa<VPWidenStoreRecipe>(MemOp->getAsRecipe());
       std::optional<Instruction::CastOps> CastHint;
-      VPUser *MaybeCast = Opcode == Instruction::Store
-                              ? StoredValue->getDefiningRecipe()
-                              : R.getVPSingleValue()->getSingleUser();
+      VPUser *MaybeCast = IsStore ? StoredValue->getDefiningRecipe()
+                                  : R.getVPSingleValue()->getSingleUser();
       if (auto *Cast = dyn_cast_if_present<VPWidenCastRecipe>(MaybeCast))
         CastHint = Cast->getOpcode();
 
-      unsigned VFMultiple = TTI.getPreferredVFMultipleForMemoryOp(
-          Opcode, AccessType, VF, UF, /*IsMasked=*/false, CastHint);
-      assert((VFMultiple != 0 && UF % VFMultiple == 0) &&
-             "VFMultiple must divide UF");
-
-      if (VFMultiple == 1)
+      VectorType *VectorAccessType = VectorType::get(AccessType, VF);
+      if (!TTI.hasMultipleVectorLoadStore(
+              /*NumVectors=*/UF, VectorAccessType, IsStore,
+              TargetTransformInfo::MaskSource::MS_None, CastHint))
         continue;
 
+      DebugLoc DL = R.getDebugLoc();
       VPValue *Ptr = MemOp->getAddr();
-      VPValue *VFMultipleVPV = Plan.getConstantInt(I64Ty, VFMultiple);
-      VPValue *Align = Plan.getConstantInt(I64Ty, MemOp->getAlign().value());
+      VPValue *Align = Plan.getConstantInt(64, MemOp->getAlign().value());
+      VPValue *Multiplier = Plan.getConstantInt(64, 1);
 
       VPBuilder Builder(VPBB, R.getIterator());
-      if (Opcode == Instruction::Load) {
+      if (IsStore) {
+        VPValue *WideStoredValue = Builder.createNaryOp(
+            VPInstruction::ConcatVectors, {StoredValue}, DL);
+        Builder.createNaryOp(VPInstruction::WideVectorStore,
+                             {Multiplier, Ptr, Align, WideStoredValue}, nullptr,
+                             {}, *MemOp, R.getDebugLoc());
+      } else {
         VPValue *OldLoad = R.getVPSingleValue();
         VPValue *Load = Builder.createNaryOp(
-            VPInstruction::VFMultipleLoad, {VFMultipleVPV, Ptr, Align}, nullptr,
+            VPInstruction::WideVectorLoad, {Multiplier, Ptr, Align}, nullptr,
             {}, *MemOp, R.getDebugLoc(), "", OldLoad->getScalarType());
-        OldLoad->replaceAllUsesWith(Load);
-      } else {
-        assert(Opcode == Instruction::Store);
-        Builder.createNaryOp(VPInstruction::VFMultipleStore,
-                             {VFMultipleVPV, Ptr, Align, StoredValue}, nullptr,
-                             {}, *MemOp, R.getDebugLoc());
+        VPValue *Extract =
+            Builder.createNaryOp(VPInstruction::ExtractVectorForPart,
+                                 {Load, Plan.getConstantInt(64, 0)}, DL);
+        OldLoad->replaceAllUsesWith(Extract);
       }
 
       R.eraseFromParent();
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 9683947802304..c801871370a01 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -484,9 +484,8 @@ struct VPlanTransforms {
 
   /// Widens memory operations by a factor of UF based on a target hook.
   /// This allows targets to use wider memory operations when profitable.
-  static void widenMemoryAccessesToVFMultiple(VPlan &Plan, ElementCount VF,
-                                              unsigned UF,
-                                              const TargetTransformInfo &TTI);
+  static void widenMemoryAccessesByUF(VPlan &Plan, ElementCount VF, unsigned UF,
+                                      const TargetTransformInfo &TTI);
 
   // Materialize vector trip counts for constants early if it can simply be
   // computed as (Original TC / VF * UF) * VF * UF.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 2f298c02c9c66..47d23bdbd1d66 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -74,9 +74,6 @@ class UnrollState {
     return Plan.getConstantInt(CanIVIntTy, Part);
   }
 
-  /// Unroll a VFMultipleLoad or VFMultipleStore VPInstruction.
-  void unrollMemOpWithVFMultiple(VPInstruction *VPI);
-
 public:
   UnrollState(VPlan &Plan, unsigned UF) : Plan(Plan), UF(UF) {}
 
@@ -293,76 +290,12 @@ void UnrollState::unrollHeaderPHIByUF(VPHeaderPHIRecipe *R,
   }
 }
 
-void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
-  assert((VPI->getOpcode() == VPInstruction::VFMultipleLoad ||
-          VPI->getOpcode() == VPInstruction::VFMultipleStore) &&
-         "expected a vf-multiple load/store");
-
-  unsigned VFMultiple = cast<VPConstantInt>(VPI->getOperand(0))->getZExtValue();
-  assert(VFMultiple > 1 && UF % VFMultiple == 0 &&
-         "expected VFMultiple to divide UF");
-
-  SmallVector<VPInstruction *, 4> Groups(UF / VFMultiple, nullptr);
-  Groups[0] = VPI;
-
-  // A memory op with a VFMultiple is widened to VF * VFMultiple elements, so
-  // after unrolling by UF we materialize UF / VFMultiple such ops, each
-  // covering VFMultiple unroll parts.
-  VPBuilder Builder = VPBuilder::getToInsertAfter(VPI);
-  for (unsigned Group = 1; Group < Groups.size(); ++Group) {
-    auto *Copy = Builder.insert(VPI->clone());
-    remapOperands(Copy, Group * VFMultiple);
-    Groups[Group] = Copy;
-  }
-
-  if (VPI->getOpcode() == VPInstruction::VFMultipleStore) {
-    VPValue *StoredValue = VPI->getOperand(3);
-    for (unsigned Group = 0; Group < Groups.size(); ++Group) {
-      VPInstruction *Store = Groups[Group];
-      // Add the value to store for each unroll part in this group.
-      for (unsigned Part = 0; Part < VFMultiple; ++Part) {
-        VPValue *UnrollPart =
-            getValueForPart(StoredValue, Group * VFMultiple + Part);
-        if (Part == 0)
-          Store->setOperand(3, UnrollPart);
-        else
-          Store->addOperand(UnrollPart);
-      }
-    }
-    return;
-  }
-
-  // We need to extract each unroll part as a subvector.
-  auto *ExtractPart0 = Builder.createNaryOp(VPInstruction::ExtractVectorForPart,
-                                            {Groups[0], getConstantInt(0)});
-  // First VPI with an extract of the first unroll part (ExtractPart0).
-  VPI->replaceUsesWithIf(
-      ExtractPart0, [&](VPUser &U, unsigned) { return &U != ExtractPart0; });
-
-  // Create extracts for the remaining unroll parts and remap later uses of
-  // ExtractPart0 to the correct unrolled part.
-  for (unsigned Part = 1; Part != UF; ++Part) {
-    VPInstruction *Group = Groups[Part / VFMultiple];
-    unsigned IndexInGroup = Part % VFMultiple;
-    auto *Extract = Builder.createNaryOp(
-        VPInstruction::ExtractVectorForPart,
-        {Group->getVPSingleValue(), getConstantInt(IndexInGroup)});
-    addRecipeForPart(ExtractPart0, Extract, Part);
-  }
-}
-
 /// Handle non-header-phi recipes.
 void UnrollState::unrollRecipeByUF(VPRecipeBase &R) {
   if (match(&R, m_CombineOr(m_BranchOnCond(), m_BranchOnCount())))
     return;
 
   if (auto *VPI = dyn_cast<VPInstruction>(&R)) {
-    if (VPI->getOpcode() == VPInstruction::VFMultipleLoad ||
-        VPI->getOpcode() == VPInstruction::VFMultipleStore) {
-      unrollMemOpWithVFMultiple(VPI);
-      return;
-    }
-
     if (vputils::onlyFirstPartUsed(VPI)) {
       addUniformForAllParts(VPI);
       return;
@@ -496,6 +429,8 @@ void UnrollState::unrollBlock(VPBlockBase *VPB) {
     // value.
     VPValue *Op1;
     if (match(&R, m_VPInstruction<VPInstruction::AnyOf>(m_VPValue(Op1))) ||
+        match(&R,
+              m_VPInstruction<VPInstruction::ConcatVectors>(m_VPValue(Op1))) ||
         match(&R, m_FirstActiveLane(m_VPValue(Op1))) ||
         match(&R, m_LastActiveLane(m_VPValue(Op1))) ||
         match(&R, m_ComputeReductionResult(m_VPValue(Op1)))) {
@@ -553,6 +488,13 @@ void UnrollState::unrollBlock(VPBlockBase *VPB) {
       continue;
     }
 
+    if (match(&R,
+              m_CombineOr(m_VPInstruction<VPInstruction::WideVectorLoad>(),
+                          m_VPInstruction<VPInstruction::WideVectorStore>()))) {
+      cast<VPInstruction>(&R)->setOperand(0, Plan.getConstantInt(64, UF));
+      continue;
+    }
+
     auto *SingleDef = dyn_cast<VPSingleDefRecipe>(&R);
     if (SingleDef && vputils::isUniformAcrossVFsAndUFs(SingleDef)) {
       addUniformForAllParts(SingleDef);
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll
index 529909d353e17..4d2ecb32f9326 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll
@@ -3,7 +3,7 @@
 
 target triple = "aarch64-unknown-linux-gnu"
 
-; Tests `widenMemoryAccessesToVFMultiple` with a non-power-of-two unroll factor.
+; Tests `widenMemoryAccessesByUF` with a non-power-of-two unroll factor.
 ; On AArch64, this should be rejected (which results in `vscale x 4` loads/stores).
 
 define void @mixed_i64_i32_accesses(ptr noalias %x, ptr noalias %y, i64 %n) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
index 3b41923deada9..f7c2390e258a9 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -16,46 +16,41 @@ define void @mixed_i64_i32_accesses(ptr noalias %x, ptr noalias %y, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[UMAX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
 ; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[UMAX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[UMAX]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw nsw i64 [[TMP2]], 1
-; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[TMP3]], i64 [[TMP4]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP3]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = load <vscale x 16 x i64>, ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv16i64(<vscale x 16 x i64> [[TMP4]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv16i64(<vscale x 16 x i64> [[TMP4]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv16i64(<vscale x 16 x i64> [[TMP4]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv16i64(<vscale x 16 x i64> [[TMP4]], i64 12)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = add <vscale x 4 x i64> [[TMP6]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <vscale x 4 x i64> [[TMP7]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <vscale x 4 x i64> [[TMP8]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add <vscale x 4 x i64> [[TMP9]], splat (i64 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP14]], <vscale x 4 x i64> [[TMP11]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP15]], ptr [[TMP3]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP13]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP17]], ptr [[TMP5]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> [[TMP30]], <vscale x 4 x i64> [[TMP12]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> [[TMP14]], <vscale x 4 x i64> [[TMP13]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i64> [[TMP15]], ptr [[TMP3]], align 8
 ; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP18]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = load <vscale x 16 x i32>, ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP17]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP17]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP17]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP17]], i64 12)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = add <vscale x 4 x i32> [[TMP19]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = add <vscale x 4 x i32> [[TMP20]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = add <vscale x 4 x i32> [[TMP21]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = add <vscale x 4 x i32> [[TMP22]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP23]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP27]], <vscale x 4 x i32> [[TMP24]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP28]], <vscale x 4 x i32> [[TMP25]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 12)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP30]], ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP23]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP32]], <vscale x 4 x i32> [[TMP24]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP27]], <vscale x 4 x i32> [[TMP25]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP28]], <vscale x 4 x i32> [[TMP26]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP29]], ptr [[TMP18]], align 4
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
@@ -95,52 +90,47 @@ define void @mixed_i32_more_frequent_than_i64(ptr noalias %x, ptr noalias %y, pt
 ; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[UMAX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
 ; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[UMAX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[UMAX]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw nsw i64 [[TMP2]], 1
-; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[TMP3]], i64 [[TMP4]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP3]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = load <vscale x 16 x i64>, ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv16i64(<vscale x 16 x i64> [[TMP4]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv16i64(<vscale x 16 x i64> [[TMP4]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv16i64(<vscale x 16 x i64> [[TMP4]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv16i64(<vscale x 16 x i64> [[TMP4]], i64 12)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = add <vscale x 4 x i64> [[TMP6]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <vscale x 4 x i64> [[TMP7]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <vscale x 4 x i64> [[TMP8]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add <vscale x 4 x i64> [[TMP9]], splat (i64 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP14]], <vscale x 4 x i64> [[TMP11]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP15]], ptr [[TMP3]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP13]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP17]], ptr [[TMP5]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> [[TMP30]], <vscale x 4 x i64> [[TMP12]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> [[TMP14]], <vscale x 4 x i64> [[TMP13]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i64> [[TMP15]], ptr [[TMP3]], align 8
 ; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP18]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = load <vscale x 16 x i32>, ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP17]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP17]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP17]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP17]], i64 12)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = add <vscale x 4 x i32> [[TMP19]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = add <vscale x 4 x i32> [[TMP20]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = add <vscale x 4 x i32> [[TMP21]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = add <vscale x 4 x i32> [[TMP22]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP23]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP27]], <vscale x 4 x i32> [[TMP24]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP28]], <vscale x 4 x i32> [[TMP25]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 12)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP30]], ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP45:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP23]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP45]], <vscale x 4 x i32> [[TMP24]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP27]], <vscale x 4 x i32> [[TMP25]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP28]], <vscale x 4 x i32> [[TMP26]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP29]], ptr [[TMP18]], align 4
 ; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = getelementptr inbounds i32, ptr [[Z]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD4:%.*]] = load <vscale x 16 x i32>, ptr [[TMP31]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP34:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP35:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP46:%.*]] = load <vscale x 16 x i32>, ptr [[TMP31]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP46]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP46]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP34:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP46]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP35:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[TMP46]], i64 12)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP36:%.*]] = add <vscale x 4 x i32> [[TMP32]], splat (i32 2)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP37:%.*]] = add <vscale x 4 x i32> [[TMP33]], splat (i32 2)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP38:%.*]] = add <vscale x 4 x i32> [[TMP34]], splat (i32 2)
@@ -262,11 +252,11 @@ define i64 @i64_sum_reduction_scaled_partial_reduce(ptr noalias %a, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP11:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP3]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = load <vscale x 8 x i64>, ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[TMP13]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[TMP13]], i64 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[TMP13]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[TMP13]], i64 6)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP8]] = add <vscale x 2 x i64> [[VEC_PHI]], [[TMP4]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP9]] = add <vscale x 2 x i64> [[VEC_PHI1]], [[TMP5]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_PHI2]], [[TMP6]]
@@ -318,11 +308,11 @@ define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP25:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP26:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i64, ptr [[DATA]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP6]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = load <vscale x 8 x i64>, ptr [[TMP6]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[TMP17]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[TMP17]], i64 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[TMP17]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[TMP17]], i64 6)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP7]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP8]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP9]]
@@ -405,13 +395,11 @@ define void @extending_load(ptr noalias %dst, ptr %src, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = mul nsw <vscale x 4 x i64> [[TMP11]], splat (i64 42)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = mul nsw <vscale x 4 x i64> [[TMP12]], splat (i64 42)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = getelementptr inbounds nuw i64, ptr [[DST]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = getelementptr inbounds nuw i64, ptr [[TMP17]], i64 [[TMP4]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP13]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP19]], <vscale x 4 x i64> [[TMP14]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP20]], ptr [[TMP17]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP15]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP21]], <vscale x 4 x i64> [[TMP16]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP22]], ptr [[TMP18]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> poison, <vscale x 4 x i64> [[TMP13]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> [[TMP18]], <vscale x 4 x i64> [[TMP14]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> [[TMP19]], <vscale x 4 x i64> [[TMP15]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 16 x i64> @llvm.vector.insert.nxv16i64.nxv4i64(<vscale x 16 x i64> [[TMP20]], <vscale x 4 x i64> [[TMP16]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i64> [[TMP21]], ptr [[TMP17]], align 8
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP23]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP17:![0-9]+]]
@@ -573,11 +561,11 @@ define void @gather_nxv4i32_stride2(ptr noalias %a, ptr noalias %b, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    [[STRIDED_VEC7:%.*]] = call { <vscale x 4 x float>, <vscale x 4 x float> } @llvm.vector.deinterleave2.nxv8f32(<vscale x 8 x float> [[WIDE_VEC6]])
 ; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = extractvalue { <vscale x 4 x float>, <vscale x 4 x float> } [[STRIDED_VEC7]], 0
 ; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> poison, <vscale x 4 x float> [[TMP24]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP29]], <vscale x 4 x float> [[TMP25]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP30]], <vscale x 4 x float> [[TMP26]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP31]], <vscale x 4 x float> [[TMP27]], i64 12)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x float> [[TMP32]], ptr [[TMP28]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> poison, <vscale x 4 x float> [[TMP24]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP30]], <vscale x 4 x float> [[TMP25]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP31]], <vscale x 4 x float> [[TMP26]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP32]], <vscale x 4 x float> [[TMP27]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x float> [[TMP29]], ptr [[TMP28]], align 4
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP33]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP23:![0-9]+]]
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
index 82b1201c22629..05464ccc940a4 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
@@ -1,7 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter-out-after "middle.block:" --version 6
-; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-before="widenMemoryAccessesToVFMultiple$" 2>&1 \
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-before="widenMemoryAccessesByUF$" 2>&1 \
 ; RUN:   | FileCheck --check-prefix=BEFORE-SCALE %s
-; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-after="widenMemoryAccessesToVFMultiple$" 2>&1 \
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-after="widenMemoryAccessesByUF$" 2>&1 \
 ; RUN:   | FileCheck --check-prefix=AFTER-SCALE %s
 ; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-after="unrollByUF$" 2>&1 \
 ; RUN:   | FileCheck --check-prefix=AFTER-UNROLL %s
@@ -69,10 +69,12 @@ define void @i64_load_store(ptr noalias %x, i64 %n) {
 ; AFTER-SCALE-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
 ; AFTER-SCALE-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
 ; AFTER-SCALE-NEXT:      vp<[[VP9:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-SCALE-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP9]]>, ir<8>
-; AFTER-SCALE-NEXT:      WIDEN ir<%add> = add vp<[[VP10]]>, ir<1>
-; AFTER-SCALE-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-SCALE-NEXT:      EMIT vf-multiple store ir<2>, vp<[[VP11]]>, ir<8>, ir<%add>
+; AFTER-SCALE-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = wide vector load ir<1>, vp<[[VP9]]>, ir<8>
+; AFTER-SCALE-NEXT:      EMIT vp<[[VP11:%[0-9]+]]> = extract-vector-for-part vp<[[VP10]]>, ir<0>
+; AFTER-SCALE-NEXT:      WIDEN ir<%add> = add vp<[[VP11]]>, ir<1>
+; AFTER-SCALE-NEXT:      vp<[[VP12:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-SCALE-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = concat-vectors ir<%add>
+; AFTER-SCALE-NEXT:      EMIT wide vector store ir<1>, vp<[[VP12]]>, ir<8>, vp<[[VP13]]>
 ; AFTER-SCALE-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
 ; AFTER-SCALE-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; AFTER-SCALE-NEXT:    No successors
@@ -105,24 +107,19 @@ define void @i64_load_store(ptr noalias %x, i64 %n) {
 ; AFTER-UNROLL-NEXT:    vector.body:
 ; AFTER-UNROLL-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
 ; AFTER-UNROLL-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
-; AFTER-UNROLL-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-UNROLL-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP9]]>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP10]]>, ir<8>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP11]]>, ir<8>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP14:%[0-9]+]]> = extract-vector-for-part vp<[[VP12]]>, ir<0>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP15:%[0-9]+]]> = extract-vector-for-part vp<[[VP12]]>, ir<1>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP16:%[0-9]+]]> = extract-vector-for-part vp<[[VP13]]>, ir<0>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP17:%[0-9]+]]> = extract-vector-for-part vp<[[VP13]]>, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add> = add vp<[[VP14]]>, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.1 = add vp<[[VP15]]>, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.2 = add vp<[[VP16]]>, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.3 = add vp<[[VP17]]>, ir<1>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP18:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
-; AFTER-UNROLL-NEXT:      vp<[[VP19:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-UNROLL-NEXT:      vp<[[VP20:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP18]]>
-; AFTER-UNROLL-NEXT:      EMIT vf-multiple store ir<2>, vp<[[VP19]]>, ir<8>, ir<%add>, ir<%add>.1
-; AFTER-UNROLL-NEXT:      EMIT vf-multiple store ir<2>, vp<[[VP20]]>, ir<8>, ir<%add>.2, ir<%add>.3
+; AFTER-UNROLL-NEXT:      vp<[[VP9:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = wide vector load ir<4>, vp<[[VP9]]>, ir<8>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP11:%[0-9]+]]> = extract-vector-for-part vp<[[VP10]]>, ir<0>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = extract-vector-for-part vp<[[VP10]]>, ir<1>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = extract-vector-for-part vp<[[VP10]]>, ir<2>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP14:%[0-9]+]]> = extract-vector-for-part vp<[[VP10]]>, ir<3>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add> = add vp<[[VP11]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.1 = add vp<[[VP12]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.2 = add vp<[[VP13]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.3 = add vp<[[VP14]]>, ir<1>
+; AFTER-UNROLL-NEXT:      vp<[[VP15:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP16:%[0-9]+]]> = concat-vectors ir<%add>, ir<%add>.1, ir<%add>.2, ir<%add>.3
+; AFTER-UNROLL-NEXT:      EMIT wide vector store ir<4>, vp<[[VP15]]>, ir<8>, vp<[[VP16]]>
 ; AFTER-UNROLL-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
 ; AFTER-UNROLL-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; AFTER-UNROLL-NEXT:    No successors

>From fb5d24194f051e3b381ba3b475d15d06d67ef15d Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Thu, 1 Oct 2026 12:17:35 +0000
Subject: [PATCH 18/20] Rebase fixups

---
 .../AArch64/multi-vector-mem-ops.ll              | 16 ++++++++--------
 1 file changed, 8 insertions(+), 8 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
index f7c2390e258a9..e0b7897250634 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -303,10 +303,10 @@ define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold)
 ; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP28:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP29:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP30:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP23:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP24:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP25:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP26:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP24:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP25:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP26:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP33:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i64, ptr [[DATA]], i64 [[INDEX]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = load <vscale x 8 x i64>, ptr [[TMP6]], align 8
 ; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[TMP17]], i64 0)
@@ -325,10 +325,10 @@ define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = or <vscale x 2 x i1> [[TMP32]], [[TMP18]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = or <vscale x 2 x i1> [[TMP19]], [[TMP20]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call i1 @llvm.vector.reduce.or.nxv2i1(<vscale x 2 x i1> [[TMP21]])
-; UNMASKED-SVE2P1-NEXT:    [[TMP23]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP11]], <vscale x 2 x i1> [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP24]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP12]], <vscale x 2 x i1> [[TMP3]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP25]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP13]], <vscale x 2 x i1> [[TMP4]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP26]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP14]], <vscale x 2 x i1> [[TMP5]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP24]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP15]], <vscale x 2 x i1> [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP25]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP16]], <vscale x 2 x i1> [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP26]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP18]], <vscale x 2 x i1> [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP33]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP20]], <vscale x 2 x i1> [[TMP5]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP27]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP7]], <vscale x 2 x i64> [[VEC_PHI]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP28]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP8]], <vscale x 2 x i64> [[VEC_PHI1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP29]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP9]], <vscale x 2 x i64> [[VEC_PHI2]]

>From 666cf08969a8819030a4c4937fab169b6580feb7 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Fri, 2 Oct 2026 08:55:56 +0000
Subject: [PATCH 19/20] Fixups

---
 llvm/include/llvm/Analysis/TargetTransformInfo.h   | 14 +++++++-------
 .../llvm/Analysis/TargetTransformInfoImpl.h        |  8 ++++----
 llvm/lib/Analysis/TargetTransformInfo.cpp          | 10 +++++-----
 .../Target/AArch64/AArch64TargetTransformInfo.cpp  | 12 ++++++------
 .../Target/AArch64/AArch64TargetTransformInfo.h    |  6 +++---
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp    |  8 +++++---
 llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp  |  6 +++---
 7 files changed, 33 insertions(+), 31 deletions(-)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index a15237f31b5f7..dcc0a8c406ec7 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1001,13 +1001,13 @@ class TargetTransformInfo {
                                 unsigned Opcode1,
                                 const SmallBitVector &OpcodeMask) const;
 
-  enum MaskSource {
+  enum class MaskSource {
     /// The operation is unmasked.
-    MS_None,
+    None,
     /// The operation is masked with an arbitrary predicate.
-    MS_Masked,
+    ArbitraryPredicate,
     /// The operation is masked with a contiguous active lane mask.
-    MS_ActiveLaneMask,
+    ActiveLaneMask,
   };
 
   /// Return true if the target supports loading or storing \p NumVectors
@@ -1017,9 +1017,9 @@ class TargetTransformInfo {
   ///
   /// \p CastHint is non-null if a stored value is produced by a cast or a
   /// loaded value is consumed by one.
-  LLVM_ABI bool hasMultipleVectorLoadStore(
-      unsigned NumVectors, VectorType *VectorTy = nullptr, bool IsStore = false,
-      MaskSource Mask = MS_None,
+  LLVM_ABI bool hasMultiVectorLoadStore(
+      unsigned NumVectors, MaskSource Mask, VectorType *VectorTy = nullptr,
+      bool IsStore = false,
       std::optional<Instruction::CastOps> CastHint = std::nullopt) const;
 
   /// Return true if we should be enabling ordered reductions for the target.
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index cd4acc0731c7d..004602490fbbd 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -429,10 +429,10 @@ class LLVM_ABI TargetTransformInfoImplBase {
     return false;
   }
 
-  virtual bool hasMultipleVectorLoadStore(
-      unsigned NumVectors, VectorType *VectorTy, bool IsStore,
-      TTI::MaskSource Mask,
-      std::optional<Instruction::CastOps> CastHint) const {
+  virtual bool
+  hasMultiVectorLoadStore(unsigned NumVectors, TTI::MaskSource Mask,
+                          VectorType *VectorTy, bool IsStore,
+                          std::optional<Instruction::CastOps> CastHint) const {
     return false;
   }
 
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 99a6202fc2d7a..df91c170ed915 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -554,11 +554,11 @@ bool TargetTransformInfo::isLegalStridedLoadStore(Type *DataType,
   return TTIImpl->isLegalStridedLoadStore(DataType, Alignment);
 }
 
-bool TargetTransformInfo::hasMultipleVectorLoadStore(
-    unsigned NumVectors, VectorType *VectorTy, bool IsStore,
-    TTI::MaskSource Mask, std::optional<Instruction::CastOps> CastHint) const {
-  return TTIImpl->hasMultipleVectorLoadStore(NumVectors, VectorTy, IsStore,
-                                             Mask, CastHint);
+bool TargetTransformInfo::hasMultiVectorLoadStore(
+    unsigned NumVectors, TTI::MaskSource Mask, VectorType *VectorTy,
+    bool IsStore, std::optional<Instruction::CastOps> CastHint) const {
+  return TTIImpl->hasMultiVectorLoadStore(NumVectors, Mask, VectorTy, IsStore,
+                                          CastHint);
 }
 
 bool TargetTransformInfo::isLegalInterleavedAccessType(
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 12945052e1244..12e4b493903cb 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -5959,14 +5959,14 @@ bool AArch64TTIImpl::isLegalSpeculativeLoad(Type *DataType,
          Size.getFixedValue() <= 16;
 }
 
-bool AArch64TTIImpl::hasMultipleVectorLoadStore(
-    unsigned NumVectors, VectorType *VectorTy, bool IsStore,
-    TTI::MaskSource Mask, std::optional<Instruction::CastOps> CastHint) const {
+bool AArch64TTIImpl::hasMultiVectorLoadStore(
+    unsigned NumVectors, TTI::MaskSource Mask, VectorType *VectorTy,
+    bool IsStore, std::optional<Instruction::CastOps> CastHint) const {
   if (NumVectors <= 1 || !ST->enableSubRegLiveness() || !ST->hasSVE2p1())
     return false;
 
   // TODO: Support masked multi-vector loads/stores.
-  if (Mask != TTI::MS_None)
+  if (Mask != TTI::MaskSource::None)
     return false;
 
   // A null vector type queries whether the target supports multi-vector memory
@@ -5987,8 +5987,8 @@ bool AArch64TTIImpl::hasMultipleVectorLoadStore(
   // For unpredicated loads/stores allow any pow-of-two multiple of a vector >=
   // to a single z-register. We can split operations wider than a single
   // multi-vector load/store during ISEL.
-  unsigned VectorWidth = DL.getTypeSizeInBits(VectorTy).getKnownMinValue();
-  return VectorWidth % 128 == 0 && isPowerOf2_32(NumVectors);
+  return DL.getTypeSizeInBits(VectorTy).isKnownMultipleOf(128) &&
+         isPowerOf2_32(NumVectors);
 }
 
 unsigned
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index 010b1f9142edd..b1aec1c391eff 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -282,9 +282,9 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
   bool isLegalSpeculativeLoad(Type *DataType,
                               unsigned AddressSpace) const override;
 
-  bool hasMultipleVectorLoadStore(
-      unsigned NumVectors, VectorType *VectorTy, bool IsStore,
-      TTI::MaskSource Mask,
+  bool hasMultiVectorLoadStore(
+      unsigned NumVectors, TTI::MaskSource Mask, VectorType *VectorTy,
+      bool IsStore,
       std::optional<Instruction::CastOps> CastHint) const override;
 
   void getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 0fcbdabeb256e..e6ad6cecc5969 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5750,11 +5750,13 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
 
   RUN_VPLAN_PASS(VPlanTransforms::replaceWideCanonicalIVWithWideIV, BestVPlan,
                  *PSE.getSE(), TTI, Config.CostKind, BestVF, BestUF);
-  // TODO: Move to VPlan transform stage once the transition to the VPlan-based
-  // cost model is complete for better cost estimates.
-  if (TTI.hasMultipleVectorLoadStore(BestUF))
+  if (!BestVPlan.hasTailFolded() &&
+      TTI.hasMultiVectorLoadStore(BestUF,
+                                  TargetTransformInfo::MaskSource::None))
     RUN_VPLAN_PASS(VPlanTransforms::widenMemoryAccessesByUF, BestVPlan, BestVF,
                    BestUF, TTI);
+  // TODO: Move to VPlan transform stage once the transition to the VPlan-based
+  // cost model is complete for better cost estimates.
   RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
   RUN_VPLAN_PASS(VPlanTransforms::materializePacksAndUnpacks, BestVPlan);
   RUN_VPLAN_PASS(VPlanTransforms::materializeBroadcasts, BestVPlan);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index d98f885073360..b7a0f626fbbb5 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4209,9 +4209,9 @@ void VPlanTransforms::widenMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
         CastHint = Cast->getOpcode();
 
       VectorType *VectorAccessType = VectorType::get(AccessType, VF);
-      if (!TTI.hasMultipleVectorLoadStore(
-              /*NumVectors=*/UF, VectorAccessType, IsStore,
-              TargetTransformInfo::MaskSource::MS_None, CastHint))
+      if (!TTI.hasMultiVectorLoadStore(
+              /*NumVectors=*/UF, TargetTransformInfo::MaskSource::None,
+              VectorAccessType, IsStore, CastHint))
         continue;
 
       DebugLoc DL = R.getDebugLoc();

>From 9792a7e00d9d3b58baf4ed2e49af5b6c70331131 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Fri, 2 Oct 2026 08:59:03 +0000
Subject: [PATCH 20/20] Add comment

---
 llvm/include/llvm/Analysis/TargetTransformInfo.h | 1 +
 1 file changed, 1 insertion(+)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index dcc0a8c406ec7..80066cf05f7e2 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1001,6 +1001,7 @@ class TargetTransformInfo {
                                 unsigned Opcode1,
                                 const SmallBitVector &OpcodeMask) const;
 
+  /// Enum describing the source/producer of a mask.
   enum class MaskSource {
     /// The operation is unmasked.
     None,



More information about the llvm-commits mailing list