[llvm] [LV] Add support for widening loads/stores to a VF multiple (PR #217670)

Benjamin Maxwell via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 11 03:29:04 PDT 2026


https://github.com/MacDue updated https://github.com/llvm/llvm-project/pull/217670

>From 0dcec2a7e6910095ea299b24ba90ec81cc8caa22 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Thu, 20 Aug 2026 14:39:53 +0000
Subject: [PATCH 01/11] [LV] Add support for widening loads/stores to a VF
 multiple

This patch adds support for widening loads and stores by a "VF multiple"
that must divide the UF. It is currently limited to unmasked operations,
but we plan to extend it to masked loops via the wide active lane mask.

For now, this is driven by a new TTI hook,
`getPreferredVFMultipleForMemoryOp`. A small VPlan transform uses that
hook to set the VF multiple on `VPWidenLoadRecipe` and
`VPWidenStoreRecipe`.

When VPlan unrolling encounters a load or store with a VF multiple
greater than 1, it inserts extracts for the unroll parts of widened
loads and concatenates multiple unroll parts for widened stores.

On AArch64 this is used to target multi-vector load/store instructions.
---
 .../llvm/Analysis/TargetTransformInfo.h       |   9 +
 .../llvm/Analysis/TargetTransformInfoImpl.h   |   8 +
 llvm/lib/Analysis/TargetTransformInfo.cpp     |   7 +
 .../AArch64/AArch64TargetTransformInfo.cpp    |  31 ++
 .../AArch64/AArch64TargetTransformInfo.h      |   4 +
 .../Transforms/Vectorize/LoopVectorize.cpp    |   2 +
 llvm/lib/Transforms/Vectorize/VPlan.h         |  27 +-
 .../Transforms/Vectorize/VPlanPatternMatch.h  |  37 +-
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp |  28 +-
 .../Transforms/Vectorize/VPlanTransforms.cpp  |  41 ++
 .../Transforms/Vectorize/VPlanTransforms.h    |   5 +
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp |  66 ++++
 .../AArch64/multi-vector-mem-ops.ll           | 374 ++++++++++++++++++
 .../vplan-printing-multi-vector-mem-ops.ll    | 151 +++++++
 .../VPlan/vplan-print-before-after-all.ll     |   1 +
 15 files changed, 784 insertions(+), 7 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index cf5e940eeb1f4..4d0962cd420da 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -979,6 +979,15 @@ class TargetTransformInfo {
                                 unsigned Opcode1,
                                 const SmallBitVector &OpcodeMask) const;
 
+  /// Return the preferred multiple of VF to use for a contiguous load/store.
+  /// Returning 1 leaves the operation at VF. The returned value must divide UF.
+  ///
+  /// \p Opcode must be either Instruction::Load or Instruction::Store.
+  LLVM_ABI unsigned
+  getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataType,
+                                    ElementCount VF, unsigned UF,
+                                    bool IsMasked = false) const;
+
   /// Return true if we should be enabling ordered reductions for the target.
   LLVM_ABI bool enableOrderedReductions() const;
 
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index d9e7f82496d49..9e23ec8a42b18 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -422,6 +422,14 @@ class LLVM_ABI TargetTransformInfoImplBase {
     return false;
   }
 
+  virtual unsigned getPreferredVFMultipleForMemoryOp(unsigned Opcode,
+                                                     Type *DataType,
+                                                     ElementCount VF,
+                                                     unsigned UF,
+                                                     bool IsMasked) const {
+    return 1;
+  }
+
   virtual bool isLegalInterleavedAccessType(VectorType *VTy, unsigned Factor,
                                             Align Alignment,
                                             unsigned AddrSpace) const {
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 71f55f2e7e046..8bdc293478fee 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -547,6 +547,13 @@ bool TargetTransformInfo::isLegalStridedLoadStore(Type *DataType,
   return TTIImpl->isLegalStridedLoadStore(DataType, Alignment);
 }
 
+unsigned TargetTransformInfo::getPreferredVFMultipleForMemoryOp(
+    unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
+    bool IsMasked) const {
+  return TTIImpl->getPreferredVFMultipleForMemoryOp(Opcode, DataType, VF, UF,
+                                                    IsMasked);
+}
+
 bool TargetTransformInfo::isLegalInterleavedAccessType(
     VectorType *VTy, unsigned Factor, Align Alignment,
     unsigned AddrSpace) const {
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index d872c2ec34c25..c7d27eaadd1c0 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -5859,6 +5859,37 @@ bool AArch64TTIImpl::isLegalMaskedExpandLoad(Type *DataTy,
          (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2());
 }
 
+unsigned
+AArch64TTIImpl::getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataTy,
+                                                  ElementCount VF, unsigned UF,
+                                                  bool IsMasked) const {
+  assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
+         "expected load/store opcode");
+  if (IsMasked)
+    return 1; // TODO: Support masked multi-vector loads/stores.
+
+  if (!ST->enableSubRegLiveness())
+    return 1;
+
+  if ((Opcode != Instruction::Load && Opcode != Instruction::Store) ||
+      !ST->hasSVE2p1() || !VF.isScalable() || !isPowerOf2_32(UF))
+    return 1;
+
+  unsigned VectorWidth = VF.getKnownMinValue() * DL.getTypeSizeInBits(DataTy);
+  if (VectorWidth % 128 != 0)
+    return 1;
+
+  for (unsigned TargetWidth : {512u, 256u}) {
+    if (TargetWidth % VectorWidth == 0) {
+      unsigned Scale = TargetWidth / VectorWidth;
+      if (Scale <= UF)
+        return Scale;
+    }
+  }
+
+  return 1;
+}
+
 unsigned
 AArch64TTIImpl::getMaxInterleaveFactor(ElementCount VF,
                                        bool HasUnorderedReductions) const {
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index dc840f5c7565f..ea0f3039f1285 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -279,6 +279,10 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
 
   bool isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const override;
 
+  unsigned getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataType,
+                                             ElementCount VF, unsigned UF,
+                                             bool IsMasked) const override;
+
   void getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
                                TTI::UnrollingPreferences &UP,
                                OptimizationRemarkEmitter *ORE) const override;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 04abdac5ee680..514bb051d9ae6 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5754,6 +5754,8 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
                  *PSE.getSE(), TTI, Config.CostKind, BestVF, BestUF);
   // TODO: Move to VPlan transform stage once the transition to the VPlan-based
   // cost model is complete for better cost estimates.
+  RUN_VPLAN_PASS(VPlanTransforms::scaleMemoryAccessesByUF, BestVPlan, BestVF,
+                 BestUF, CM.TTI);
   RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
   RUN_VPLAN_PASS(VPlanTransforms::materializePacksAndUnpacks, BestVPlan);
   RUN_VPLAN_PASS(VPlanTransforms::materializeBroadcasts, BestVPlan);
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 2393403b3a837..ce31583b4c3a5 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1325,6 +1325,8 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
     WideActiveLaneMask,
     // Extracts each unrolled part of a (VF * UF) widened vector/mask.
     ExtractVectorForPart,
+    // Concatenates its unrolled part operands into one widened vector.
+    ConcatVectorParts,
     ExplicitVectorLength,
     // Represents the incoming loop-invariant alias-mask. All memory accesses
     // in the loop must stay within the active lanes.
@@ -3761,6 +3763,10 @@ class LLVM_ABI_FOR_TEST VPWidenMemoryRecipe : public VPIRMetadata {
   /// Whether the memory access is masked.
   bool IsMasked = false;
 
+  /// Multiple of VF used to widen this memory operation. The final operation
+  /// loads or stores VF * VFMultiple elements
+  unsigned VFMultiple = 1;
+
   void setMask(VPValue *Mask) {
     assert(!IsMasked && "cannot re-set mask");
     if (!Mask)
@@ -3807,6 +3813,12 @@ class LLVM_ABI_FOR_TEST VPWidenMemoryRecipe : public VPIRMetadata {
   InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const;
 
   Instruction &getIngredient() const { return Ingredient; }
+
+  /// Set the VF multiple for this memory operation.
+  void setVFMultiple(unsigned VFMultiple) { this->VFMultiple = VFMultiple; }
+
+  /// Returns the VF multiple of this memory operation.
+  unsigned getVFMultiple() const { return VFMultiple; }
 };
 
 /// A recipe for widening load operations, using the address to load from and an
@@ -3822,8 +3834,11 @@ struct LLVM_ABI_FOR_TEST VPWidenLoadRecipe final : public VPSingleDefRecipe,
   }
 
   VPWidenLoadRecipe *clone() override {
-    return new VPWidenLoadRecipe(cast<LoadInst>(Ingredient), getAddr(),
-                                 getMask(), Consecutive, *this, getDebugLoc());
+    auto *R =
+        new VPWidenLoadRecipe(cast<LoadInst>(Ingredient), getAddr(), getMask(),
+                              Consecutive, *this, getDebugLoc());
+    R->setVFMultiple(VFMultiple);
+    return R;
   }
 
   VP_CLASSOF_IMPL(VPRecipeBase::VPWidenLoadSC);
@@ -3927,9 +3942,11 @@ struct LLVM_ABI_FOR_TEST VPWidenStoreRecipe final : public VPRecipeBase,
   }
 
   VPWidenStoreRecipe *clone() override {
-    return new VPWidenStoreRecipe(cast<StoreInst>(Ingredient), getAddr(),
-                                  getStoredValue(), getMask(), Consecutive,
-                                  *this, getDebugLoc());
+    auto *R = new VPWidenStoreRecipe(cast<StoreInst>(Ingredient), getAddr(),
+                                     getStoredValue(), getMask(), Consecutive,
+                                     *this, getDebugLoc());
+    R->setVFMultiple(VFMultiple);
+    return R;
   }
 
   VP_CLASSOF_IMPL(VPRecipeBase::VPWidenStoreSC);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
index d5fd4c2ffb511..4023ccd6d76e0 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
@@ -292,7 +292,10 @@ struct Recipe_match {
     // Check for recipes that do not have opcodes.
     if constexpr (std::is_same_v<RecipeTy, VPScalarIVStepsRecipe> ||
                   std::is_same_v<RecipeTy, VPDerivedIVRecipe> ||
-                  std::is_same_v<RecipeTy, VPVectorEndPointerRecipe>)
+                  std::is_same_v<RecipeTy, VPVectorEndPointerRecipe> ||
+                  std::is_same_v<RecipeTy, VPVectorPointerRecipe> ||
+                  std::is_same_v<RecipeTy, VPWidenLoadRecipe> ||
+                  std::is_same_v<RecipeTy, VPWidenStoreRecipe>)
       return DefR;
     else
       return DefR && DefR->getOpcode() == Opcode;
@@ -1025,6 +1028,38 @@ m_MaskedStore(const Addr_t &Addr, const Val_t &Val, const Mask_t &Mask) {
   return Store_match<Addr_t, Val_t, Mask_t>(Addr, Val, Mask);
 }
 
+template <typename Op0_t, typename Op1_t>
+using VectorPointerRecipe_match =
+    Recipe_match<std::tuple<Op0_t, Op1_t>, 0,
+                 /*Commutative*/ false, VPVectorPointerRecipe>;
+
+template <typename Op0_t, typename Op1_t>
+VectorPointerRecipe_match<Op0_t, Op1_t> m_VecPtr(const Op0_t &Op0,
+                                                 const Op1_t &Op1) {
+  return VectorPointerRecipe_match<Op0_t, Op1_t>(Op0, Op1);
+}
+
+template <typename Op0_t>
+using VPWidenLoadRecipe_match =
+    Recipe_match<std::tuple<Op0_t>, 0,
+                 /*Commutative*/ false, VPWidenLoadRecipe>;
+
+template <typename Op0_t>
+VPWidenLoadRecipe_match<Op0_t> m_WidenLoad(const Op0_t &Op0) {
+  return VPWidenLoadRecipe_match<Op0_t>(Op0);
+}
+
+template <typename Op0_t, typename Op1_t>
+using VPWidenStoreRecipe_match =
+    Recipe_match<std::tuple<Op0_t, Op1_t>, 0,
+                 /*Commutative*/ false, VPWidenStoreRecipe>;
+
+template <typename Op0_t, typename Op1_t>
+VPWidenStoreRecipe_match<Op0_t, Op1_t> m_WidenStore(const Op0_t &Op0,
+                                                    const Op1_t &Op1) {
+  return VPWidenStoreRecipe_match<Op0_t, Op1_t>(Op0, Op1);
+}
+
 template <typename Op0_t, typename Op1_t>
 using VectorEndPointerRecipe_match =
     Recipe_match<std::tuple<Op0_t, Op1_t>, 0,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 049e8b70c70d8..b8155e9a2bdf2 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -570,6 +570,9 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
     return StructTy->getTypeAtIndex(
         cast<VPConstantInt>(Operands[1])->getZExtValue());
   }
+  case VPInstruction::ConcatVectorParts:
+  case VPInstruction::ExtractVectorForPart:
+    return Op0Ty;
   case VPInstruction::FirstActiveLane:
   case VPInstruction::LastActiveLane:
   case VPInstruction::NumActiveLanes:
@@ -697,6 +700,7 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
   case VPInstruction::LastActiveLane:
   case VPInstruction::ExtractLane:
   case VPInstruction::ExtractLastActive:
+  case VPInstruction::ConcatVectorParts:
     // Cannot determine the number of operands from the opcode.
     return -1u;
   }
@@ -1145,6 +1149,19 @@ Value *VPInstruction::generate(VPTransformState &State) {
                                          vputils::getIntrinsicID(this), Args,
                                          /*FMFSource=*/nullptr, getName());
   }
+  case VPInstruction::ConcatVectorParts: {
+    unsigned VectorOps = getNumOperands();
+    auto *WideDataTy = VectorType::get(
+        getScalarType(), State.VF.multiplyCoefficientBy(VectorOps));
+    Value *WideData = PoisonValue::get(WideDataTy);
+
+    for (unsigned I = 0; I < VectorOps; ++I) {
+      Value *Part = State.get(getOperand(I));
+      WideData = Builder.CreateInsertVector(WideDataTy, WideData, Part,
+                                            I * State.VF.getKnownMinValue());
+    }
+    return WideData;
+  }
   default:
     llvm_unreachable("Unsupported opcode for instruction");
   }
@@ -1693,6 +1710,7 @@ bool VPInstruction::opcodeMayReadOrWriteFromMemory() const {
   case VPInstruction::ActiveLaneMask:
   case VPInstruction::WideActiveLaneMask:
   case VPInstruction::IncomingAliasMask:
+  case VPInstruction::ConcatVectorParts:
   case VPInstruction::ExitingIVValue:
   case VPInstruction::ExplicitVectorLength:
   case VPInstruction::FirstActiveLane:
@@ -1827,6 +1845,9 @@ void VPInstruction::printRecipe(raw_ostream &O, const Twine &Indent,
   case VPInstruction::IncomingAliasMask:
     O << "incoming-alias-mask";
     break;
+  case VPInstruction::ConcatVectorParts:
+    O << "concat-vector-parts";
+    break;
   case VPInstruction::ExplicitVectorLength:
     O << "EXPLICIT-VECTOR-LENGTH";
     break;
@@ -4341,7 +4362,8 @@ InstructionCost VPWidenMemoryRecipe::computeCost(ElementCount VF,
 
 void VPWidenLoadRecipe::execute(VPTransformState &State) {
   Type *ScalarDataTy = getScalarType();
-  auto *DataTy = VectorType::get(ScalarDataTy, State.VF);
+  auto *DataTy =
+      VectorType::get(ScalarDataTy, State.VF.multiplyCoefficientBy(VFMultiple));
   bool CreateGather = !isConsecutive();
 
   auto &Builder = State.Builder;
@@ -4371,6 +4393,8 @@ void VPWidenLoadRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
   O << Indent << "WIDEN ";
   printAsOperand(O, SlotTracker);
   O << " = load ";
+  if (VFMultiple > 1)
+    O << "x" << VFMultiple << ' ';
   printOperands(O, SlotTracker);
 }
 #endif
@@ -4458,6 +4482,8 @@ void VPWidenStoreRecipe::execute(VPTransformState &State) {
 void VPWidenStoreRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
                                      VPSlotTracker &SlotTracker) const {
   O << Indent << "WIDEN store ";
+  if (VFMultiple > 1)
+    O << "x" << VFMultiple << ' ';
   printOperands(O, SlotTracker);
 }
 #endif
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 831eb957c8caf..f89356f6f5815 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -3989,6 +3989,47 @@ void VPlanTransforms::sinkPredicatedStores(VPlan &Plan,
   }
 }
 
+void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
+                                              unsigned UF,
+                                              const TargetTransformInfo &TTI) {
+  if (UF == 1)
+    return;
+
+  for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
+           vp_depth_first_deep(Plan.getVectorLoopRegion()->getEntry()))) {
+    for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
+      uint64_t Stride;
+      VPValue *StoredValue = nullptr;
+      auto m_ConstantStrideVecPtr =
+          m_VecPtr(m_VPValue(), m_ConstantInt(Stride));
+      if ((!match(&R, m_WidenLoad(m_ConstantStrideVecPtr)) &&
+           !match(&R, m_WidenStore(m_ConstantStrideVecPtr,
+                                   m_VPValue(StoredValue)))) ||
+          Stride != 1)
+        continue;
+
+      auto *MemOp = cast<VPWidenMemoryRecipe>(&R);
+      if (!MemOp->isConsecutive())
+        continue;
+
+      // TODO: Support masked loads/stores. This requires widening the header
+      // mask to the same factor as the memory operation.
+      assert(!MemOp->isMasked() && "Masked accesses are not supported yet");
+
+      Type *AccessType = StoredValue ? StoredValue->getScalarType()
+                                     : R.getVPSingleValue()->getScalarType();
+      unsigned Opcode = isa<VPWidenLoadRecipe>(MemOp->getAsRecipe())
+                            ? Instruction::Load
+                            : Instruction::Store;
+      unsigned ScaleFactor = TTI.getPreferredVFMultipleForMemoryOp(
+          Opcode, AccessType, VF, UF, /*IsMasked=*/false);
+      assert((ScaleFactor != 0 && UF % ScaleFactor == 0) &&
+             "ScaleFactor must divide UF");
+      MemOp->setVFMultiple(ScaleFactor);
+    }
+  }
+}
+
 /// Returns true if \p V is VPWidenLoadRecipe or VPInterleaveRecipe that can be
 /// converted to a narrower recipe. \p V is used by a wide recipe that feeds a
 /// store interleave group at index \p Idx, \p WideMember0 is the recipe feeding
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 1c7b17942b795..c049ff16d85a1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -460,6 +460,11 @@ struct VPlanTransforms {
   static void sinkPredicatedStores(VPlan &Plan, PredicatedScalarEvolution &PSE,
                                    const Loop *L);
 
+  /// Widens memory operations by a factor of UF based on a target hook.
+  /// This allows targets to use wider memory operations when profitable.
+  static void scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF, unsigned UF,
+                                      const TargetTransformInfo &TTI);
+
   // Materialize vector trip counts for constants early if it can simply be
   // computed as (Original TC / VF * UF) * VF * UF.
   static void
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 52a16debc7acf..d4960fd35c6ec 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -74,6 +74,9 @@ class UnrollState {
     return Plan.getConstantInt(CanIVIntTy, Part);
   }
 
+  /// Unroll a VPWidenLoadRecipe or VPWidenStoreRecipe with a VFMultiple > 1.
+  void unrollMemOpWithVFMultiple(VPRecipeBase &R, unsigned VFMultiple);
+
 public:
   UnrollState(VPlan &Plan, unsigned UF) : Plan(Plan), UF(UF) {}
 
@@ -290,6 +293,62 @@ void UnrollState::unrollHeaderPHIByUF(VPHeaderPHIRecipe *R,
   }
 }
 
+void UnrollState::unrollMemOpWithVFMultiple(VPRecipeBase &R,
+                                            unsigned VFMultiple) {
+  assert(VFMultiple > 1 && UF % VFMultiple == 0);
+  SmallVector<VPRecipeBase *, 4> Groups(UF / VFMultiple, nullptr);
+  Groups[0] = &R;
+
+  // A memory op with a VFMultiple is widened to VF * VFMultiple elements, so
+  // after unrolling by UF we materialize UF / VFMultiple such ops, each
+  // covering VFMultiple unroll parts.
+  VPBuilder Builder = VPBuilder::getToInsertAfter(&R);
+  for (unsigned Group = 1; Group < Groups.size(); ++Group) {
+    auto *Copy = Builder.insert(R.clone());
+    remapOperands(Copy, Group * VFMultiple);
+    Groups[Group] = Copy;
+  }
+
+  if (auto *Store = dyn_cast<VPWidenStoreRecipe>(&R)) {
+    VPValue *StoredValue = Store->getStoredValue();
+    for (unsigned Group = 0; Group < Groups.size(); ++Group) {
+      VPRecipeBase *Store = Groups[Group];
+      Builder.setInsertPoint(Store);
+      SmallVector<VPValue *, 4> Parts;
+      // We need to concatenate VFMultiple parts to form the stored value.
+      for (unsigned Part = 0; Part < VFMultiple; ++Part)
+        Parts.push_back(
+            getValueForPart(StoredValue, Group * VFMultiple + Part));
+      auto *Concat =
+          Builder.createNaryOp(VPInstruction::ConcatVectorParts, Parts);
+      Groups[Group]->setOperand(1, Concat);
+      ToSkip.insert(Concat);
+    }
+  } else {
+    assert(isa<VPWidenLoadRecipe>(R) && "Expected a load recipe");
+    // We need to extract each unroll part as a subvector.
+    auto ExtractPart0 = Builder.createNaryOp(
+        VPInstruction::ExtractVectorForPart,
+        {Groups[0]->getVPSingleValue(), getConstantInt(0)});
+    // First replace R with an extract of the first unroll part (ExtractPart0).
+    R.getVPSingleValue()->replaceUsesWithIf(
+        ExtractPart0, [&](VPUser &U, unsigned) { return &U != ExtractPart0; });
+    ToSkip.insert(ExtractPart0);
+
+    // Create extracts for the remaining unroll parts and remap later uses of
+    // ExtractPart0 to the correct unrolled part.
+    for (unsigned Part = 1; Part != UF; ++Part) {
+      VPRecipeBase *Group = Groups[Part / VFMultiple];
+      unsigned IndexInGroup = Part % VFMultiple;
+      auto *Extract = Builder.createNaryOp(
+          VPInstruction::ExtractVectorForPart,
+          {Group->getVPSingleValue(), getConstantInt(IndexInGroup)});
+      addRecipeForPart(ExtractPart0, Extract, Part);
+      ToSkip.insert(Extract);
+    }
+  }
+}
+
 /// Handle non-header-phi recipes.
 void UnrollState::unrollRecipeByUF(VPRecipeBase &R) {
   if (match(&R, m_CombineOr(m_BranchOnCond(), m_BranchOnCount())))
@@ -301,6 +360,13 @@ void UnrollState::unrollRecipeByUF(VPRecipeBase &R) {
       return;
     }
   }
+
+  if (auto *WidenMem = dyn_cast<VPWidenMemoryRecipe>(&R);
+      WidenMem && WidenMem->getVFMultiple() > 1) {
+    unrollMemOpWithVFMultiple(R, WidenMem->getVFMultiple());
+    return;
+  }
+
   if (auto *RepR = dyn_cast<VPReplicateRecipe>(&R)) {
     if (isa<StoreInst>(RepR->getUnderlyingValue()) &&
         RepR->getOperand(1)->isDefinedOutsideLoopRegions()) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
new file mode 100644
index 0000000000000..f20e7aff70ef7
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -0,0 +1,374 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^middle.block:" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -sve-tail-folding=disabled %s | FileCheck %s --check-prefix=UNMASKED-SVE2P1
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @mixed_i64_i32_accesses(
+; UNMASKED-SVE2P1-LABEL: define void @mixed_i64_i32_accesses(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[X:%.*]], ptr noalias [[Y:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = shl nuw nsw i64 [[TMP3]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i64, ptr [[TMP5]], i64 [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 8 x i64>, ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = add <vscale x 4 x i64> [[TMP6]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <vscale x 4 x i64> [[TMP7]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <vscale x 4 x i64> [[TMP8]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add <vscale x 4 x i64> [[TMP9]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP17]], ptr [[TMP5]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP18]], <vscale x 4 x i64> [[TMP13]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP19]], ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 16 x i32>, ptr [[TMP20]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = add <vscale x 4 x i32> [[TMP21]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = add <vscale x 4 x i32> [[TMP22]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = add <vscale x 4 x i32> [[TMP23]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = add <vscale x 4 x i32> [[TMP24]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP25]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP30]], <vscale x 4 x i32> [[TMP27]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP31]], <vscale x 4 x i32> [[TMP28]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP32]], ptr [[TMP20]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP33]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+  ptr noalias %x, ptr noalias %y, i64 %n) {
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %px = getelementptr inbounds i64, ptr %x, i64 %iv
+  %vx = load i64, ptr %px, align 8
+  %ax = add i64 %vx, 1
+  store i64 %ax, ptr %px, align 8
+  %py = getelementptr inbounds i32, ptr %y, i64 %iv
+  %vy = load i32, ptr %py, align 4
+  %ay = add i32 %vy, 1
+  store i32 %ay, ptr %py, align 4
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+define void @mixed_i32_more_frequent_than_i64(
+; UNMASKED-SVE2P1-LABEL: define void @mixed_i32_more_frequent_than_i64(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[X:%.*]], ptr noalias [[Y:%.*]], ptr noalias [[Z:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = shl nuw nsw i64 [[TMP3]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i64, ptr [[TMP5]], i64 [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 8 x i64>, ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD3]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD3]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = add <vscale x 4 x i64> [[TMP6]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <vscale x 4 x i64> [[TMP7]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <vscale x 4 x i64> [[TMP8]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add <vscale x 4 x i64> [[TMP9]], splat (i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP17]], ptr [[TMP5]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP18]], <vscale x 4 x i64> [[TMP13]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP19]], ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 16 x i32>, ptr [[TMP20]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = add <vscale x 4 x i32> [[TMP21]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = add <vscale x 4 x i32> [[TMP22]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = add <vscale x 4 x i32> [[TMP23]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = add <vscale x 4 x i32> [[TMP24]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP25]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP30]], <vscale x 4 x i32> [[TMP27]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP31]], <vscale x 4 x i32> [[TMP28]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP32]], ptr [[TMP20]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = getelementptr inbounds i32, ptr [[Z]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 16 x i32>, ptr [[TMP33]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP34:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP35:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP36:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP37:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP38:%.*]] = add <vscale x 4 x i32> [[TMP34]], splat (i32 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP39:%.*]] = add <vscale x 4 x i32> [[TMP35]], splat (i32 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP40:%.*]] = add <vscale x 4 x i32> [[TMP36]], splat (i32 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP41:%.*]] = add <vscale x 4 x i32> [[TMP37]], splat (i32 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP42:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP38]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP43:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP42]], <vscale x 4 x i32> [[TMP39]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP44:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP43]], <vscale x 4 x i32> [[TMP40]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP45:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP44]], <vscale x 4 x i32> [[TMP41]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP45]], ptr [[TMP33]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP46:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP46]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+  ptr noalias %x, ptr noalias %y, ptr noalias %z, i64 %n) {
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %px = getelementptr inbounds i64, ptr %x, i64 %iv
+  %vx = load i64, ptr %px, align 8
+  %ax = add i64 %vx, 1
+  store i64 %ax, ptr %px, align 8
+  %py = getelementptr inbounds i32, ptr %y, i64 %iv
+  %vy = load i32, ptr %py, align 4
+  %ay = add i32 %vy, 1
+  store i32 %ay, ptr %py, align 4
+  %pz = getelementptr inbounds i32, ptr %z, i64 %iv
+  %vz = load i32, ptr %pz, align 4
+  %az = add i32 %vz, 2
+  store i32 %az, ptr %pz, align 4
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+define void @first_order_recurrence_i64_scaled_load_and_store(
+; UNMASKED-SVE2P1-LABEL: define void @first_order_recurrence_i64_scaled_load_and_store(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[SRC:%.*]], ptr noalias [[DST:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[FIRST:%.*]] = load i64, ptr [[SRC]], align 8
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[VECTOR_RECUR_INIT:%.*]] = insertelement <2 x i64> poison, i64 [[FIRST]], i32 1
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VECTOR_RECUR:%.*]] = phi <2 x i64> [ [[VECTOR_RECUR_INIT]], %[[VECTOR_PH]] ], [ [[WIDE_LOAD3:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = add nuw i64 1, [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i64, ptr [[SRC]], i64 [[TMP0]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i64, ptr [[TMP1]], i64 2
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[TMP1]], i64 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i64, ptr [[TMP1]], i64 6
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <2 x i64>, ptr [[TMP1]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD1:%.*]] = load <2 x i64>, ptr [[TMP2]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <2 x i64>, ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3]] = load <2 x i64>, ptr [[TMP4]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = shufflevector <2 x i64> [[VECTOR_RECUR]], <2 x i64> [[WIDE_LOAD]], <2 x i32> <i32 1, i32 2>
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = shufflevector <2 x i64> [[WIDE_LOAD]], <2 x i64> [[WIDE_LOAD1]], <2 x i32> <i32 1, i32 2>
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = shufflevector <2 x i64> [[WIDE_LOAD1]], <2 x i64> [[WIDE_LOAD2]], <2 x i32> <i32 1, i32 2>
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = shufflevector <2 x i64> [[WIDE_LOAD2]], <2 x i64> [[WIDE_LOAD3]], <2 x i32> <i32 1, i32 2>
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = add <2 x i64> [[TMP5]], [[WIDE_LOAD]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = add <2 x i64> [[TMP6]], [[WIDE_LOAD1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <2 x i64> [[TMP7]], [[WIDE_LOAD2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <2 x i64> [[TMP8]], [[WIDE_LOAD3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i64, ptr [[DST]], i64 [[TMP0]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i64, ptr [[TMP13]], i64 2
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i64, ptr [[TMP13]], i64 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i64, ptr [[TMP13]], i64 6
+; UNMASKED-SVE2P1-NEXT:    store <2 x i64> [[TMP9]], ptr [[TMP13]], align 8
+; UNMASKED-SVE2P1-NEXT:    store <2 x i64> [[TMP10]], ptr [[TMP14]], align 8
+; UNMASKED-SVE2P1-NEXT:    store <2 x i64> [[TMP11]], ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    store <2 x i64> [[TMP12]], ptr [[TMP16]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+  ptr noalias %src, ptr noalias %dst) {
+entry:
+  %first = load i64, ptr %src, align 8
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 1, %entry ], [ %next, %loop ]
+  %prev = phi i64 [ %first, %entry ], [ %cur, %loop ]
+  %cur.ptr = getelementptr inbounds i64, ptr %src, i64 %iv
+  %cur = load i64, ptr %cur.ptr, align 8
+  %sum = add i64 %prev, %cur
+  %dst.ptr = getelementptr inbounds i64, ptr %dst, i64 %iv
+  store i64 %sum, ptr %dst.ptr, align 8
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp eq i64 %next, 1025
+  br i1 %cmp, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+define i64 @i64_sum_reduction_scaled_partial_reduce(ptr noalias %a, i64 %n) {
+; UNMASKED-SVE2P1-LABEL: define i64 @i64_sum_reduction_scaled_partial_reduce(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 3
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP11:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP12:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP4]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9]] = add <vscale x 2 x i64> [[VEC_PHI]], [[TMP5]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_PHI1]], [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP11]] = add <vscale x 2 x i64> [[VEC_PHI2]], [[TMP7]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP12]] = add <vscale x 2 x i64> [[VEC_PHI3]], [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %acc = phi i64 [ 0, %entry ], [ %sum, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  %ld = load i64, ptr %p, align 8
+  %sum = add i64 %acc, %ld
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret i64 %sum
+}
+
+define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold) {
+; UNMASKED-SVE2P1-LABEL: define i64 @find_last_i64_scaled_load(
+; UNMASKED-SVE2P1-SAME: i64 [[N:%.*]], ptr noalias [[DATA:%.*]], i64 [[THRESHOLD:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 3
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[THRESHOLD]], i64 0
+; UNMASKED-SVE2P1-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP28:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP29:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP30:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP31:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP24:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP25:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP26:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP27:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i64, ptr [[DATA]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP7]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP9]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP10]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP11]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = freeze <vscale x 2 x i1> [[TMP12]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = freeze <vscale x 2 x i1> [[TMP13]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = or <vscale x 2 x i1> [[TMP16]], [[TMP17]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = freeze <vscale x 2 x i1> [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = or <vscale x 2 x i1> [[TMP18]], [[TMP19]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = freeze <vscale x 2 x i1> [[TMP15]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = or <vscale x 2 x i1> [[TMP20]], [[TMP21]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call i1 @llvm.vector.reduce.or.nxv2i1(<vscale x 2 x i1> [[TMP22]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP24]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP12]], <vscale x 2 x i1> [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP25]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP13]], <vscale x 2 x i1> [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP26]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP14]], <vscale x 2 x i1> [[TMP5]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP27]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP15]], <vscale x 2 x i1> [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP28]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP8]], <vscale x 2 x i64> [[VEC_PHI]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP29]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP9]], <vscale x 2 x i64> [[VEC_PHI1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP30]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP10]], <vscale x 2 x i64> [[VEC_PHI2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP31]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP11]], <vscale x 2 x i64> [[VEC_PHI3]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP32]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %data.phi = phi i64 [ -1, %entry ], [ %select.data, %loop ]
+  %p = getelementptr inbounds i64, ptr %data, i64 %iv
+  %ld = load i64, ptr %p, align 8
+  %select.cmp = icmp slt i64 %threshold, %ld
+  %select.data = select i1 %select.cmp, i64 %ld, i64 %data.phi
+  %next = add nuw i64 %iv, 1
+  %exit.cmp = icmp eq i64 %next, %n
+  br i1 %exit.cmp, label %exit, label %loop
+
+exit:
+  ret i64 %select.data
+}
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
new file mode 100644
index 0000000000000..3943cb5f28103
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
@@ -0,0 +1,151 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --filter-out-after "middle.block:" --version 6
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-before="scaleMemoryAccessesByUF$" 2>&1 \
+; RUN:   | FileCheck --check-prefix=BEFORE-SCALE %s
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-after="scaleMemoryAccessesByUF$" 2>&1 \
+; RUN:   | FileCheck --check-prefix=AFTER-SCALE %s
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-width="vscale x 4" -force-vector-interleave=4 -aarch64-enable-subreg-liveness-tracking -disable-output -S %s -vplan-print-after="unrollByUF$" 2>&1 \
+; RUN:   | FileCheck --check-prefix=AFTER-UNROLL %s
+
+define void @i64_load_store(ptr noalias %x, i64 %n) {
+; BEFORE-SCALE-LABEL: VPlan for loop in 'i64_load_store'
+; BEFORE-SCALE:  VPlan 'Initial VPlan for VF={vscale x 4},UF>=1' {
+; BEFORE-SCALE-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; BEFORE-SCALE-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; BEFORE-SCALE-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; BEFORE-SCALE-NEXT:  vp<[[VP3:%[0-9]+]]> = original trip-count
+; BEFORE-SCALE-EMPTY:
+; BEFORE-SCALE-NEXT:  ir-bb<entry>:
+; BEFORE-SCALE-NEXT:    EMIT vp<[[VP3]]> = EXPAND SCEV (1 umax %n)
+; BEFORE-SCALE-NEXT:    EMIT-SCALAR vp<[[VP4:%[0-9]+]]> = call i64 @llvm.vscale()
+; BEFORE-SCALE-NEXT:    EMIT vp<[[VP5:%[0-9]+]]> = mul nuw vp<[[VP4]]>, ir<16>
+; BEFORE-SCALE-NEXT:    EMIT vp<%min.iters.check> = icmp ult vp<[[VP3]]>, vp<[[VP5]]>
+; BEFORE-SCALE-NEXT:    EMIT branch-on-cond vp<%min.iters.check>
+; BEFORE-SCALE-NEXT:  Successor(s): scalar.ph, vector.ph
+; BEFORE-SCALE-EMPTY:
+; BEFORE-SCALE-NEXT:  vector.ph:
+; BEFORE-SCALE-NEXT:  Successor(s): vector loop
+; BEFORE-SCALE-EMPTY:
+; BEFORE-SCALE-NEXT:  <x1> vector loop: {
+; BEFORE-SCALE-NEXT:  vp<[[VP7:%[0-9]+]]> = CANONICAL-IV
+; BEFORE-SCALE-EMPTY:
+; BEFORE-SCALE-NEXT:    vector.body:
+; BEFORE-SCALE-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
+; BEFORE-SCALE-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
+; BEFORE-SCALE-NEXT:      vp<[[VP9:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; BEFORE-SCALE-NEXT:      WIDEN ir<%ld> = load vp<[[VP9]]>
+; BEFORE-SCALE-NEXT:      WIDEN ir<%add> = add ir<%ld>, ir<1>
+; BEFORE-SCALE-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; BEFORE-SCALE-NEXT:      WIDEN store vp<[[VP10]]>, ir<%add>
+; BEFORE-SCALE-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
+; BEFORE-SCALE-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; BEFORE-SCALE-NEXT:    No successors
+; BEFORE-SCALE-NEXT:  }
+; BEFORE-SCALE-NEXT:  Successor(s): middle.block
+; BEFORE-SCALE-EMPTY:
+; BEFORE-SCALE-NEXT:  middle.block:
+;
+; AFTER-SCALE-LABEL: VPlan for loop in 'i64_load_store'
+; AFTER-SCALE:  VPlan 'Initial VPlan for VF={vscale x 4},UF>=1' {
+; AFTER-SCALE-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; AFTER-SCALE-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; AFTER-SCALE-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; AFTER-SCALE-NEXT:  vp<[[VP3:%[0-9]+]]> = original trip-count
+; AFTER-SCALE-EMPTY:
+; AFTER-SCALE-NEXT:  ir-bb<entry>:
+; AFTER-SCALE-NEXT:    EMIT vp<[[VP3]]> = EXPAND SCEV (1 umax %n)
+; AFTER-SCALE-NEXT:    EMIT-SCALAR vp<[[VP4:%[0-9]+]]> = call i64 @llvm.vscale()
+; AFTER-SCALE-NEXT:    EMIT vp<[[VP5:%[0-9]+]]> = mul nuw vp<[[VP4]]>, ir<16>
+; AFTER-SCALE-NEXT:    EMIT vp<%min.iters.check> = icmp ult vp<[[VP3]]>, vp<[[VP5]]>
+; AFTER-SCALE-NEXT:    EMIT branch-on-cond vp<%min.iters.check>
+; AFTER-SCALE-NEXT:  Successor(s): scalar.ph, vector.ph
+; AFTER-SCALE-EMPTY:
+; AFTER-SCALE-NEXT:  vector.ph:
+; AFTER-SCALE-NEXT:  Successor(s): vector loop
+; AFTER-SCALE-EMPTY:
+; AFTER-SCALE-NEXT:  <x1> vector loop: {
+; AFTER-SCALE-NEXT:  vp<[[VP7:%[0-9]+]]> = CANONICAL-IV
+; AFTER-SCALE-EMPTY:
+; AFTER-SCALE-NEXT:    vector.body:
+; AFTER-SCALE-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
+; AFTER-SCALE-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
+; AFTER-SCALE-NEXT:      vp<[[VP9:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-SCALE-NEXT:      WIDEN ir<%ld> = load x2 vp<[[VP9]]>
+; AFTER-SCALE-NEXT:      WIDEN ir<%add> = add ir<%ld>, ir<1>
+; AFTER-SCALE-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-SCALE-NEXT:      WIDEN store x2 vp<[[VP10]]>, ir<%add>
+; AFTER-SCALE-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
+; AFTER-SCALE-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; AFTER-SCALE-NEXT:    No successors
+; AFTER-SCALE-NEXT:  }
+; AFTER-SCALE-NEXT:  Successor(s): middle.block
+; AFTER-SCALE-EMPTY:
+; AFTER-SCALE-NEXT:  middle.block:
+;
+; AFTER-UNROLL-LABEL: VPlan for loop in 'i64_load_store'
+; AFTER-UNROLL:  VPlan 'Initial VPlan for VF={vscale x 4},UF={4}' {
+; AFTER-UNROLL-NEXT:  Live-in vp<[[VP0:%[0-9]+]]> = VF
+; AFTER-UNROLL-NEXT:  Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; AFTER-UNROLL-NEXT:  Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; AFTER-UNROLL-NEXT:  vp<[[VP3:%[0-9]+]]> = original trip-count
+; AFTER-UNROLL-EMPTY:
+; AFTER-UNROLL-NEXT:  ir-bb<entry>:
+; AFTER-UNROLL-NEXT:    EMIT vp<[[VP3]]> = EXPAND SCEV (1 umax %n)
+; AFTER-UNROLL-NEXT:    EMIT-SCALAR vp<[[VP4:%[0-9]+]]> = call i64 @llvm.vscale()
+; AFTER-UNROLL-NEXT:    EMIT vp<[[VP5:%[0-9]+]]> = mul nuw vp<[[VP4]]>, ir<16>
+; AFTER-UNROLL-NEXT:    EMIT vp<%min.iters.check> = icmp ult vp<[[VP3]]>, vp<[[VP5]]>
+; AFTER-UNROLL-NEXT:    EMIT branch-on-cond vp<%min.iters.check>
+; AFTER-UNROLL-NEXT:  Successor(s): scalar.ph, vector.ph
+; AFTER-UNROLL-EMPTY:
+; AFTER-UNROLL-NEXT:  vector.ph:
+; AFTER-UNROLL-NEXT:  Successor(s): vector loop
+; AFTER-UNROLL-EMPTY:
+; AFTER-UNROLL-NEXT:  <x1> vector loop: {
+; AFTER-UNROLL-NEXT:  vp<[[VP7:%[0-9]+]]> = CANONICAL-IV
+; AFTER-UNROLL-EMPTY:
+; AFTER-UNROLL-NEXT:    vector.body:
+; AFTER-UNROLL-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
+; AFTER-UNROLL-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
+; AFTER-UNROLL-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-UNROLL-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP9]]>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%ld> = load x2 vp<[[VP10]]>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%ld>.1 = load x2 vp<[[VP11]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = extract-vector-for-part ir<%ld>, ir<0>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = extract-vector-for-part ir<%ld>, ir<1>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP14:%[0-9]+]]> = extract-vector-for-part ir<%ld>.1, ir<0>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP15:%[0-9]+]]> = extract-vector-for-part ir<%ld>.1, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add> = add vp<[[VP12]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.1 = add vp<[[VP13]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.2 = add vp<[[VP14]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.3 = add vp<[[VP15]]>, ir<1>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP16:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
+; AFTER-UNROLL-NEXT:      vp<[[VP17:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-UNROLL-NEXT:      vp<[[VP18:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP16]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP19:%[0-9]+]]> = concat-vector-parts ir<%add>, ir<%add>.1
+; AFTER-UNROLL-NEXT:      WIDEN store x2 vp<[[VP17]]>, vp<[[VP19]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP20:%[0-9]+]]> = concat-vector-parts ir<%add>.2, ir<%add>.3
+; AFTER-UNROLL-NEXT:      WIDEN store x2 vp<[[VP18]]>, vp<[[VP20]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
+; AFTER-UNROLL-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
+; AFTER-UNROLL-NEXT:    No successors
+; AFTER-UNROLL-NEXT:  }
+; AFTER-UNROLL-NEXT:  Successor(s): middle.block
+; AFTER-UNROLL-EMPTY:
+; AFTER-UNROLL-NEXT:  middle.block:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %ptr = getelementptr inbounds i64, ptr %x, i64 %iv
+  %ld = load i64, ptr %ptr, align 8
+  %add = add i64 %ld, 1
+  store i64 %add, ptr %ptr, align 8
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
index b5ee488ae5bd5..861721ea7c799 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
@@ -69,6 +69,7 @@
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] printOptimizedVPlan
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::addMinimumIterationCheck
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::replaceWideCanonicalIVWithWideIV
+; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::scaleMemoryAccessesByUF
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::unrollByUF
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::materializePacksAndUnpacks
 ; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::materializeBroadcasts

>From f787eadb6a2ebeb0aa4ede15e249f257f40c0d8c Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Thu, 20 Aug 2026 19:33:58 +0000
Subject: [PATCH 02/11] Rebase fixups

---
 .../AArch64/multi-vector-mem-ops.ll           | 20 ++++++++-----------
 1 file changed, 8 insertions(+), 12 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
index f20e7aff70ef7..416eea04db1cb 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -17,8 +17,7 @@ define void @mixed_i64_i32_accesses(
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
 ; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
-; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 2
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
@@ -57,7 +56,7 @@ define void @mixed_i64_i32_accesses(
 ; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP30]], <vscale x 4 x i32> [[TMP27]], i64 8)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP31]], <vscale x 4 x i32> [[TMP28]], i64 12)
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP32]], ptr [[TMP20]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP33]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
@@ -98,8 +97,7 @@ define void @mixed_i32_more_frequent_than_i64(
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
 ; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
-; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 2
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
@@ -153,7 +151,7 @@ define void @mixed_i32_more_frequent_than_i64(
 ; UNMASKED-SVE2P1-NEXT:    [[TMP44:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP43]], <vscale x 4 x i32> [[TMP40]], i64 8)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP45:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP44]], <vscale x 4 x i32> [[TMP41]], i64 12)
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP45]], ptr [[TMP33]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP46:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP46]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
@@ -257,8 +255,7 @@ define i64 @i64_sum_reduction_scaled_partial_reduce(ptr noalias %a, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 3
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
@@ -277,7 +274,7 @@ define i64 @i64_sum_reduction_scaled_partial_reduce(ptr noalias %a, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_PHI1]], [[TMP6]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP11]] = add <vscale x 2 x i64> [[VEC_PHI2]], [[TMP7]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12]] = add <vscale x 2 x i64> [[VEC_PHI3]], [[TMP8]]
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
@@ -308,8 +305,7 @@ define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold)
 ; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 3
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[THRESHOLD]], i64 0
 ; UNMASKED-SVE2P1-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
@@ -350,7 +346,7 @@ define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP29]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP9]], <vscale x 2 x i64> [[VEC_PHI1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP30]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP10]], <vscale x 2 x i64> [[VEC_PHI2]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP31]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP11]], <vscale x 2 x i64> [[VEC_PHI3]]
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP32]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:

>From 999d7989e15b210e868f2457de32516099622640 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Wed, 2 Sep 2026 10:46:03 +0000
Subject: [PATCH 03/11] Fixups

---
 llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp | 3 +--
 1 file changed, 1 insertion(+), 2 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index c7d27eaadd1c0..d4d8bda2dc759 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -5871,8 +5871,7 @@ AArch64TTIImpl::getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataTy,
   if (!ST->enableSubRegLiveness())
     return 1;
 
-  if ((Opcode != Instruction::Load && Opcode != Instruction::Store) ||
-      !ST->hasSVE2p1() || !VF.isScalable() || !isPowerOf2_32(UF))
+  if (!ST->hasSVE2p1() || !VF.isScalable() || !isPowerOf2_32(UF))
     return 1;
 
   unsigned VectorWidth = VF.getKnownMinValue() * DL.getTypeSizeInBits(DataTy);

>From dadde4a9abf5bf09865130867ad3a20746034ad6 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Wed, 2 Sep 2026 13:43:35 +0000
Subject: [PATCH 04/11] Check for extending loads

---
 .../llvm/Analysis/TargetTransformInfo.h       | 10 +--
 .../llvm/Analysis/TargetTransformInfoImpl.h   |  8 +--
 llvm/lib/Analysis/TargetTransformInfo.cpp     |  4 +-
 .../AArch64/AArch64TargetTransformInfo.cpp    | 14 ++--
 .../AArch64/AArch64TargetTransformInfo.h      |  7 +-
 .../Transforms/Vectorize/VPlanTransforms.cpp  | 10 ++-
 .../AArch64/multi-vector-mem-ops.ll           | 70 +++++++++++++++++++
 7 files changed, 104 insertions(+), 19 deletions(-)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index 4d0962cd420da..9e306152c2c80 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -981,12 +981,14 @@ class TargetTransformInfo {
 
   /// Return the preferred multiple of VF to use for a contiguous load/store.
   /// Returning 1 leaves the operation at VF. The returned value must divide UF.
+  /// \p CastHint is non-null if the stored or loaded valued is produced by or
+  /// consumed a cast instruction respectively.
   ///
   /// \p Opcode must be either Instruction::Load or Instruction::Store.
-  LLVM_ABI unsigned
-  getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataType,
-                                    ElementCount VF, unsigned UF,
-                                    bool IsMasked = false) const;
+  LLVM_ABI unsigned getPreferredVFMultipleForMemoryOp(
+      unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
+      bool IsMasked = false,
+      std::optional<Instruction::CastOps> CastHint = std::nullopt) const;
 
   /// Return true if we should be enabling ordered reductions for the target.
   LLVM_ABI bool enableOrderedReductions() const;
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 9e23ec8a42b18..408055f4a8dc5 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -422,11 +422,9 @@ class LLVM_ABI TargetTransformInfoImplBase {
     return false;
   }
 
-  virtual unsigned getPreferredVFMultipleForMemoryOp(unsigned Opcode,
-                                                     Type *DataType,
-                                                     ElementCount VF,
-                                                     unsigned UF,
-                                                     bool IsMasked) const {
+  virtual unsigned getPreferredVFMultipleForMemoryOp(
+      unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
+      bool IsMasked, std::optional<Instruction::CastOps> CastHint) const {
     return 1;
   }
 
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 8bdc293478fee..7b840b830d45a 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -549,9 +549,9 @@ bool TargetTransformInfo::isLegalStridedLoadStore(Type *DataType,
 
 unsigned TargetTransformInfo::getPreferredVFMultipleForMemoryOp(
     unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
-    bool IsMasked) const {
+    bool IsMasked, std::optional<Instruction::CastOps> CastHint) const {
   return TTIImpl->getPreferredVFMultipleForMemoryOp(Opcode, DataType, VF, UF,
-                                                    IsMasked);
+                                                    IsMasked, CastHint);
 }
 
 bool TargetTransformInfo::isLegalInterleavedAccessType(
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index d4d8bda2dc759..38d7edc8a1f5a 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -5859,15 +5859,21 @@ bool AArch64TTIImpl::isLegalMaskedExpandLoad(Type *DataTy,
          (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2());
 }
 
-unsigned
-AArch64TTIImpl::getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataTy,
-                                                  ElementCount VF, unsigned UF,
-                                                  bool IsMasked) const {
+unsigned AArch64TTIImpl::getPreferredVFMultipleForMemoryOp(
+    unsigned Opcode, Type *DataTy, ElementCount VF, unsigned UF, bool IsMasked,
+    std::optional<Instruction::CastOps> CastHint) const {
   assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
          "expected load/store opcode");
   if (IsMasked)
     return 1; // TODO: Support masked multi-vector loads/stores.
 
+  // Conservatively, avoid using multi-vector loads when it's possible we could
+  // use extending loads instead. Note: We can ignore stores as we only use
+  // truncating stores when the store vector-width is < a full SVE vector.
+  if (Opcode == Instruction::Load &&
+      (CastHint == Instruction::ZExt || CastHint == Instruction::SExt))
+    return 1;
+
   if (!ST->enableSubRegLiveness())
     return 1;
 
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index ea0f3039f1285..37c553dbbb0b4 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -279,9 +279,10 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
 
   bool isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const override;
 
-  unsigned getPreferredVFMultipleForMemoryOp(unsigned Opcode, Type *DataType,
-                                             ElementCount VF, unsigned UF,
-                                             bool IsMasked) const override;
+  unsigned getPreferredVFMultipleForMemoryOp(
+      unsigned Opcode, Type *DataType, ElementCount VF, unsigned UF,
+      bool IsMasked,
+      std::optional<Instruction::CastOps> CastHint) const override;
 
   void getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
                                TTI::UnrollingPreferences &UP,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index f89356f6f5815..e4988e0ab8d1e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4021,8 +4021,16 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
       unsigned Opcode = isa<VPWidenLoadRecipe>(MemOp->getAsRecipe())
                             ? Instruction::Load
                             : Instruction::Store;
+
+      std::optional<Instruction::CastOps> CastHint;
+      VPUser *MaybeCast = Opcode == Instruction::Store
+                              ? StoredValue->getDefiningRecipe()
+                              : R.getVPSingleValue()->getSingleUser();
+      if (auto *Cast = dyn_cast_if_present<VPWidenCastRecipe>(MaybeCast))
+        CastHint = Cast->getOpcode();
+
       unsigned ScaleFactor = TTI.getPreferredVFMultipleForMemoryOp(
-          Opcode, AccessType, VF, UF, /*IsMasked=*/false);
+          Opcode, AccessType, VF, UF, /*IsMasked=*/false, CastHint);
       assert((ScaleFactor != 0 && UF % ScaleFactor == 0) &&
              "ScaleFactor must divide UF");
       MemOp->setVFMultiple(ScaleFactor);
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
index 416eea04db1cb..27aa10c8d6023 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -368,3 +368,73 @@ loop:
 exit:
   ret i64 %select.data
 }
+
+; Negative test: We should not use a wide load when the result of the load is extended.
+; In this case, it's better to use SVE extending loads (rather than a multi-vector load).
+define void @extending_load(ptr noalias %dst, ptr %src, i64 %n) {
+; UNMASKED-SVE2P1-LABEL: define void @extending_load(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[DST:%.*]], ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ITER_CHECK:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw nsw i64 [[TMP2]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = mul nuw nsw i64 [[TMP2]], 3
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP3]], i64 [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP3]], i64 [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP3]], i64 [[TMP5]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP3]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 4 x i32>, ptr [[TMP6]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 4 x i32>, ptr [[TMP7]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD4:%.*]] = load <vscale x 4 x i32>, ptr [[TMP8]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = sext <vscale x 4 x i32> [[WIDE_LOAD]] to <vscale x 4 x i64>
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = sext <vscale x 4 x i32> [[WIDE_LOAD2]] to <vscale x 4 x i64>
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = sext <vscale x 4 x i32> [[WIDE_LOAD3]] to <vscale x 4 x i64>
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = sext <vscale x 4 x i32> [[WIDE_LOAD4]] to <vscale x 4 x i64>
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = mul nsw <vscale x 4 x i64> [[TMP9]], splat (i64 42)
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = mul nsw <vscale x 4 x i64> [[TMP10]], splat (i64 42)
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = mul nsw <vscale x 4 x i64> [[TMP11]], splat (i64 42)
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = mul nsw <vscale x 4 x i64> [[TMP12]], splat (i64 42)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = getelementptr inbounds nuw i64, ptr [[DST]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = getelementptr inbounds nuw i64, ptr [[TMP17]], i64 [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP13]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP19]], <vscale x 4 x i64> [[TMP14]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP20]], ptr [[TMP17]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP15]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP21]], <vscale x 4 x i64> [[TMP16]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP22]], ptr [[TMP18]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP23]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP14:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %src.ptr = getelementptr inbounds nuw i32, ptr %src, i64 %iv
+  %src.val = load i32, ptr %src.ptr, align 4
+  %conv = sext i32 %src.val to i64
+  %mul = mul nsw i64 %conv, 42
+  %dst.ptr = getelementptr inbounds nuw i64, ptr %dst, i64 %iv
+  store i64 %mul, ptr %dst.ptr, align 8
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exit.cmp = icmp eq i64 %iv.next, %n
+  br i1 %exit.cmp, label %exit, label %loop
+
+exit:
+  ret void
+}

>From 6f74924d7ef34f885871d5a382db0b12f46ec165 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Thu, 3 Sep 2026 16:04:44 +0000
Subject: [PATCH 05/11] Fixups

---
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp   |  2 +-
 llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h | 12 ------------
 llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp |  9 ++-------
 3 files changed, 3 insertions(+), 20 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 514bb051d9ae6..7f0aab9ba62a6 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5755,7 +5755,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
   // TODO: Move to VPlan transform stage once the transition to the VPlan-based
   // cost model is complete for better cost estimates.
   RUN_VPLAN_PASS(VPlanTransforms::scaleMemoryAccessesByUF, BestVPlan, BestVF,
-                 BestUF, CM.TTI);
+                 BestUF, TTI);
   RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
   RUN_VPLAN_PASS(VPlanTransforms::materializePacksAndUnpacks, BestVPlan);
   RUN_VPLAN_PASS(VPlanTransforms::materializeBroadcasts, BestVPlan);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
index 4023ccd6d76e0..1dc6a89c596eb 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
@@ -293,7 +293,6 @@ struct Recipe_match {
     if constexpr (std::is_same_v<RecipeTy, VPScalarIVStepsRecipe> ||
                   std::is_same_v<RecipeTy, VPDerivedIVRecipe> ||
                   std::is_same_v<RecipeTy, VPVectorEndPointerRecipe> ||
-                  std::is_same_v<RecipeTy, VPVectorPointerRecipe> ||
                   std::is_same_v<RecipeTy, VPWidenLoadRecipe> ||
                   std::is_same_v<RecipeTy, VPWidenStoreRecipe>)
       return DefR;
@@ -1028,17 +1027,6 @@ m_MaskedStore(const Addr_t &Addr, const Val_t &Val, const Mask_t &Mask) {
   return Store_match<Addr_t, Val_t, Mask_t>(Addr, Val, Mask);
 }
 
-template <typename Op0_t, typename Op1_t>
-using VectorPointerRecipe_match =
-    Recipe_match<std::tuple<Op0_t, Op1_t>, 0,
-                 /*Commutative*/ false, VPVectorPointerRecipe>;
-
-template <typename Op0_t, typename Op1_t>
-VectorPointerRecipe_match<Op0_t, Op1_t> m_VecPtr(const Op0_t &Op0,
-                                                 const Op1_t &Op1) {
-  return VectorPointerRecipe_match<Op0_t, Op1_t>(Op0, Op1);
-}
-
 template <typename Op0_t>
 using VPWidenLoadRecipe_match =
     Recipe_match<std::tuple<Op0_t>, 0,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index e4988e0ab8d1e..f16aac3bb94a3 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -3998,14 +3998,9 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
   for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
            vp_depth_first_deep(Plan.getVectorLoopRegion()->getEntry()))) {
     for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
-      uint64_t Stride;
       VPValue *StoredValue = nullptr;
-      auto m_ConstantStrideVecPtr =
-          m_VecPtr(m_VPValue(), m_ConstantInt(Stride));
-      if ((!match(&R, m_WidenLoad(m_ConstantStrideVecPtr)) &&
-           !match(&R, m_WidenStore(m_ConstantStrideVecPtr,
-                                   m_VPValue(StoredValue)))) ||
-          Stride != 1)
+      if (!match(&R, m_WidenLoad(m_VPValue())) &&
+          !match(&R, m_WidenStore(m_VPValue(), m_VPValue(StoredValue))))
         continue;
 
       auto *MemOp = cast<VPWidenMemoryRecipe>(&R);

>From 0c09c1be33eb1b99ae6152139957e35bab6d09cc Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 7 Sep 2026 11:20:54 +0000
Subject: [PATCH 06/11] Fixups

---
 .../Transforms/Vectorize/VPlanPatternMatch.h  |  12 +
 .../Transforms/Vectorize/VPlanTransforms.cpp  |  22 +-
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp |  12 +-
 ...-vector-mem-ops-non-power-of-two-unroll.ll |  74 +++
 .../AArch64/multi-vector-mem-ops.ll           | 445 ++++++++++++------
 5 files changed, 411 insertions(+), 154 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll

diff --git a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
index 1dc6a89c596eb..4023ccd6d76e0 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
@@ -293,6 +293,7 @@ struct Recipe_match {
     if constexpr (std::is_same_v<RecipeTy, VPScalarIVStepsRecipe> ||
                   std::is_same_v<RecipeTy, VPDerivedIVRecipe> ||
                   std::is_same_v<RecipeTy, VPVectorEndPointerRecipe> ||
+                  std::is_same_v<RecipeTy, VPVectorPointerRecipe> ||
                   std::is_same_v<RecipeTy, VPWidenLoadRecipe> ||
                   std::is_same_v<RecipeTy, VPWidenStoreRecipe>)
       return DefR;
@@ -1027,6 +1028,17 @@ m_MaskedStore(const Addr_t &Addr, const Val_t &Val, const Mask_t &Mask) {
   return Store_match<Addr_t, Val_t, Mask_t>(Addr, Val, Mask);
 }
 
+template <typename Op0_t, typename Op1_t>
+using VectorPointerRecipe_match =
+    Recipe_match<std::tuple<Op0_t, Op1_t>, 0,
+                 /*Commutative*/ false, VPVectorPointerRecipe>;
+
+template <typename Op0_t, typename Op1_t>
+VectorPointerRecipe_match<Op0_t, Op1_t> m_VecPtr(const Op0_t &Op0,
+                                                 const Op1_t &Op1) {
+  return VectorPointerRecipe_match<Op0_t, Op1_t>(Op0, Op1);
+}
+
 template <typename Op0_t>
 using VPWidenLoadRecipe_match =
     Recipe_match<std::tuple<Op0_t>, 0,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index f16aac3bb94a3..b10cc38e062d7 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -3996,16 +3996,20 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
     return;
 
   for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
-           vp_depth_first_deep(Plan.getVectorLoopRegion()->getEntry()))) {
+           vp_depth_first_shallow(Plan.getVectorLoopRegion()->getEntry()))) {
     for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
+      uint64_t Stride;
       VPValue *StoredValue = nullptr;
-      if (!match(&R, m_WidenLoad(m_VPValue())) &&
-          !match(&R, m_WidenStore(m_VPValue(), m_VPValue(StoredValue))))
+      auto m_ConstantStrideVecPtr =
+          m_VecPtr(m_VPValue(), m_ConstantInt(Stride));
+      if ((!match(&R, m_WidenLoad(m_ConstantStrideVecPtr)) &&
+           !match(&R, m_WidenStore(m_ConstantStrideVecPtr,
+                                   m_VPValue(StoredValue)))) ||
+          Stride != 1)
         continue;
 
       auto *MemOp = cast<VPWidenMemoryRecipe>(&R);
-      if (!MemOp->isConsecutive())
-        continue;
+      assert(MemOp->isConsecutive() && "Expected consecutive load/store");
 
       // TODO: Support masked loads/stores. This requires widening the header
       // mask to the same factor as the memory operation.
@@ -4024,11 +4028,11 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
       if (auto *Cast = dyn_cast_if_present<VPWidenCastRecipe>(MaybeCast))
         CastHint = Cast->getOpcode();
 
-      unsigned ScaleFactor = TTI.getPreferredVFMultipleForMemoryOp(
+      unsigned VFMultiple = TTI.getPreferredVFMultipleForMemoryOp(
           Opcode, AccessType, VF, UF, /*IsMasked=*/false, CastHint);
-      assert((ScaleFactor != 0 && UF % ScaleFactor == 0) &&
-             "ScaleFactor must divide UF");
-      MemOp->setVFMultiple(ScaleFactor);
+      assert((VFMultiple != 0 && UF % VFMultiple == 0) &&
+             "VFMultiple must divide UF");
+      MemOp->setVFMultiple(VFMultiple);
     }
   }
 }
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index d4960fd35c6ec..3e270b2725c9f 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -295,7 +295,8 @@ void UnrollState::unrollHeaderPHIByUF(VPHeaderPHIRecipe *R,
 
 void UnrollState::unrollMemOpWithVFMultiple(VPRecipeBase &R,
                                             unsigned VFMultiple) {
-  assert(VFMultiple > 1 && UF % VFMultiple == 0);
+  assert(VFMultiple > 1 && UF % VFMultiple == 0 &&
+         "expected VFMultiple to divide UF");
   SmallVector<VPRecipeBase *, 4> Groups(UF / VFMultiple, nullptr);
   Groups[0] = &R;
 
@@ -361,10 +362,11 @@ void UnrollState::unrollRecipeByUF(VPRecipeBase &R) {
     }
   }
 
-  if (auto *WidenMem = dyn_cast<VPWidenMemoryRecipe>(&R);
-      WidenMem && WidenMem->getVFMultiple() > 1) {
-    unrollMemOpWithVFMultiple(R, WidenMem->getVFMultiple());
-    return;
+  if (auto *WidenMem = dyn_cast<VPWidenMemoryRecipe>(&R)) {
+    if (WidenMem && WidenMem->getVFMultiple() > 1) {
+      unrollMemOpWithVFMultiple(R, WidenMem->getVFMultiple());
+      return;
+    }
   }
 
   if (auto *RepR = dyn_cast<VPReplicateRecipe>(&R)) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll
new file mode 100644
index 0000000000000..ac9e55117364e
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops-non-power-of-two-unroll.ll
@@ -0,0 +1,74 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^middle.block:" --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve2p1 -force-vector-interleave=3 -aarch64-enable-subreg-liveness-tracking -sve-tail-folding=disabled %s | FileCheck %s --check-prefix=CHECK
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; Tests `scaleMemoryAccessesByUF` with a non-power-of-two unroll factor.
+; On AArch64, this should be rejected (which results in `vscale x 4` loads/stores).
+
+define void @mixed_i64_i32_accesses(ptr noalias %x, ptr noalias %y, i64 %n) {
+; CHECK-LABEL: define void @mixed_i64_i32_accesses(
+; CHECK-SAME: ptr noalias [[X:%.*]], ptr noalias [[Y:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP2:%.*]] = mul nuw i64 [[TMP1]], 12
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP5:%.*]] = shl nuw nsw i64 [[TMP3]], 1
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i64, ptr [[TMP4]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i64, ptr [[TMP4]], i64 [[TMP5]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 4 x i64>, ptr [[TMP4]], align 8
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 4 x i64>, ptr [[TMP6]], align 8
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 4 x i64>, ptr [[TMP7]], align 8
+; CHECK-NEXT:    [[TMP8:%.*]] = add <vscale x 4 x i64> [[WIDE_LOAD]], splat (i64 1)
+; CHECK-NEXT:    [[TMP9:%.*]] = add <vscale x 4 x i64> [[WIDE_LOAD1]], splat (i64 1)
+; CHECK-NEXT:    [[TMP10:%.*]] = add <vscale x 4 x i64> [[WIDE_LOAD2]], splat (i64 1)
+; CHECK-NEXT:    store <vscale x 4 x i64> [[TMP8]], ptr [[TMP4]], align 8
+; CHECK-NEXT:    store <vscale x 4 x i64> [[TMP9]], ptr [[TMP6]], align 8
+; CHECK-NEXT:    store <vscale x 4 x i64> [[TMP10]], ptr [[TMP7]], align 8
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[TMP11]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[TMP11]], i64 [[TMP5]]
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 4 x i32>, ptr [[TMP11]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD4:%.*]] = load <vscale x 4 x i32>, ptr [[TMP12]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD5:%.*]] = load <vscale x 4 x i32>, ptr [[TMP13]], align 4
+; CHECK-NEXT:    [[TMP14:%.*]] = add <vscale x 4 x i32> [[WIDE_LOAD3]], splat (i32 1)
+; CHECK-NEXT:    [[TMP15:%.*]] = add <vscale x 4 x i32> [[WIDE_LOAD4]], splat (i32 1)
+; CHECK-NEXT:    [[TMP16:%.*]] = add <vscale x 4 x i32> [[WIDE_LOAD5]], splat (i32 1)
+; CHECK-NEXT:    store <vscale x 4 x i32> [[TMP14]], ptr [[TMP11]], align 4
+; CHECK-NEXT:    store <vscale x 4 x i32> [[TMP15]], ptr [[TMP12]], align 4
+; CHECK-NEXT:    store <vscale x 4 x i32> [[TMP16]], ptr [[TMP13]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %px = getelementptr inbounds i64, ptr %x, i64 %iv
+  %vx = load i64, ptr %px, align 8
+  %ax = add i64 %vx, 1
+  store i64 %ax, ptr %px, align 8
+  %py = getelementptr inbounds i32, ptr %y, i64 %iv
+  %vy = load i32, ptr %py, align 4
+  %ay = add i32 %vy, 1
+  store i32 %ay, ptr %py, align 4
+  %next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
index 27aa10c8d6023..b90e29519e24e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/multi-vector-mem-ops.ll
@@ -3,30 +3,30 @@
 
 target triple = "aarch64-unknown-linux-gnu"
 
-define void @mixed_i64_i32_accesses(
+define void @mixed_i64_i32_accesses(ptr noalias %x, ptr noalias %y, i64 %n) {
 ; UNMASKED-SVE2P1-LABEL: define void @mixed_i64_i32_accesses(
 ; UNMASKED-SVE2P1-SAME: ptr noalias [[X:%.*]], ptr noalias [[Y:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
-; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
-; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 4
-; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1-NEXT:  [[ITER_CHECK:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[UMAX]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
-; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 4
-; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[UMAX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[UMAX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[UMAX]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = shl nuw nsw i64 [[TMP3]], 1
-; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i64, ptr [[TMP5]], i64 [[TMP14]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 8 x i64>, ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw nsw i64 [[TMP2]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[TMP3]], i64 [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
 ; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 0)
@@ -35,33 +35,32 @@ define void @mixed_i64_i32_accesses(
 ; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <vscale x 4 x i64> [[TMP7]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <vscale x 4 x i64> [[TMP8]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add <vscale x 4 x i64> [[TMP9]], splat (i64 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP14]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP15]], ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP13]], i64 4)
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP17]], ptr [[TMP5]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP18]], <vscale x 4 x i64> [[TMP13]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP19]], ptr [[TMP15]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 16 x i32>, ptr [[TMP20]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = add <vscale x 4 x i32> [[TMP19]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = add <vscale x 4 x i32> [[TMP20]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = add <vscale x 4 x i32> [[TMP21]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = add <vscale x 4 x i32> [[TMP22]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = add <vscale x 4 x i32> [[TMP23]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = add <vscale x 4 x i32> [[TMP24]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP25]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP30]], <vscale x 4 x i32> [[TMP27]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP31]], <vscale x 4 x i32> [[TMP28]], i64 12)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP32]], ptr [[TMP20]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP33]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP23]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP27]], <vscale x 4 x i32> [[TMP24]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP28]], <vscale x 4 x i32> [[TMP25]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP30]], ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
-  ptr noalias %x, ptr noalias %y, i64 %n) {
 entry:
   br label %loop
 
@@ -83,80 +82,79 @@ exit:
   ret void
 }
 
-define void @mixed_i32_more_frequent_than_i64(
+define void @mixed_i32_more_frequent_than_i64(ptr noalias %x, ptr noalias %y, ptr noalias %z, i64 %n) {
 ; UNMASKED-SVE2P1-LABEL: define void @mixed_i32_more_frequent_than_i64(
 ; UNMASKED-SVE2P1-SAME: ptr noalias [[X:%.*]], ptr noalias [[Y:%.*]], ptr noalias [[Z:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
-; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 4
-; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1-NEXT:  [[ITER_CHECK:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[UMAX]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
-; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 4
-; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[UMAX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
-; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 2
-; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[UMAX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[UMAX]], [[N_MOD_VF]]
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = shl nuw nsw i64 [[TMP3]], 1
-; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i64, ptr [[TMP5]], i64 [[TMP14]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 8 x i64>, ptr [[TMP15]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[X]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw nsw i64 [[TMP2]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i64, ptr [[TMP3]], i64 [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 8 x i64>, ptr [[TMP5]], align 8
 ; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD3]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD3]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 4 x i64> @llvm.vector.extract.nxv4i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD2]], i64 4)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = add <vscale x 4 x i64> [[TMP6]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add <vscale x 4 x i64> [[TMP7]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = add <vscale x 4 x i64> [[TMP8]], splat (i64 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add <vscale x 4 x i64> [[TMP9]], splat (i64 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP10]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP14]], <vscale x 4 x i64> [[TMP11]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP15]], ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP16]], <vscale x 4 x i64> [[TMP13]], i64 4)
 ; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP17]], ptr [[TMP5]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> poison, <vscale x 4 x i64> [[TMP12]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 8 x i64> @llvm.vector.insert.nxv8i64.nxv4i64(<vscale x 8 x i64> [[TMP18]], <vscale x 4 x i64> [[TMP13]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 8 x i64> [[TMP19]], ptr [[TMP15]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 16 x i32>, ptr [[TMP20]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD1]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[Y]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = add <vscale x 4 x i32> [[TMP19]], splat (i32 1)
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = add <vscale x 4 x i32> [[TMP20]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = add <vscale x 4 x i32> [[TMP21]], splat (i32 1)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = add <vscale x 4 x i32> [[TMP22]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = add <vscale x 4 x i32> [[TMP23]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = add <vscale x 4 x i32> [[TMP24]], splat (i32 1)
-; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP25]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP30]], <vscale x 4 x i32> [[TMP27]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP31]], <vscale x 4 x i32> [[TMP28]], i64 12)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP32]], ptr [[TMP20]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = getelementptr inbounds i32, ptr [[Z]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 16 x i32>, ptr [[TMP33]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[TMP34:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP35:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP36:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP37:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD2]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP23]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP27]], <vscale x 4 x i32> [[TMP24]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP28]], <vscale x 4 x i32> [[TMP25]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP29]], <vscale x 4 x i32> [[TMP26]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP30]], ptr [[TMP18]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = getelementptr inbounds i32, ptr [[Z]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD4:%.*]] = load <vscale x 16 x i32>, ptr [[TMP31]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP34:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP35:%.*]] = call <vscale x 4 x i32> @llvm.vector.extract.nxv4i32.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD4]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    [[TMP36:%.*]] = add <vscale x 4 x i32> [[TMP32]], splat (i32 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP37:%.*]] = add <vscale x 4 x i32> [[TMP33]], splat (i32 2)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP38:%.*]] = add <vscale x 4 x i32> [[TMP34]], splat (i32 2)
 ; UNMASKED-SVE2P1-NEXT:    [[TMP39:%.*]] = add <vscale x 4 x i32> [[TMP35]], splat (i32 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP40:%.*]] = add <vscale x 4 x i32> [[TMP36]], splat (i32 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP41:%.*]] = add <vscale x 4 x i32> [[TMP37]], splat (i32 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP42:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP38]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP43:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP42]], <vscale x 4 x i32> [[TMP39]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP44:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP43]], <vscale x 4 x i32> [[TMP40]], i64 8)
-; UNMASKED-SVE2P1-NEXT:    [[TMP45:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP44]], <vscale x 4 x i32> [[TMP41]], i64 12)
-; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP45]], ptr [[TMP33]], align 4
-; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP46:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP46]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP40:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> poison, <vscale x 4 x i32> [[TMP36]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP41:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP40]], <vscale x 4 x i32> [[TMP37]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP42:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP41]], <vscale x 4 x i32> [[TMP38]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP43:%.*]] = call <vscale x 16 x i32> @llvm.vector.insert.nxv16i32.nxv4i32(<vscale x 16 x i32> [[TMP42]], <vscale x 4 x i32> [[TMP39]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x i32> [[TMP43]], ptr [[TMP31]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP44:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP44]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
-  ptr noalias %x, ptr noalias %y, ptr noalias %z, i64 %n) {
 entry:
   br label %loop
 
@@ -182,7 +180,7 @@ exit:
   ret void
 }
 
-define void @first_order_recurrence_i64_scaled_load_and_store(
+define void @first_order_recurrence_i64_scaled_load_and_store(ptr noalias %src, ptr noalias %dst) {
 ; UNMASKED-SVE2P1-LABEL: define void @first_order_recurrence_i64_scaled_load_and_store(
 ; UNMASKED-SVE2P1-SAME: ptr noalias [[SRC:%.*]], ptr noalias [[DST:%.*]]) #[[ATTR0]] {
 ; UNMASKED-SVE2P1-NEXT:  [[ENTRY:.*:]]
@@ -224,7 +222,6 @@ define void @first_order_recurrence_i64_scaled_load_and_store(
 ; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
-  ptr noalias %src, ptr noalias %dst) {
 entry:
   %first = load i64, ptr %src, align 8
   br label %loop
@@ -260,23 +257,23 @@ define i64 @i64_sum_reduction_scaled_partial_reduce(ptr noalias %a, i64 %n) {
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP11:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP12:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP4]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
-; UNMASKED-SVE2P1-NEXT:    [[TMP9]] = add <vscale x 2 x i64> [[VEC_PHI]], [[TMP5]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_PHI1]], [[TMP6]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP11]] = add <vscale x 2 x i64> [[VEC_PHI2]], [[TMP7]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP12]] = add <vscale x 2 x i64> [[VEC_PHI3]], [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP11:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP3]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8]] = add <vscale x 2 x i64> [[VEC_PHI]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP9]] = add <vscale x 2 x i64> [[VEC_PHI1]], [[TMP5]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP10]] = add <vscale x 2 x i64> [[VEC_PHI2]], [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP11]] = add <vscale x 2 x i64> [[VEC_PHI3]], [[TMP7]]
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -312,43 +309,43 @@ define i64 @find_last_i64_scaled_load(i64 %n, ptr noalias %data, i64 %threshold)
 ; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP28:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP29:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP30:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP31:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP27:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP28:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP29:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[VEC_PHI3:%.*]] = phi <vscale x 2 x i64> [ splat (i64 -1), %[[VECTOR_PH]] ], [ [[TMP30:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP23:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP24:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP25:%.*]], %[[VECTOR_BODY]] ]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP26:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = phi <vscale x 2 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP27:%.*]], %[[VECTOR_BODY]] ]
-; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i64, ptr [[DATA]], i64 [[INDEX]]
-; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP7]], align 8
-; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
-; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
-; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
-; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i64, ptr [[DATA]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 8 x i64>, ptr [[TMP6]], align 8
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 2)
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = call <vscale x 2 x i64> @llvm.vector.extract.nxv2i64.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD]], i64 6)
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP7]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP8]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP9]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP10]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = icmp slt <vscale x 2 x i64> [[BROADCAST_SPLAT]], [[TMP11]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = freeze <vscale x 2 x i1> [[TMP11]]
 ; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = freeze <vscale x 2 x i1> [[TMP12]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = freeze <vscale x 2 x i1> [[TMP13]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = or <vscale x 2 x i1> [[TMP16]], [[TMP17]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = freeze <vscale x 2 x i1> [[TMP14]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = or <vscale x 2 x i1> [[TMP18]], [[TMP19]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = freeze <vscale x 2 x i1> [[TMP15]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = or <vscale x 2 x i1> [[TMP20]], [[TMP21]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = call i1 @llvm.vector.reduce.or.nxv2i1(<vscale x 2 x i1> [[TMP22]])
-; UNMASKED-SVE2P1-NEXT:    [[TMP24]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP12]], <vscale x 2 x i1> [[TMP3]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP25]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP13]], <vscale x 2 x i1> [[TMP4]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP26]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP14]], <vscale x 2 x i1> [[TMP5]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP27]] = select i1 [[TMP23]], <vscale x 2 x i1> [[TMP15]], <vscale x 2 x i1> [[TMP6]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP28]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP8]], <vscale x 2 x i64> [[VEC_PHI]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP29]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP9]], <vscale x 2 x i64> [[VEC_PHI1]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP30]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP10]], <vscale x 2 x i64> [[VEC_PHI2]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP31]] = select i1 [[TMP23]], <vscale x 2 x i64> [[TMP11]], <vscale x 2 x i64> [[VEC_PHI3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = or <vscale x 2 x i1> [[TMP15]], [[TMP16]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = freeze <vscale x 2 x i1> [[TMP13]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = or <vscale x 2 x i1> [[TMP17]], [[TMP18]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = freeze <vscale x 2 x i1> [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = or <vscale x 2 x i1> [[TMP19]], [[TMP20]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = call i1 @llvm.vector.reduce.or.nxv2i1(<vscale x 2 x i1> [[TMP21]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP23]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP11]], <vscale x 2 x i1> [[TMP2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP24]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP12]], <vscale x 2 x i1> [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP25]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP13]], <vscale x 2 x i1> [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP26]] = select i1 [[TMP22]], <vscale x 2 x i1> [[TMP14]], <vscale x 2 x i1> [[TMP5]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP27]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP7]], <vscale x 2 x i64> [[VEC_PHI]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP28]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP8]], <vscale x 2 x i64> [[VEC_PHI1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP29]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP9]], <vscale x 2 x i64> [[VEC_PHI2]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP30]] = select i1 [[TMP22]], <vscale x 2 x i64> [[TMP10]], <vscale x 2 x i64> [[VEC_PHI3]]
 ; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
-; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP32]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
 ; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
 ;
 entry:
@@ -438,3 +435,171 @@ loop:
 exit:
   ret void
 }
+
+; TODO: Support multi-vector operations with negative strides.
+define void @reverse_stride(ptr noalias %c, ptr noalias  %a, ptr noalias %b, i64 %n) {
+; UNMASKED-SVE2P1-LABEL: define void @reverse_stride(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[C:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ITER_CHECK:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; UNMASKED-SVE2P1-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[N]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = sub i64 [[TMP0]], [[SMIN]]
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP1]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP2]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP1]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP2]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP1]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP1]], [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = sub i64 [[N]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = sub i64 [[N]], [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i32, ptr [[A]], i64 [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = sub nuw nsw i64 [[TMP4]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = sub i64 0, [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 [[TMP9]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = sub i64 [[TMP9]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 [[TMP27]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = mul i64 -2, [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = sub i64 [[TMP29]], [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP34:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 [[TMP33]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP35:%.*]] = mul i64 -3, [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP36:%.*]] = sub i64 [[TMP35]], [[TMP8]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP37:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 [[TMP36]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = load <vscale x 4 x i32>, ptr [[TMP10]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = load <vscale x 4 x i32>, ptr [[TMP28]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = load <vscale x 4 x i32>, ptr [[TMP34]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = load <vscale x 4 x i32>, ptr [[TMP37]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = getelementptr inbounds nuw i32, ptr [[B]], i64 [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[TMP15]], i64 [[TMP9]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP38:%.*]] = getelementptr inbounds i32, ptr [[TMP15]], i64 [[TMP27]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP39:%.*]] = getelementptr inbounds i32, ptr [[TMP15]], i64 [[TMP33]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP40:%.*]] = getelementptr inbounds i32, ptr [[TMP15]], i64 [[TMP36]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = load <vscale x 4 x i32>, ptr [[TMP16]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = load <vscale x 4 x i32>, ptr [[TMP38]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = load <vscale x 4 x i32>, ptr [[TMP39]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = load <vscale x 4 x i32>, ptr [[TMP40]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = add nsw <vscale x 4 x i32> [[TMP17]], [[TMP11]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = add nsw <vscale x 4 x i32> [[TMP18]], [[TMP12]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = add nsw <vscale x 4 x i32> [[TMP19]], [[TMP13]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = add nsw <vscale x 4 x i32> [[TMP20]], [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = getelementptr inbounds nuw i32, ptr [[C]], i64 [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = getelementptr inbounds i32, ptr [[TMP25]], i64 [[TMP9]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = getelementptr inbounds i32, ptr [[TMP25]], i64 [[TMP27]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP41:%.*]] = getelementptr inbounds i32, ptr [[TMP25]], i64 [[TMP33]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = getelementptr inbounds i32, ptr [[TMP25]], i64 [[TMP36]]
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 4 x i32> [[TMP21]], ptr [[TMP26]], align 4
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 4 x i32> [[TMP22]], ptr [[TMP30]], align 4
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 4 x i32> [[TMP23]], ptr [[TMP41]], align 4
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 4 x i32> [[TMP24]], ptr [[TMP32]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP17:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ %dec, %for.body ], [ %n, %entry ]
+  %a.ptr = getelementptr inbounds nuw i32, ptr %a, i64 %iv
+  %a.val = load i32, ptr %a.ptr, align 4
+  %b.ptr = getelementptr inbounds nuw i32, ptr %b, i64 %iv
+  %b.val = load i32, ptr %b.ptr, align 4
+  %add = add nsw i32 %b.val, %a.val
+  %c.ptr = getelementptr inbounds nuw i32, ptr %c, i64 %iv
+  store i32 %add, ptr %c.ptr, align 4
+  %dec = add nsw i64 %iv, -1
+  %cmp = icmp sgt i64 %iv, 0
+  br i1 %cmp, label %for.body, label %exit
+
+exit:
+  ret void
+}
+
+; Negative test (load): A strided load (stride = 2). Only the store should be widened to a VF multiple.
+define void @gather_nxv4i32_stride2(ptr noalias %a, ptr noalias %b, i64 %n) {
+; UNMASKED-SVE2P1-LABEL: define void @gather_nxv4i32_stride2(
+; UNMASKED-SVE2P1-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; UNMASKED-SVE2P1-NEXT:  [[ITER_CHECK:.*:]]
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ule i64 [[N]], 4
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[VEC_EPILOG_SCALAR_PH:label %.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; UNMASKED-SVE2P1-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 4
+; UNMASKED-SVE2P1-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ule i64 [[N]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[MIN_ITERS_CHECK1]], [[VEC_EPILOG_PH:label %.*]], label %[[VECTOR_PH:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_PH]]:
+; UNMASKED-SVE2P1-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
+; UNMASKED-SVE2P1-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[N_MOD_VF]], 0
+; UNMASKED-SVE2P1-NEXT:    [[TMP4:%.*]] = select i1 [[TMP3]], i64 [[TMP1]], i64 [[N_MOD_VF]]
+; UNMASKED-SVE2P1-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP4]]
+; UNMASKED-SVE2P1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; UNMASKED-SVE2P1:       [[VECTOR_BODY]]:
+; UNMASKED-SVE2P1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNMASKED-SVE2P1-NEXT:    [[TMP5:%.*]] = add i64 [[TMP2]], 0
+; UNMASKED-SVE2P1-NEXT:    [[TMP6:%.*]] = mul i64 [[TMP5]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP7:%.*]] = add i64 [[INDEX]], [[TMP6]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP8:%.*]] = shl i64 [[TMP2]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP9:%.*]] = add i64 [[TMP8]], 0
+; UNMASKED-SVE2P1-NEXT:    [[TMP10:%.*]] = mul i64 [[TMP9]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP11:%.*]] = add i64 [[INDEX]], [[TMP10]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP12:%.*]] = mul i64 [[TMP2]], 3
+; UNMASKED-SVE2P1-NEXT:    [[TMP13:%.*]] = add i64 [[TMP12]], 0
+; UNMASKED-SVE2P1-NEXT:    [[TMP14:%.*]] = mul i64 [[TMP13]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP15:%.*]] = add i64 [[INDEX]], [[TMP14]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP16:%.*]] = shl i64 [[INDEX]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP17:%.*]] = shl i64 [[TMP7]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP18:%.*]] = shl i64 [[TMP11]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP19:%.*]] = shl i64 [[TMP15]], 1
+; UNMASKED-SVE2P1-NEXT:    [[TMP20:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[TMP16]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP21:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[TMP17]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP22:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[TMP18]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP23:%.*]] = getelementptr inbounds float, ptr [[B]], i64 [[TMP19]]
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_VEC:%.*]] = load <vscale x 8 x float>, ptr [[TMP20]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[STRIDED_VEC:%.*]] = call { <vscale x 4 x float>, <vscale x 4 x float> } @llvm.vector.deinterleave2.nxv8f32(<vscale x 8 x float> [[WIDE_VEC]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP24:%.*]] = extractvalue { <vscale x 4 x float>, <vscale x 4 x float> } [[STRIDED_VEC]], 0
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_VEC2:%.*]] = load <vscale x 8 x float>, ptr [[TMP21]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[STRIDED_VEC3:%.*]] = call { <vscale x 4 x float>, <vscale x 4 x float> } @llvm.vector.deinterleave2.nxv8f32(<vscale x 8 x float> [[WIDE_VEC2]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP25:%.*]] = extractvalue { <vscale x 4 x float>, <vscale x 4 x float> } [[STRIDED_VEC3]], 0
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_VEC4:%.*]] = load <vscale x 8 x float>, ptr [[TMP22]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[STRIDED_VEC5:%.*]] = call { <vscale x 4 x float>, <vscale x 4 x float> } @llvm.vector.deinterleave2.nxv8f32(<vscale x 8 x float> [[WIDE_VEC4]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP26:%.*]] = extractvalue { <vscale x 4 x float>, <vscale x 4 x float> } [[STRIDED_VEC5]], 0
+; UNMASKED-SVE2P1-NEXT:    [[WIDE_VEC6:%.*]] = load <vscale x 8 x float>, ptr [[TMP23]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[STRIDED_VEC7:%.*]] = call { <vscale x 4 x float>, <vscale x 4 x float> } @llvm.vector.deinterleave2.nxv8f32(<vscale x 8 x float> [[WIDE_VEC6]])
+; UNMASKED-SVE2P1-NEXT:    [[TMP27:%.*]] = extractvalue { <vscale x 4 x float>, <vscale x 4 x float> } [[STRIDED_VEC7]], 0
+; UNMASKED-SVE2P1-NEXT:    [[TMP28:%.*]] = getelementptr inbounds float, ptr [[A]], i64 [[INDEX]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP29:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> poison, <vscale x 4 x float> [[TMP24]], i64 0)
+; UNMASKED-SVE2P1-NEXT:    [[TMP30:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP29]], <vscale x 4 x float> [[TMP25]], i64 4)
+; UNMASKED-SVE2P1-NEXT:    [[TMP31:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP30]], <vscale x 4 x float> [[TMP26]], i64 8)
+; UNMASKED-SVE2P1-NEXT:    [[TMP32:%.*]] = call <vscale x 16 x float> @llvm.vector.insert.nxv16f32.nxv4f32(<vscale x 16 x float> [[TMP31]], <vscale x 4 x float> [[TMP27]], i64 12)
+; UNMASKED-SVE2P1-NEXT:    store <vscale x 16 x float> [[TMP32]], ptr [[TMP28]], align 4
+; UNMASKED-SVE2P1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; UNMASKED-SVE2P1-NEXT:    [[TMP33:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; UNMASKED-SVE2P1-NEXT:    br i1 [[TMP33]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
+; UNMASKED-SVE2P1:       [[MIDDLE_BLOCK]]:
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ %iv.next, %for.body ], [ 0, %entry ]
+  %iv.stride2 = mul i64 %iv, 2
+  %b.ptr = getelementptr inbounds float, ptr %b, i64 %iv.stride2
+  %b.val = load float, ptr %b.ptr, align 4
+  %a.ptr = getelementptr inbounds float, ptr %a, i64 %iv
+  store float %b.val, ptr %a.ptr, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond.not = icmp eq i64 %iv.next, %n
+  br i1 %exitcond.not, label %exit, label %for.body
+
+exit:
+  ret void
+}

>From 6f67c83d311c740e700de9af69181453a89e541a Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 7 Sep 2026 13:14:38 +0000
Subject: [PATCH 07/11] Rework opcodes

---
 llvm/lib/Transforms/Vectorize/VPlan.h         | 38 +++++------
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp | 55 ++++++++++------
 .../Transforms/Vectorize/VPlanTransforms.cpp  | 27 +++++++-
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp | 65 ++++++++++---------
 .../vplan-printing-multi-vector-mem-ops.ll    | 40 ++++++------
 5 files changed, 131 insertions(+), 94 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index ce31583b4c3a5..129870f69c6d7 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1323,10 +1323,18 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
     // WideActiveLaneMask is used for control flow and is unrolled by widening,
     // with one extract vector created per unroll part.
     WideActiveLaneMask,
+    // Signature: (VFMultiple, Address, Alignment) -> Wide Vector
+    // Loads a single wide vector of `VFMultiple * VF` elements. VFMultiple must
+    // divide UF. After unrolling, each section of VF elements in the wide
+    // vector corresponds to an unroll part.
+    VFMultipleLoad,
+    // Signature: (VFMultiple, Address, Alignment, Vectors...)
+    // Concatenates VFMultiple vector operands into a single wide vector of
+    // `VFMultiple * VF` elements and stores it. After unrolling, each vector
+    // operand corresponds to an unroll part.
+    VFMultipleStore,
     // Extracts each unrolled part of a (VF * UF) widened vector/mask.
     ExtractVectorForPart,
-    // Concatenates its unrolled part operands into one widened vector.
-    ConcatVectorParts,
     ExplicitVectorLength,
     // Represents the incoming loop-invariant alias-mask. All memory accesses
     // in the loop must stay within the active lanes.
@@ -1530,6 +1538,7 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
     case VPInstruction::BranchOnCond:
     case VPInstruction::BranchOnTwoConds:
     case VPInstruction::BranchOnCount:
+    case VPInstruction::VFMultipleStore:
       return false;
     default:
       return true;
@@ -3763,10 +3772,6 @@ class LLVM_ABI_FOR_TEST VPWidenMemoryRecipe : public VPIRMetadata {
   /// Whether the memory access is masked.
   bool IsMasked = false;
 
-  /// Multiple of VF used to widen this memory operation. The final operation
-  /// loads or stores VF * VFMultiple elements
-  unsigned VFMultiple = 1;
-
   void setMask(VPValue *Mask) {
     assert(!IsMasked && "cannot re-set mask");
     if (!Mask)
@@ -3813,12 +3818,6 @@ class LLVM_ABI_FOR_TEST VPWidenMemoryRecipe : public VPIRMetadata {
   InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const;
 
   Instruction &getIngredient() const { return Ingredient; }
-
-  /// Set the VF multiple for this memory operation.
-  void setVFMultiple(unsigned VFMultiple) { this->VFMultiple = VFMultiple; }
-
-  /// Returns the VF multiple of this memory operation.
-  unsigned getVFMultiple() const { return VFMultiple; }
 };
 
 /// A recipe for widening load operations, using the address to load from and an
@@ -3834,11 +3833,8 @@ struct LLVM_ABI_FOR_TEST VPWidenLoadRecipe final : public VPSingleDefRecipe,
   }
 
   VPWidenLoadRecipe *clone() override {
-    auto *R =
-        new VPWidenLoadRecipe(cast<LoadInst>(Ingredient), getAddr(), getMask(),
-                              Consecutive, *this, getDebugLoc());
-    R->setVFMultiple(VFMultiple);
-    return R;
+    return new VPWidenLoadRecipe(cast<LoadInst>(Ingredient), getAddr(),
+                                 getMask(), Consecutive, *this, getDebugLoc());
   }
 
   VP_CLASSOF_IMPL(VPRecipeBase::VPWidenLoadSC);
@@ -3942,11 +3938,9 @@ struct LLVM_ABI_FOR_TEST VPWidenStoreRecipe final : public VPRecipeBase,
   }
 
   VPWidenStoreRecipe *clone() override {
-    auto *R = new VPWidenStoreRecipe(cast<StoreInst>(Ingredient), getAddr(),
-                                     getStoredValue(), getMask(), Consecutive,
-                                     *this, getDebugLoc());
-    R->setVFMultiple(VFMultiple);
-    return R;
+    return new VPWidenStoreRecipe(cast<StoreInst>(Ingredient), getAddr(),
+                                  getStoredValue(), getMask(), Consecutive,
+                                  *this, getDebugLoc());
   }
 
   VP_CLASSOF_IMPL(VPRecipeBase::VPWidenStoreSC);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index b8155e9a2bdf2..c569d3604739d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -497,6 +497,7 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
     for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
       AssertOperandType(Idx, Op0Ty);
     return Type::getVoidTy(Ctx);
+  case VPInstruction::VFMultipleStore:
   case Instruction::Store:
     return Type::getVoidTy(Ctx);
   case Instruction::ICmp:
@@ -570,9 +571,9 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
     return StructTy->getTypeAtIndex(
         cast<VPConstantInt>(Operands[1])->getZExtValue());
   }
-  case VPInstruction::ConcatVectorParts:
   case VPInstruction::ExtractVectorForPart:
     return Op0Ty;
+  case VPInstruction::VFMultipleLoad:
   case VPInstruction::FirstActiveLane:
   case VPInstruction::LastActiveLane:
   case VPInstruction::NumActiveLanes:
@@ -681,6 +682,7 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
   case Instruction::Select:
   case VPInstruction::WideActiveLaneMask:
   case VPInstruction::ReductionStartVector:
+  case VPInstruction::VFMultipleLoad:
     return 3;
   case Instruction::Call:
     return getCalledFnOperandIndex(operands()) + 1;
@@ -700,7 +702,7 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
   case VPInstruction::LastActiveLane:
   case VPInstruction::ExtractLane:
   case VPInstruction::ExtractLastActive:
-  case VPInstruction::ConcatVectorParts:
+  case VPInstruction::VFMultipleStore:
     // Cannot determine the number of operands from the opcode.
     return -1u;
   }
@@ -1149,18 +1151,32 @@ Value *VPInstruction::generate(VPTransformState &State) {
                                          vputils::getIntrinsicID(this), Args,
                                          /*FMFSource=*/nullptr, getName());
   }
-  case VPInstruction::ConcatVectorParts: {
-    unsigned VectorOps = getNumOperands();
+  case VPInstruction::VFMultipleLoad: {
+    unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
     auto *WideDataTy = VectorType::get(
-        getScalarType(), State.VF.multiplyCoefficientBy(VectorOps));
-    Value *WideData = PoisonValue::get(WideDataTy);
+        getScalarType(), State.VF.multiplyCoefficientBy(VFMultiple));
+
+    Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
+    Align Alignment = Align(cast<VPConstantInt>(getOperand(2))->getZExtValue());
+    return Builder.CreateAlignedLoad(WideDataTy, Addr, Alignment,
+                                     "vf.multiple.load");
+  }
+  case VPInstruction::VFMultipleStore: {
+    unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
+    Type *ScalarStoreTy = getOperand(3)->getScalarType();
+    auto *WideDataTy = VectorType::get(
+        ScalarStoreTy, State.VF.multiplyCoefficientBy(VFMultiple));
 
-    for (unsigned I = 0; I < VectorOps; ++I) {
-      Value *Part = State.get(getOperand(I));
+    Value *WideData = PoisonValue::get(WideDataTy);
+    for (unsigned I = 0; I < VFMultiple; ++I) {
+      Value *Part = State.get(getOperand(I + 3));
       WideData = Builder.CreateInsertVector(WideDataTy, WideData, Part,
                                             I * State.VF.getKnownMinValue());
     }
-    return WideData;
+
+    Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
+    Align Alignment = Align(cast<VPConstantInt>(getOperand(2))->getZExtValue());
+    return Builder.CreateAlignedStore(WideData, Addr, Alignment);
   }
   default:
     llvm_unreachable("Unsupported opcode for instruction");
@@ -1639,6 +1655,10 @@ void VPInstruction::addOperand(VPValue *Op) {
            "matching operand 1's type and i1, respectively");
     break;
   }
+  case VPInstruction::VFMultipleStore:
+    assert(Ty == getOperand(3)->getScalarType() &&
+           "appended operand must match operand 3's scalar type");
+    break;
   default:
     llvm_unreachable("opcode does not support growing the operand list "
                      "outside of construction");
@@ -1710,7 +1730,6 @@ bool VPInstruction::opcodeMayReadOrWriteFromMemory() const {
   case VPInstruction::ActiveLaneMask:
   case VPInstruction::WideActiveLaneMask:
   case VPInstruction::IncomingAliasMask:
-  case VPInstruction::ConcatVectorParts:
   case VPInstruction::ExitingIVValue:
   case VPInstruction::ExplicitVectorLength:
   case VPInstruction::FirstActiveLane:
@@ -1842,12 +1861,15 @@ void VPInstruction::printRecipe(raw_ostream &O, const Twine &Indent,
   case VPInstruction::WideActiveLaneMask:
     O << "wide active lane mask";
     break;
+  case VPInstruction::VFMultipleLoad:
+    O << "vf-multiple load";
+    break;
+  case VPInstruction::VFMultipleStore:
+    O << "vf-multiple store";
+    break;
   case VPInstruction::IncomingAliasMask:
     O << "incoming-alias-mask";
     break;
-  case VPInstruction::ConcatVectorParts:
-    O << "concat-vector-parts";
-    break;
   case VPInstruction::ExplicitVectorLength:
     O << "EXPLICIT-VECTOR-LENGTH";
     break;
@@ -4362,8 +4384,7 @@ InstructionCost VPWidenMemoryRecipe::computeCost(ElementCount VF,
 
 void VPWidenLoadRecipe::execute(VPTransformState &State) {
   Type *ScalarDataTy = getScalarType();
-  auto *DataTy =
-      VectorType::get(ScalarDataTy, State.VF.multiplyCoefficientBy(VFMultiple));
+  auto *DataTy = VectorType::get(ScalarDataTy, State.VF);
   bool CreateGather = !isConsecutive();
 
   auto &Builder = State.Builder;
@@ -4393,8 +4414,6 @@ void VPWidenLoadRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
   O << Indent << "WIDEN ";
   printAsOperand(O, SlotTracker);
   O << " = load ";
-  if (VFMultiple > 1)
-    O << "x" << VFMultiple << ' ';
   printOperands(O, SlotTracker);
 }
 #endif
@@ -4482,8 +4501,6 @@ void VPWidenStoreRecipe::execute(VPTransformState &State) {
 void VPWidenStoreRecipe::printRecipe(raw_ostream &O, const Twine &Indent,
                                      VPSlotTracker &SlotTracker) const {
   O << Indent << "WIDEN store ";
-  if (VFMultiple > 1)
-    O << "x" << VFMultiple << ' ';
   printOperands(O, SlotTracker);
 }
 #endif
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index b10cc38e062d7..ccb15fcab6a73 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -3995,6 +3995,8 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
   if (UF == 1)
     return;
 
+  Type *IVTy = Plan.getVectorLoopRegion()->getCanonicalIVType();
+
   for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
            vp_depth_first_shallow(Plan.getVectorLoopRegion()->getEntry()))) {
     for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
@@ -4032,7 +4034,30 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
           Opcode, AccessType, VF, UF, /*IsMasked=*/false, CastHint);
       assert((VFMultiple != 0 && UF % VFMultiple == 0) &&
              "VFMultiple must divide UF");
-      MemOp->setVFMultiple(VFMultiple);
+
+      if (VFMultiple == 1)
+        continue;
+
+      VPBuilder Builder(VPBB, R.getIterator());
+
+      VPValue *Ptr = MemOp->getAddr();
+      VPValue *VFMultipleVPV = Plan.getConstantInt(IVTy, VFMultiple);
+      VPValue *Align = Plan.getConstantInt(IVTy, MemOp->getAlign().value());
+
+      if (Opcode == Instruction::Load) {
+        VPValue *OldLoad = R.getVPSingleValue();
+        VPValue *Load = Builder.createNaryOp(
+            VPInstruction::VFMultipleLoad, {VFMultipleVPV, Ptr, Align}, nullptr,
+            {}, {}, DebugLoc::getUnknown(), "", OldLoad->getScalarType());
+        OldLoad->replaceAllUsesWith(Load);
+      } else {
+        assert(Opcode == Instruction::Store);
+        Builder.createNaryOp(VPInstruction::VFMultipleStore,
+                             {VFMultipleVPV, Ptr, Align, StoredValue},
+                             R.getDebugLoc());
+      }
+
+      R.eraseFromParent();
     }
   }
 }
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 3e270b2725c9f..5c950a0fd8028 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -74,8 +74,8 @@ class UnrollState {
     return Plan.getConstantInt(CanIVIntTy, Part);
   }
 
-  /// Unroll a VPWidenLoadRecipe or VPWidenStoreRecipe with a VFMultiple > 1.
-  void unrollMemOpWithVFMultiple(VPRecipeBase &R, unsigned VFMultiple);
+  /// Unroll a VFMultipleLoad or VFMultipleStore VPInstruction.
+  void unrollMemOpWithVFMultiple(VPInstruction *VPI);
 
 public:
   UnrollState(VPlan &Plan, unsigned UF) : Plan(Plan), UF(UF) {}
@@ -293,53 +293,57 @@ void UnrollState::unrollHeaderPHIByUF(VPHeaderPHIRecipe *R,
   }
 }
 
-void UnrollState::unrollMemOpWithVFMultiple(VPRecipeBase &R,
-                                            unsigned VFMultiple) {
+void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
+  assert(VPI->getOpcode() == VPInstruction::VFMultipleLoad ||
+         VPI->getOpcode() == VPInstruction::VFMultipleStore);
+
+  unsigned VFMultiple = cast<VPConstantInt>(VPI->getOperand(0))->getZExtValue();
   assert(VFMultiple > 1 && UF % VFMultiple == 0 &&
          "expected VFMultiple to divide UF");
-  SmallVector<VPRecipeBase *, 4> Groups(UF / VFMultiple, nullptr);
-  Groups[0] = &R;
+
+  SmallVector<VPInstruction *, 4> Groups(UF / VFMultiple, nullptr);
+  Groups[0] = VPI;
 
   // A memory op with a VFMultiple is widened to VF * VFMultiple elements, so
   // after unrolling by UF we materialize UF / VFMultiple such ops, each
   // covering VFMultiple unroll parts.
-  VPBuilder Builder = VPBuilder::getToInsertAfter(&R);
+  VPBuilder Builder = VPBuilder::getToInsertAfter(VPI);
   for (unsigned Group = 1; Group < Groups.size(); ++Group) {
-    auto *Copy = Builder.insert(R.clone());
+    auto *Copy = Builder.insert(VPI->clone());
     remapOperands(Copy, Group * VFMultiple);
     Groups[Group] = Copy;
   }
 
-  if (auto *Store = dyn_cast<VPWidenStoreRecipe>(&R)) {
-    VPValue *StoredValue = Store->getStoredValue();
+  if (VPI->getOpcode() == VPInstruction::VFMultipleStore) {
+    VPValue *StoredValue = VPI->getOperand(3);
     for (unsigned Group = 0; Group < Groups.size(); ++Group) {
-      VPRecipeBase *Store = Groups[Group];
-      Builder.setInsertPoint(Store);
-      SmallVector<VPValue *, 4> Parts;
-      // We need to concatenate VFMultiple parts to form the stored value.
-      for (unsigned Part = 0; Part < VFMultiple; ++Part)
-        Parts.push_back(
-            getValueForPart(StoredValue, Group * VFMultiple + Part));
-      auto *Concat =
-          Builder.createNaryOp(VPInstruction::ConcatVectorParts, Parts);
-      Groups[Group]->setOperand(1, Concat);
-      ToSkip.insert(Concat);
+      VPInstruction *Store = Groups[Group];
+      // Add the value to store for each unroll part in this group.
+      for (unsigned Part = 0; Part < VFMultiple; ++Part) {
+        VPValue *UnrollPart =
+            getValueForPart(StoredValue, Group * VFMultiple + Part);
+        if (Part == 0)
+          Store->setOperand(3, UnrollPart);
+        else
+          Store->addOperand(UnrollPart);
+      }
     }
   } else {
-    assert(isa<VPWidenLoadRecipe>(R) && "Expected a load recipe");
+    assert(VPI->getOpcode() == VPInstruction::VFMultipleLoad &&
+           "Expected a load recipe");
     // We need to extract each unroll part as a subvector.
     auto ExtractPart0 = Builder.createNaryOp(
         VPInstruction::ExtractVectorForPart,
         {Groups[0]->getVPSingleValue(), getConstantInt(0)});
-    // First replace R with an extract of the first unroll part (ExtractPart0).
-    R.getVPSingleValue()->replaceUsesWithIf(
+    // First VPI with an extract of the first unroll part (ExtractPart0).
+    VPI->getVPSingleValue()->replaceUsesWithIf(
         ExtractPart0, [&](VPUser &U, unsigned) { return &U != ExtractPart0; });
     ToSkip.insert(ExtractPart0);
 
     // Create extracts for the remaining unroll parts and remap later uses of
     // ExtractPart0 to the correct unrolled part.
     for (unsigned Part = 1; Part != UF; ++Part) {
-      VPRecipeBase *Group = Groups[Part / VFMultiple];
+      VPInstruction *Group = Groups[Part / VFMultiple];
       unsigned IndexInGroup = Part % VFMultiple;
       auto *Extract = Builder.createNaryOp(
           VPInstruction::ExtractVectorForPart,
@@ -356,15 +360,14 @@ void UnrollState::unrollRecipeByUF(VPRecipeBase &R) {
     return;
 
   if (auto *VPI = dyn_cast<VPInstruction>(&R)) {
-    if (vputils::onlyFirstPartUsed(VPI)) {
-      addUniformForAllParts(VPI);
+    if (VPI->getOpcode() == VPInstruction::VFMultipleLoad ||
+        VPI->getOpcode() == VPInstruction::VFMultipleStore) {
+      unrollMemOpWithVFMultiple(VPI);
       return;
     }
-  }
 
-  if (auto *WidenMem = dyn_cast<VPWidenMemoryRecipe>(&R)) {
-    if (WidenMem && WidenMem->getVFMultiple() > 1) {
-      unrollMemOpWithVFMultiple(R, WidenMem->getVFMultiple());
+    if (vputils::onlyFirstPartUsed(VPI)) {
+      addUniformForAllParts(VPI);
       return;
     }
   }
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
index 3943cb5f28103..15dc08babf853 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/AArch64/vplan-printing-multi-vector-mem-ops.ll
@@ -69,10 +69,10 @@ define void @i64_load_store(ptr noalias %x, i64 %n) {
 ; AFTER-SCALE-NEXT:      vp<[[VP8:%[0-9]+]]> = SCALAR-STEPS vp<[[VP7]]>, ir<1>, vp<[[VP0]]>
 ; AFTER-SCALE-NEXT:      CLONE ir<%ptr> = getelementptr inbounds ir<%x>, vp<[[VP8]]>
 ; AFTER-SCALE-NEXT:      vp<[[VP9:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-SCALE-NEXT:      WIDEN ir<%ld> = load x2 vp<[[VP9]]>
-; AFTER-SCALE-NEXT:      WIDEN ir<%add> = add ir<%ld>, ir<1>
-; AFTER-SCALE-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-SCALE-NEXT:      WIDEN store x2 vp<[[VP10]]>, ir<%add>
+; AFTER-SCALE-NEXT:      EMIT vp<[[VP10:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP9]]>, ir<8>
+; AFTER-SCALE-NEXT:      WIDEN ir<%add> = add vp<[[VP10]]>, ir<1>
+; AFTER-SCALE-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-SCALE-NEXT:      EMIT vf-multiple store ir<2>, vp<[[VP11]]>, ir<8>, ir<%add>
 ; AFTER-SCALE-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
 ; AFTER-SCALE-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; AFTER-SCALE-NEXT:    No successors
@@ -108,23 +108,21 @@ define void @i64_load_store(ptr noalias %x, i64 %n) {
 ; AFTER-UNROLL-NEXT:      EMIT vp<[[VP9:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
 ; AFTER-UNROLL-NEXT:      vp<[[VP10:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
 ; AFTER-UNROLL-NEXT:      vp<[[VP11:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP9]]>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%ld> = load x2 vp<[[VP10]]>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%ld>.1 = load x2 vp<[[VP11]]>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = extract-vector-for-part ir<%ld>, ir<0>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = extract-vector-for-part ir<%ld>, ir<1>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP14:%[0-9]+]]> = extract-vector-for-part ir<%ld>.1, ir<0>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP15:%[0-9]+]]> = extract-vector-for-part ir<%ld>.1, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add> = add vp<[[VP12]]>, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.1 = add vp<[[VP13]]>, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.2 = add vp<[[VP14]]>, ir<1>
-; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.3 = add vp<[[VP15]]>, ir<1>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP16:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
-; AFTER-UNROLL-NEXT:      vp<[[VP17:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
-; AFTER-UNROLL-NEXT:      vp<[[VP18:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP16]]>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP19:%[0-9]+]]> = concat-vector-parts ir<%add>, ir<%add>.1
-; AFTER-UNROLL-NEXT:      WIDEN store x2 vp<[[VP17]]>, vp<[[VP19]]>
-; AFTER-UNROLL-NEXT:      EMIT vp<[[VP20:%[0-9]+]]> = concat-vector-parts ir<%add>.2, ir<%add>.3
-; AFTER-UNROLL-NEXT:      WIDEN store x2 vp<[[VP18]]>, vp<[[VP20]]>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP12:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP10]]>, ir<8>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP13:%[0-9]+]]> = vf-multiple load ir<2>, vp<[[VP11]]>, ir<8>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP14:%[0-9]+]]> = extract-vector-for-part vp<[[VP12]]>, ir<0>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP15:%[0-9]+]]> = extract-vector-for-part vp<[[VP12]]>, ir<1>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP16:%[0-9]+]]> = extract-vector-for-part vp<[[VP13]]>, ir<0>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP17:%[0-9]+]]> = extract-vector-for-part vp<[[VP13]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add> = add vp<[[VP14]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.1 = add vp<[[VP15]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.2 = add vp<[[VP16]]>, ir<1>
+; AFTER-UNROLL-NEXT:      WIDEN ir<%add>.3 = add vp<[[VP17]]>, ir<1>
+; AFTER-UNROLL-NEXT:      EMIT vp<[[VP18:%[0-9]+]]> = mul nuw nsw vp<[[VP0]]>, ir<2>
+; AFTER-UNROLL-NEXT:      vp<[[VP19:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>
+; AFTER-UNROLL-NEXT:      vp<[[VP20:%[0-9]+]]> = vector-pointer inbounds i64, ir<%ptr>, ir<1>, vp<[[VP18]]>
+; AFTER-UNROLL-NEXT:      EMIT vf-multiple store ir<2>, vp<[[VP19]]>, ir<8>, ir<%add>, ir<%add>.1
+; AFTER-UNROLL-NEXT:      EMIT vf-multiple store ir<2>, vp<[[VP20]]>, ir<8>, ir<%add>.2, ir<%add>.3
 ; AFTER-UNROLL-NEXT:      EMIT vp<%index.next> = add nuw vp<[[VP7]]>, vp<[[VP1]]>
 ; AFTER-UNROLL-NEXT:      EMIT branch-on-count vp<%index.next>, vp<[[VP2]]>
 ; AFTER-UNROLL-NEXT:    No successors

>From f75c8c0a09d5f86ed3fe651c8890f270f3aab585 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 7 Sep 2026 14:09:53 +0000
Subject: [PATCH 08/11] Fixups

---
 llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp    | 12 +++++++++---
 llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp |  3 +--
 2 files changed, 10 insertions(+), 5 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index c569d3604739d..8523680aa2187 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -67,7 +67,8 @@ bool VPRecipeBase::mayWriteToMemory() const {
   case VPInstructionSC: {
     auto *VPI = cast<VPInstruction>(this);
     // Loads read from memory but don't write to memory.
-    if (VPI->getOpcode() == Instruction::Load)
+    if (VPI->getOpcode() == Instruction::Load ||
+        VPI->getOpcode() == VPInstruction::VFMultipleLoad)
       return false;
     return VPI->opcodeMayReadOrWriteFromMemory();
   }
@@ -126,8 +127,13 @@ bool VPRecipeBase::mayReadFromMemory() const {
   switch (getVPRecipeID()) {
   case VPExpressionSC:
     return cast<VPExpressionRecipe>(this)->mayReadOrWriteMemory();
-  case VPInstructionSC:
-    return cast<VPInstruction>(this)->opcodeMayReadOrWriteFromMemory();
+  case VPInstructionSC: {
+    auto *VPI = cast<VPInstruction>(this);
+    // Stores write to memory but don't read from memory.
+    if (VPI->getOpcode() == VPInstruction::VFMultipleStore)
+      return false;
+    return VPI->opcodeMayReadOrWriteFromMemory();
+  }
   case VPWidenLoadEVLSC:
   case VPWidenLoadSC:
     return true;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index ccb15fcab6a73..a0b637f04849a 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4038,12 +4038,11 @@ void VPlanTransforms::scaleMemoryAccessesByUF(VPlan &Plan, ElementCount VF,
       if (VFMultiple == 1)
         continue;
 
-      VPBuilder Builder(VPBB, R.getIterator());
-
       VPValue *Ptr = MemOp->getAddr();
       VPValue *VFMultipleVPV = Plan.getConstantInt(IVTy, VFMultiple);
       VPValue *Align = Plan.getConstantInt(IVTy, MemOp->getAlign().value());
 
+      VPBuilder Builder(VPBB, R.getIterator());
       if (Opcode == Instruction::Load) {
         VPValue *OldLoad = R.getVPSingleValue();
         VPValue *Load = Builder.createNaryOp(

>From f1426f53d433bf0542753d52a93a6880fa58c279 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 7 Sep 2026 16:33:45 +0000
Subject: [PATCH 09/11] Fixups

---
 llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp | 3 +++
 1 file changed, 3 insertions(+)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 8523680aa2187..c95700e7f7117 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1814,6 +1814,9 @@ bool VPInstruction::usesFirstLaneOnly(const VPValue *Op) const {
   case VPInstruction::WidePtrAdd:
     // WidePtrAdd supports scalar and vector base addresses.
     return false;
+  case VPInstruction::VFMultipleLoad:
+  case VPInstruction::VFMultipleStore:
+    return Op == getOperand(0) || Op == getOperand(1) || Op == getOperand(2);
   case VPInstruction::ExitingIVValue:
   case VPInstruction::ExtractLane:
     return Op == getOperand(0);

>From 698bc185398a4adc52ec6bb0783c29c7e199f532 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Fri, 11 Sep 2026 09:59:38 +0000
Subject: [PATCH 10/11] Rebase fixups

---
 llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp | 6 ++----
 1 file changed, 2 insertions(+), 4 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index c95700e7f7117..342ecd3ff6afe 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1159,8 +1159,7 @@ Value *VPInstruction::generate(VPTransformState &State) {
   }
   case VPInstruction::VFMultipleLoad: {
     unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
-    auto *WideDataTy = VectorType::get(
-        getScalarType(), State.VF.multiplyCoefficientBy(VFMultiple));
+    auto *WideDataTy = VectorType::get(getScalarType(), State.VF * VFMultiple);
 
     Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
     Align Alignment = Align(cast<VPConstantInt>(getOperand(2))->getZExtValue());
@@ -1170,8 +1169,7 @@ Value *VPInstruction::generate(VPTransformState &State) {
   case VPInstruction::VFMultipleStore: {
     unsigned VFMultiple = cast<VPConstantInt>(getOperand(0))->getZExtValue();
     Type *ScalarStoreTy = getOperand(3)->getScalarType();
-    auto *WideDataTy = VectorType::get(
-        ScalarStoreTy, State.VF.multiplyCoefficientBy(VFMultiple));
+    auto *WideDataTy = VectorType::get(ScalarStoreTy, State.VF * VFMultiple);
 
     Value *WideData = PoisonValue::get(WideDataTy);
     for (unsigned I = 0; I < VFMultiple; ++I) {

>From 4d8515f6ae393404bcf75448c3583e087340d045 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Fri, 11 Sep 2026 10:24:13 +0000
Subject: [PATCH 11/11] Fixups

---
 llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp | 45 ++++++++++---------
 1 file changed, 23 insertions(+), 22 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
index 5c950a0fd8028..c127f287f18ee 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUnroll.cpp
@@ -328,29 +328,30 @@ void UnrollState::unrollMemOpWithVFMultiple(VPInstruction *VPI) {
           Store->addOperand(UnrollPart);
       }
     }
-  } else {
-    assert(VPI->getOpcode() == VPInstruction::VFMultipleLoad &&
-           "Expected a load recipe");
-    // We need to extract each unroll part as a subvector.
-    auto ExtractPart0 = Builder.createNaryOp(
+    return;
+  }
+
+  assert(VPI->getOpcode() == VPInstruction::VFMultipleLoad &&
+         "Expected a VFMultipleLoad instruction");
+  // We need to extract each unroll part as a subvector.
+  auto *ExtractPart0 =
+      Builder.createNaryOp(VPInstruction::ExtractVectorForPart,
+                           {Groups[0]->getVPSingleValue(), getConstantInt(0)});
+  // First VPI with an extract of the first unroll part (ExtractPart0).
+  VPI->getVPSingleValue()->replaceUsesWithIf(
+      ExtractPart0, [&](VPUser &U, unsigned) { return &U != ExtractPart0; });
+  ToSkip.insert(ExtractPart0);
+
+  // Create extracts for the remaining unroll parts and remap later uses of
+  // ExtractPart0 to the correct unrolled part.
+  for (unsigned Part = 1; Part != UF; ++Part) {
+    VPInstruction *Group = Groups[Part / VFMultiple];
+    unsigned IndexInGroup = Part % VFMultiple;
+    auto *Extract = Builder.createNaryOp(
         VPInstruction::ExtractVectorForPart,
-        {Groups[0]->getVPSingleValue(), getConstantInt(0)});
-    // First VPI with an extract of the first unroll part (ExtractPart0).
-    VPI->getVPSingleValue()->replaceUsesWithIf(
-        ExtractPart0, [&](VPUser &U, unsigned) { return &U != ExtractPart0; });
-    ToSkip.insert(ExtractPart0);
-
-    // Create extracts for the remaining unroll parts and remap later uses of
-    // ExtractPart0 to the correct unrolled part.
-    for (unsigned Part = 1; Part != UF; ++Part) {
-      VPInstruction *Group = Groups[Part / VFMultiple];
-      unsigned IndexInGroup = Part % VFMultiple;
-      auto *Extract = Builder.createNaryOp(
-          VPInstruction::ExtractVectorForPart,
-          {Group->getVPSingleValue(), getConstantInt(IndexInGroup)});
-      addRecipeForPart(ExtractPart0, Extract, Part);
-      ToSkip.insert(Extract);
-    }
+        {Group->getVPSingleValue(), getConstantInt(IndexInGroup)});
+    addRecipeForPart(ExtractPart0, Extract, Part);
+    ToSkip.insert(Extract);
   }
 }
 



More information about the llvm-commits mailing list