[llvm] [SeparateConstOffsetFromGEP] Share scalable vector GEP bases (PR #214769)

Jacob Crawley via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 11 04:54:32 PDT 2026


https://github.com/jacob-crawley updated https://github.com/llvm/llvm-project/pull/214769

>From 4595e371a1c1c25c1a6d451c4b936d80adf5a6ef Mon Sep 17 00:00:00 2001
From: Jacob Crawley <jacob.crawley at arm.com>
Date: Fri, 7 Aug 2026 15:08:21 +0000
Subject: [PATCH 1/4] [SeparateConstOffsetFromGEP] Share scalable vector GEP
 bases

Recognize scalable vector GEPs with a common varying index and
loop-invariant splat offsets. Compute the varying GEP once and express
each remaining offset as a scalar byte offset, avoiding repeated vector
address calculations in unrolled loops.
---
 .../Scalar/SeparateConstOffsetFromGEP.cpp     | 228 ++++++++++++++++++
 llvm/test/CodeGen/AArch64/sve-vector-gep.ll   |  34 +++
 .../scalable-vector-gep-common-base.ll        |  67 +++++
 3 files changed, 329 insertions(+)
 create mode 100644 llvm/test/CodeGen/AArch64/sve-vector-gep.ll
 create mode 100644 llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll

diff --git a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
index 4870b8c888279..481e57cf55981 100644
--- a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
+++ b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
@@ -138,6 +138,7 @@
 #include "llvm/Analysis/TargetLibraryInfo.h"
 #include "llvm/Analysis/TargetTransformInfo.h"
 #include "llvm/Analysis/ValueTracking.h"
+#include "llvm/Analysis/VectorUtils.h"
 #include "llvm/IR/BasicBlock.h"
 #include "llvm/IR/Constant.h"
 #include "llvm/IR/Constants.h"
@@ -156,6 +157,7 @@
 #include "llvm/IR/Type.h"
 #include "llvm/IR/User.h"
 #include "llvm/IR/Value.h"
+#include "llvm/IR/ValueHandle.h"
 #include "llvm/InitializePasses.h"
 #include "llvm/Pass.h"
 #include "llvm/Support/Casting.h"
@@ -394,6 +396,24 @@ class SeparateConstOffsetFromGEP {
     return {B, A};
   }
 
+  struct VectorGEPOffsetTerm {
+    Value *Scalar;
+    bool IsSub;
+  };
+
+  struct VectorGEPCandidate {
+    GetElementPtrInst *GEP;
+    Value *VaryingIndex;
+    SmallVector<VectorGEPOffsetTerm, 4> OffsetTerms;
+    uint64_t Stride;
+  };
+
+  bool collectVectorGEPCandidate(GetElementPtrInst *GEP,
+                                 VectorGEPCandidate &Candidate);
+  static bool haveSameVectorGEPOffset(const VectorGEPCandidate &LHS,
+                                      const VectorGEPCandidate &RHS);
+  bool shareScalableVectorGEPBase(BasicBlock &BB);
+
   /// Tries to split the given GEP into a variadic base and a constant offset,
   /// and returns true if the splitting succeeds.
   bool splitGEP(GetElementPtrInst *GEP);
@@ -1172,6 +1192,211 @@ bool SeparateConstOffsetFromGEP::reorderGEP(GetElementPtrInst *GEP,
   return true;
 }
 
+bool SeparateConstOffsetFromGEP::haveSameVectorGEPOffset(
+    const VectorGEPCandidate &LHS, const VectorGEPCandidate &RHS) {
+  if (LHS.OffsetTerms.size() != RHS.OffsetTerms.size())
+    return false;
+
+  for (unsigned I = 0; I != LHS.OffsetTerms.size(); ++I) {
+    const VectorGEPOffsetTerm &LT = LHS.OffsetTerms[I];
+    const VectorGEPOffsetTerm &RT = RHS.OffsetTerms[I];
+
+    if (LT.Scalar != RT.Scalar || LT.IsSub != RT.IsSub)
+      return false;
+  }
+
+  return true;
+}
+
+/// Collect a scalable vector GEP whose index is a varying value plus or minus
+/// loop invariant splats. The invariant terms become scalar byte offsets.
+bool SeparateConstOffsetFromGEP::collectVectorGEPCandidate(
+    GetElementPtrInst *GEP, VectorGEPCandidate &Candidate) {
+  if (GEP->getNumIndices() != 1 || !GEP->getPointerOperandType()->isPointerTy())
+    return false;
+
+  auto *GEPType = dyn_cast<VectorType>(GEP->getType());
+  if (!GEPType || !GEPType->getElementCount().isScalable())
+    return false;
+
+  if (GEP->isInBounds() || GEP->hasNoUnsignedWrap() ||
+      GEP->hasNoUnsignedSignedWrap())
+    return false;
+
+  Value *Index = GEP->getOperand(1);
+  auto *IndexType = dyn_cast<VectorType>(Index->getType());
+  if (!IndexType || !IndexType->getElementCount().isScalable() ||
+      !IndexType->getElementType()->isIntegerTy())
+    return false;
+
+  Type *PointerIndexType = DL->getIndexType(GEP->getPointerOperandType());
+  if (IndexType->getElementType() != PointerIndexType)
+    return false;
+
+  TypeSize ElementSize = DL->getTypeAllocSize(GEP->getSourceElementType());
+  if (ElementSize.isScalable())
+    return false;
+
+  Loop *L = LI->getLoopFor(GEP->getParent());
+  if (!L)
+    return false;
+
+  SmallVector<VectorGEPOffsetTerm, 4> OffsetTerms;
+  Value *VaryingIndex = Index;
+
+  // Peel loop-invariant splats from an add/sub chain.
+  while (auto *BO = dyn_cast<BinaryOperator>(VaryingIndex)) {
+    unsigned Opcode = BO->getOpcode();
+    if (Opcode != Instruction::Add && Opcode != Instruction::Sub)
+      break;
+
+    Value *NextIndex = nullptr;
+    Value *ScalarOffset = nullptr;
+    bool IsSub = false;
+
+    auto GetInvariantSplat = [&](Value *V) -> Value * {
+      Value *Splat = getSplatValue(V);
+      if (!Splat || Splat->getType() != IndexType->getElementType() ||
+          !L->isLoopInvariant(Splat))
+        return nullptr;
+      return Splat;
+    };
+
+    unsigned SplatOperand = 1;
+    Value *Splat = GetInvariantSplat(BO->getOperand(SplatOperand));
+
+    if (!Splat && Opcode == Instruction::Add) {
+      SplatOperand = 0;
+      Splat = GetInvariantSplat(BO->getOperand(SplatOperand));
+    }
+
+    if (Splat) {
+      NextIndex = BO->getOperand(1 - SplatOperand);
+      ScalarOffset = Splat;
+      IsSub = Opcode == Instruction::Sub;
+    }
+
+    if (!NextIndex)
+      break;
+
+    OffsetTerms.push_back({ScalarOffset, IsSub});
+    VaryingIndex = NextIndex;
+  }
+
+  if (OffsetTerms.empty())
+    return false;
+
+  Candidate = {GEP, VaryingIndex, std::move(OffsetTerms),
+               ElementSize.getFixedValue()};
+
+  return true;
+}
+
+/// Find compatible scalable vector GEPs in \p BB that share a base pointer and
+/// varying index. Rewrite them to share one scaled vector GEP, with each
+/// original GEP represented by a scalar byte offset from the common address.
+bool SeparateConstOffsetFromGEP::shareScalableVectorGEPBase(BasicBlock &BB) {
+  SmallVector<VectorGEPCandidate, 8> Candidates;
+
+  for (Instruction &I : BB) {
+    auto *GEP = dyn_cast<GetElementPtrInst>(&I);
+    if (!GEP)
+      continue;
+
+    VectorGEPCandidate Candidate;
+    if (collectVectorGEPCandidate(GEP, Candidate))
+      Candidates.push_back(std::move(Candidate));
+  }
+
+  SmallVector<bool, 8> Rewritten(Candidates.size(), false);
+  SmallVector<WeakTrackingVH, 8> DeadIndices;
+  bool Changed = false;
+
+  for (unsigned I = 0; I != Candidates.size(); ++I) {
+    if (Rewritten[I])
+      continue;
+
+    SmallVector<unsigned, 4> Group;
+    Group.push_back(I);
+
+    const VectorGEPCandidate &Leader = Candidates[I];
+    for (unsigned J = I + 1; J != Candidates.size(); ++J) {
+      if (Rewritten[J])
+        continue;
+
+      const VectorGEPCandidate &Other = Candidates[J];
+      if (Other.GEP->getPointerOperand() == Leader.GEP->getPointerOperand() &&
+          Other.GEP->getSourceElementType() ==
+              Leader.GEP->getSourceElementType() &&
+          Other.VaryingIndex == Leader.VaryingIndex)
+        Group.push_back(J);
+    }
+
+    if (Group.size() < 2)
+      continue;
+
+    bool HasDistinctOffset = false;
+    for (unsigned K = 1; K != Group.size(); ++K) {
+      if (!haveSameVectorGEPOffset(Leader, Candidates[Group[K]])) {
+        HasDistinctOffset = true;
+        break;
+      }
+    }
+
+    if (!HasDistinctOffset)
+      continue;
+
+    GetElementPtrInst *FirstGEP = Leader.GEP;
+    IRBuilder<> BaseBuilder(FirstGEP);
+    BaseBuilder.SetCurrentDebugLocation(FirstGEP->getDebugLoc());
+
+    // Computing the varying portion once avoids a vector multiply-add for every
+    // unrolled part. The remaining uniform offsets can be calculated scalarly.
+    Value *CommonBase = BaseBuilder.CreateGEP(
+        FirstGEP->getSourceElementType(), FirstGEP->getPointerOperand(),
+        Leader.VaryingIndex, "vector.gep.base");
+
+    for (unsigned CandidateIndex : Group) {
+      VectorGEPCandidate &Current = Candidates[CandidateIndex];
+      GetElementPtrInst *GEP = Current.GEP;
+
+      IRBuilder<> Builder(GEP);
+      Builder.SetCurrentDebugLocation(GEP->getDebugLoc());
+
+      Type *OffsetType =
+          cast<VectorType>(GEP->getOperand(1)->getType())->getElementType();
+      Value *Offset = ConstantInt::get(OffsetType, 0);
+
+      for (const VectorGEPOffsetTerm &Term : Current.OffsetTerms) {
+        Offset =
+            Term.IsSub
+                ? Builder.CreateSub(Offset, Term.Scalar, "vector.gep.offset")
+                : Builder.CreateAdd(Offset, Term.Scalar, "vector.gep.offset");
+      }
+
+      Value *ByteOffset = Builder.CreateMul(
+          Offset, ConstantInt::get(OffsetType, Current.Stride),
+          "vector.gep.byte.offset");
+      Value *NewGEP =
+          Builder.CreatePtrAdd(CommonBase, ByteOffset, GEP->getName());
+
+      if (auto *NewI = dyn_cast<Instruction>(NewGEP)) {
+        NewI->copyMetadata(*GEP);
+        NewI->takeName(GEP);
+      }
+
+      DeadIndices.emplace_back(GEP->getOperand(1));
+      GEP->replaceAllUsesWith(NewGEP);
+      GEP->eraseFromParent();
+      Rewritten[CandidateIndex] = true;
+      Changed = true;
+    }
+  }
+
+  RecursivelyDeleteTriviallyDeadInstructionsPermissive(DeadIndices, TLI);
+  return Changed;
+}
+
 bool SeparateConstOffsetFromGEP::splitGEP(GetElementPtrInst *GEP) {
   // Skip vector GEPs.
   if (GEP->getType()->isVectorTy())
@@ -1414,6 +1639,9 @@ bool SeparateConstOffsetFromGEP::run(Function &F) {
         Changed |= splitGEP(GEP);
     // No need to split GEP ConstantExprs because all its indices are constant
     // already.
+
+    if (LowerGEP)
+      Changed |= shareScalableVectorGEPBase(*B);
   }
 
   Changed |= reuniteExts(F);
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
new file mode 100644
index 0000000000000..10ff1ed76b7e0
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
@@ -0,0 +1,34 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --filter "movprfx" --filter "m(ad|la).*z" --filter "(add|sub).*z" --version 6
+; RUN: opt -S -passes=loop-vectorize -force-vector-width=2 -force-vector-interleave=4  -scalable-vectorization=on < %s | llc -O3 -aarch64-enable-gep-opt=true | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @scalable_vector_geps(i64 %n, ptr %base, ptr %out) #0 {
+; CHECK-LABEL: scalable_vector_geps:
+; CHECK:    movprfx z7, z0
+; CHECK:    sub z7.d, z7.d, #1 // =0x1
+; CHECK:    movprfx z7, z2
+; CHECK:    mla z7.d, p0/m, z0.d, z3.d
+; CHECK:    sub z0.d, z0.d, z1.d
+; CHECK:    movprfx z16, z7
+; CHECK:    sub z16.d, z16.d, #56 // =0x38
+; CHECK:    add z17.d, z7.d, z4.d
+; CHECK:    add z18.d, z7.d, z5.d
+; CHECK:    add z7.d, z7.d, z6.d
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %n, %entry ], [ %next, %loop ]
+  %next = add nsw i64 %iv, -1
+  %src = getelementptr [56 x i8], ptr %base, i64 %next
+  %dst = getelementptr inbounds nuw [8 x i8], ptr %out, i64 %next
+  store ptr %src, ptr %dst, align 8
+  %continue = icmp samesign ugt i64 %iv, 1
+  br i1 %continue, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+attributes #0 = { nounwind "target-cpu"="neoverse-v2" }
diff --git a/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll b/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll
new file mode 100644
index 0000000000000..35246c73b634c
--- /dev/null
+++ b/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll
@@ -0,0 +1,67 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes='separate-const-offset-from-gep<lower-gep>' < %s | FileCheck %s
+
+target triple = "aarch64-linux-gnu"
+
+define void @scalable_gep_common_base(ptr %base, ptr %out0, ptr %out1, i64 %offset, i1 %cond) {
+; CHECK-LABEL: define void @scalable_gep_common_base(
+; CHECK-SAME: ptr [[BASE:%.*]], ptr [[OUT0:%.*]], ptr [[OUT1:%.*]], i64 [[OFFSET:%.*]], i1 [[COND:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[INSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[OFFSET]], i64 0
+; CHECK-NEXT:    [[SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[INSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[ENTRY]] ], [ [[NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[NEXT]] = sub <vscale x 2 x i64> [[IV]], [[SPLAT]]
+; CHECK-NEXT:    [[VECTOR_GEP_BASE:%.*]] = getelementptr [56 x i8], ptr [[BASE]], <vscale x 2 x i64> [[IV]]
+; CHECK-NEXT:    [[GEP0:%.*]] = getelementptr i8, <vscale x 2 x ptr> [[VECTOR_GEP_BASE]], i64 -56
+; CHECK-NEXT:    [[VECTOR_GEP_OFFSET:%.*]] = sub i64 -1, [[OFFSET]]
+; CHECK-NEXT:    [[VECTOR_GEP_BYTE_OFFSET:%.*]] = mul i64 [[VECTOR_GEP_OFFSET]], 56
+; CHECK-NEXT:    [[GEP1:%.*]] = getelementptr i8, <vscale x 2 x ptr> [[VECTOR_GEP_BASE]], i64 [[VECTOR_GEP_BYTE_OFFSET]]
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[GEP0]], ptr [[OUT0]], align 16
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[GEP1]], ptr [[OUT1]], align 16
+; CHECK-NEXT:    br i1 [[COND]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %insert = insertelement <vscale x 2 x i64> poison, i64 %offset, i64 0
+  %splat = shufflevector <vscale x 2 x i64> %insert,
+  <vscale x 2 x i64> poison,
+  <vscale x 2 x i32> zeroinitializer
+  br label %loop
+
+loop:
+  %iv = phi <vscale x 2 x i64> [ zeroinitializer, %entry ], [ %next, %loop ]
+  %index0 = add <vscale x 2 x i64> %iv, splat (i64 -1)
+  %next = sub <vscale x 2 x i64> %iv, %splat
+  %index1 = add <vscale x 2 x i64> %next, splat (i64 -1)
+  %gep0 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index0
+  %gep1 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index1
+  store <vscale x 2 x ptr> %gep0, ptr %out0
+  store <vscale x 2 x ptr> %gep1, ptr %out1
+
+  br i1 %cond, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+define void @identical_scalable_geps(ptr %base, ptr %out0, ptr %out1,
+; CHECK-LABEL: define void @identical_scalable_geps(
+; CHECK-SAME: ptr [[BASE:%.*]], ptr [[OUT0:%.*]], ptr [[OUT1:%.*]], <vscale x 2 x i64> [[IV:%.*]]) {
+; CHECK-NEXT:    [[INDEX:%.*]] = add <vscale x 2 x i64> [[IV]], splat (i64 -1)
+; CHECK-NEXT:    [[GEP0:%.*]] = getelementptr [56 x i8], ptr [[BASE]], <vscale x 2 x i64> [[INDEX]]
+; CHECK-NEXT:    [[GEP1:%.*]] = getelementptr [56 x i8], ptr [[BASE]], <vscale x 2 x i64> [[INDEX]]
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[GEP0]], ptr [[OUT0]], align 16
+; CHECK-NEXT:    store <vscale x 2 x ptr> [[GEP1]], ptr [[OUT1]], align 16
+; CHECK-NEXT:    ret void
+;
+  <vscale x 2 x i64> %iv) {
+  %index = add <vscale x 2 x i64> %iv, splat (i64 -1)
+  %gep0 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index
+  %gep1 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index
+  store <vscale x 2 x ptr> %gep0, ptr %out0
+  store <vscale x 2 x ptr> %gep1, ptr %out1
+  ret void
+}

>From b6829931cf51a416542be800ea8d0788883a1dd6 Mon Sep 17 00:00:00 2001
From: Jacob Crawley <jacob.crawley at arm.com>
Date: Mon, 10 Aug 2026 14:02:23 +0000
Subject: [PATCH 2/4] Update codegen test

---
 llvm/test/CodeGen/AArch64/sve-vector-gep.ll | 87 ++++++++++++++++-----
 1 file changed, 67 insertions(+), 20 deletions(-)

diff --git a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
index 10ff1ed76b7e0..19e1460bcd0a7 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
@@ -1,31 +1,78 @@
-; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --filter "movprfx" --filter "m(ad|la).*z" --filter "(add|sub).*z" --version 6
-; RUN: opt -S -passes=loop-vectorize -force-vector-width=2 -force-vector-interleave=4  -scalable-vectorization=on < %s | llc -O3 -aarch64-enable-gep-opt=true | FileCheck %s
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -O3 -aarch64-enable-gep-opt=true < %s | FileCheck %s
 
 target triple = "aarch64-unknown-linux-gnu"
 
-define void @scalable_vector_geps(i64 %n, ptr %base, ptr %out) #0 {
+define void @scalable_vector_geps(i64 %n, ptr %base, ptr %out0, ptr %out1, ptr %out2, ptr %out3, i1 %cond) #0 {
 ; CHECK-LABEL: scalable_vector_geps:
-; CHECK:    movprfx z7, z0
-; CHECK:    sub z7.d, z7.d, #1 // =0x1
-; CHECK:    movprfx z7, z2
-; CHECK:    mla z7.d, p0/m, z0.d, z3.d
-; CHECK:    sub z0.d, z0.d, z1.d
-; CHECK:    movprfx z16, z7
-; CHECK:    sub z16.d, z16.d, #56 // =0x38
-; CHECK:    add z17.d, z7.d, z4.d
-; CHECK:    add z18.d, z7.d, z5.d
-; CHECK:    add z7.d, z7.d, z6.d
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    cntd x8
+; CHECK-NEXT:    cnth x9
+; CHECK-NEXT:    mov z3.d, #56 // =0x38
+; CHECK-NEXT:    mov z2.d, x1
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    mov z0.d, x9
+; CHECK-NEXT:    mvn x9, x8
+; CHECK-NEXT:    index z1.d, x0, #-1
+; CHECK-NEXT:    lsl x10, x9, #6
+; CHECK-NEXT:    sub x10, x10, x9, lsl #3
+; CHECK-NEXT:    sub x9, x9, x8
+; CHECK-NEXT:    lsl x11, x9, #6
+; CHECK-NEXT:    sub x8, x9, x8
+; CHECK-NEXT:    mov z4.d, x10
+; CHECK-NEXT:    sub x11, x11, x9, lsl #3
+; CHECK-NEXT:    lsl x9, x8, #6
+; CHECK-NEXT:    sub x8, x9, x8, lsl #3
+; CHECK-NEXT:    mov z5.d, x11
+; CHECK-NEXT:    mov z6.d, x8
+; CHECK-NEXT:    .p2align 5, , 16
+; CHECK-NEXT:  .LBB0_1: // %loop
+; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    movprfx z7, z2
+; CHECK-NEXT:    mla z7.d, p0/m, z1.d, z3.d
+; CHECK-NEXT:    sub z1.d, z1.d, z0.d
+; CHECK-NEXT:    movprfx z16, z7
+; CHECK-NEXT:    sub z16.d, z16.d, #56 // =0x38
+; CHECK-NEXT:    add z17.d, z7.d, z4.d
+; CHECK-NEXT:    add z18.d, z7.d, z5.d
+; CHECK-NEXT:    add z7.d, z7.d, z6.d
+; CHECK-NEXT:    str z16, [x2]
+; CHECK-NEXT:    str z17, [x3]
+; CHECK-NEXT:    str z18, [x4]
+; CHECK-NEXT:    str z7, [x5]
+; CHECK-NEXT:    tbnz w6, #0, .LBB0_1
+; CHECK-NEXT:  // %bb.2: // %exit
+; CHECK-NEXT:    ret
 entry:
+  %vscale = call i64 @llvm.vscale.i64()
+  %step = shl nuw i64 %vscale, 1
+  %insert.step = insertelement <vscale x 2 x i64> poison, i64 %step, i64 0
+  %splat.step = shufflevector <vscale x 2 x i64> %insert.step, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %insert.n = insertelement <vscale x 2 x i64> poison, i64 %n, i64 0
+  %splat.n = shufflevector <vscale x 2 x i64> %insert.n, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %initial = sub <vscale x 2 x i64> %splat.n, %lane
   br label %loop
 
 loop:
-  %iv = phi i64 [ %n, %entry ], [ %next, %loop ]
-  %next = add nsw i64 %iv, -1
-  %src = getelementptr [56 x i8], ptr %base, i64 %next
-  %dst = getelementptr inbounds nuw [8 x i8], ptr %out, i64 %next
-  store ptr %src, ptr %dst, align 8
-  %continue = icmp samesign ugt i64 %iv, 1
-  br i1 %continue, label %loop, label %exit
+  %iv0 = phi <vscale x 2 x i64> [ %initial, %entry ], [ %iv4, %loop ]
+  %iv1 = sub <vscale x 2 x i64> %iv0, %splat.step
+  %iv2 = sub <vscale x 2 x i64> %iv1, %splat.step
+  %iv3 = sub <vscale x 2 x i64> %iv2, %splat.step
+  %iv4 = sub <vscale x 2 x i64> %iv3, %splat.step
+  %index0 = add <vscale x 2 x i64> %iv0, splat (i64 -1)
+  %index1 = add <vscale x 2 x i64> %iv1, splat (i64 -1)
+  %index2 = add <vscale x 2 x i64> %iv2, splat (i64 -1)
+  %index3 = add <vscale x 2 x i64> %iv3, splat (i64 -1)
+  %gep0 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index0
+  %gep1 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index1
+  %gep2 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index2
+  %gep3 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index3
+  store <vscale x 2 x ptr> %gep0, ptr %out0
+  store <vscale x 2 x ptr> %gep1, ptr %out1
+  store <vscale x 2 x ptr> %gep2, ptr %out2
+  store <vscale x 2 x ptr> %gep3, ptr %out3
+  br i1 %cond, label %loop, label %exit
 
 exit:
   ret void

>From 47c3298c078a2f6b3f7bea62429406c3179bb816 Mon Sep 17 00:00:00 2001
From: Jacob Crawley <jacob.crawley at arm.com>
Date: Tue, 11 Aug 2026 09:36:43 +0000
Subject: [PATCH 3/4] Apply scalable only pass by default

---
 llvm/include/llvm/Transforms/Scalar.h         |  9 ++++
 .../Target/AArch64/AArch64TargetMachine.cpp   |  9 ++++
 .../Scalar/SeparateConstOffsetFromGEP.cpp     | 48 ++++++++++++++-----
 llvm/test/CodeGen/AArch64/sve-vector-gep.ll   |  2 +-
 4 files changed, 54 insertions(+), 14 deletions(-)

diff --git a/llvm/include/llvm/Transforms/Scalar.h b/llvm/include/llvm/Transforms/Scalar.h
index 24a127b1ffb17..4fc8dbfe4f558 100644
--- a/llvm/include/llvm/Transforms/Scalar.h
+++ b/llvm/include/llvm/Transforms/Scalar.h
@@ -171,6 +171,15 @@ LLVM_ABI FunctionPass *createPartiallyInlineLibCallsPass();
 LLVM_ABI FunctionPass *
 createSeparateConstOffsetFromGEPPass(bool LowerGEP = false);
 
+//===----------------------------------------------------------------------===//
+//
+// ShareScalableVectorGEPBase - Share the varying vector
+// base of compatible scalable-vector GEPs without
+// applying the scalar GEP transformations performed by
+// SeparateConstOffsetFromGEP.
+//
+LLVM_ABI FunctionPass *createShareScalableVectorGEPBasePass();
+
 //===----------------------------------------------------------------------===//
 //
 // SpeculativeExecution - Aggressively hoist instructions to enable
diff --git a/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp b/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
index dee61a4f2ac8c..7fe4b177de3af 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
@@ -133,6 +133,11 @@ static cl::opt<bool>
                  cl::desc("Enable optimizations on complex GEPs"),
                  cl::init(false));
 
+static cl::opt<bool> EnableScalableVectorGEPOpt(
+    "aarch64-enable-scalable-vector-gep-opt", cl::Hidden,
+    cl::desc("Share common bases between scalable-vector GEPs"),
+    cl::init(true));
+
 static cl::opt<bool>
     EnableSelectOpt("aarch64-select-opt", cl::Hidden,
                     cl::desc("Enable select to branch optimizations"),
@@ -683,6 +688,10 @@ void AArch64PassConfig::addIRPasses() {
     // Do loop invariant code motion in case part of the lowered result is
     // invariant.
     addPass(createLICMPass());
+  } else if (TM->getOptLevel() >= CodeGenOptLevel::Default &&
+             EnableScalableVectorGEPOpt) {
+    addPass(createShareScalableVectorGEPBasePass());
+    addPass(createLICMPass());
   }
 
   TargetPassConfig::addIRPasses();
diff --git a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
index 481e57cf55981..2a51b2dcbb7f3 100644
--- a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
+++ b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
@@ -353,12 +353,20 @@ class SeparateConstOffsetFromGEPLegacyPass : public FunctionPass {
 public:
   static char ID;
 
-  SeparateConstOffsetFromGEPLegacyPass(bool LowerGEP = false)
-      : FunctionPass(ID), LowerGEP(LowerGEP) {
+  SeparateConstOffsetFromGEPLegacyPass(
+      bool LowerGEP = false, bool ShareScalableVectorGEPBaseOnly = false)
+      : FunctionPass(ID), LowerGEP(LowerGEP),
+        ShareScalableVectorGEPBaseOnly(ShareScalableVectorGEPBaseOnly) {
     initializeSeparateConstOffsetFromGEPLegacyPassPass(
         *PassRegistry::getPassRegistry());
   }
 
+  StringRef getPassName() const override {
+    if (ShareScalableVectorGEPBaseOnly)
+      return "Share bases of scalable-vector GEPs";
+    return FunctionPass::getPassName();
+  }
+
   void getAnalysisUsage(AnalysisUsage &AU) const override {
     AU.addRequired<DominatorTreeWrapperPass>();
     AU.addRequired<TargetTransformInfoWrapperPass>();
@@ -371,6 +379,7 @@ class SeparateConstOffsetFromGEPLegacyPass : public FunctionPass {
 
 private:
   bool LowerGEP;
+  bool ShareScalableVectorGEPBaseOnly;
 };
 
 /// A pass that tries to split every GEP in the function into a variadic
@@ -380,8 +389,10 @@ class SeparateConstOffsetFromGEP {
 public:
   SeparateConstOffsetFromGEP(
       DominatorTree *DT, LoopInfo *LI, TargetLibraryInfo *TLI,
-      function_ref<TargetTransformInfo &(Function &)> GetTTI, bool LowerGEP)
-      : DT(DT), LI(LI), TLI(TLI), GetTTI(GetTTI), LowerGEP(LowerGEP) {}
+      function_ref<TargetTransformInfo &(Function &)> GetTTI, bool LowerGEP,
+      bool ShareScalableVectorGEPBaseOnly = false)
+      : DT(DT), LI(LI), TLI(TLI), GetTTI(GetTTI), LowerGEP(LowerGEP),
+        ShareScalableVectorGEPBaseOnly(ShareScalableVectorGEPBaseOnly) {}
 
   bool run(Function &F);
 
@@ -501,6 +512,10 @@ class SeparateConstOffsetFromGEP {
   /// multiple GEPs with a single index.
   bool LowerGEP;
 
+  /// Run only scalable-vector base sharing, without splitting or lowering
+  /// scalar GEPs.
+  bool ShareScalableVectorGEPBaseOnly;
+
   DenseMap<ExprKey, SmallVector<Instruction *, 2>> DominatingAdds;
   DenseMap<ExprKey, SmallVector<Instruction *, 2>> DominatingSubs;
 };
@@ -527,6 +542,10 @@ FunctionPass *llvm::createSeparateConstOffsetFromGEPPass(bool LowerGEP) {
   return new SeparateConstOffsetFromGEPLegacyPass(LowerGEP);
 }
 
+FunctionPass *llvm::createShareScalableVectorGEPBasePass() {
+  return new SeparateConstOffsetFromGEPLegacyPass(false, true);
+}
+
 // Checks if it is safe to reorder an add/sext result used in a GEP.
 //
 // An inbounds GEP does not guarantee that the index is non-negative.
@@ -1618,7 +1637,8 @@ bool SeparateConstOffsetFromGEPLegacyPass::runOnFunction(Function &F) {
   auto GetTTI = [this](Function &F) -> TargetTransformInfo & {
     return this->getAnalysis<TargetTransformInfoWrapperPass>().getTTI(F);
   };
-  SeparateConstOffsetFromGEP Impl(DT, LI, TLI, GetTTI, LowerGEP);
+  SeparateConstOffsetFromGEP Impl(DT, LI, TLI, GetTTI, LowerGEP,
+                                  ShareScalableVectorGEPBaseOnly);
   return Impl.run(F);
 }
 
@@ -1634,17 +1654,19 @@ bool SeparateConstOffsetFromGEP::run(Function &F) {
     if (!DT->isReachableFromEntry(B))
       continue;
 
-    for (Instruction &I : llvm::make_early_inc_range(*B))
-      if (GetElementPtrInst *GEP = dyn_cast<GetElementPtrInst>(&I))
-        Changed |= splitGEP(GEP);
-    // No need to split GEP ConstantExprs because all its indices are constant
-    // already.
-
-    if (LowerGEP)
+    if (!ShareScalableVectorGEPBaseOnly) {
+      for (Instruction &I : llvm::make_early_inc_range(*B))
+        if (GetElementPtrInst *GEP = dyn_cast<GetElementPtrInst>(&I))
+          Changed |= splitGEP(GEP);
+      // No need to split GEP ConstantExprs because all its indices are
+      // constant already.
+    }
+    if (LowerGEP || ShareScalableVectorGEPBaseOnly)
       Changed |= shareScalableVectorGEPBase(*B);
   }
 
-  Changed |= reuniteExts(F);
+  if (!ShareScalableVectorGEPBaseOnly)
+    Changed |= reuniteExts(F);
 
   if (VerifyNoDeadCode)
     verifyNoDeadCode(F);
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
index 19e1460bcd0a7..eb7650c913d0c 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -O3 -aarch64-enable-gep-opt=true < %s | FileCheck %s
+; RUN: llc -O3 < %s | FileCheck %s
 
 target triple = "aarch64-unknown-linux-gnu"
 

>From cded048df0e07af75e7c9d2ea1ca2ea13571d528 Mon Sep 17 00:00:00 2001
From: Jacob Crawley <jacob.crawley at arm.com>
Date: Tue, 11 Aug 2026 11:54:12 +0000
Subject: [PATCH 4/4] rm LICM from pass

---
 .../Target/AArch64/AArch64TargetMachine.cpp   |  1 -
 .../Scalar/SeparateConstOffsetFromGEP.cpp     | 16 ++++++++----
 llvm/test/CodeGen/AArch64/O3-pipeline.ll      |  2 ++
 llvm/test/CodeGen/AArch64/sve-vector-gep.ll   | 26 ++++++++++---------
 .../scalable-vector-gep-common-base.ll        |  4 +--
 5 files changed, 29 insertions(+), 20 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp b/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
index 7fe4b177de3af..652d17d6256c8 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
@@ -691,7 +691,6 @@ void AArch64PassConfig::addIRPasses() {
   } else if (TM->getOptLevel() >= CodeGenOptLevel::Default &&
              EnableScalableVectorGEPOpt) {
     addPass(createShareScalableVectorGEPBasePass());
-    addPass(createLICMPass());
   }
 
   TargetPassConfig::addIRPasses();
diff --git a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
index 2a51b2dcbb7f3..cb3ff127c355e 100644
--- a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
+++ b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
@@ -1368,6 +1368,10 @@ bool SeparateConstOffsetFromGEP::shareScalableVectorGEPBase(BasicBlock &BB) {
     GetElementPtrInst *FirstGEP = Leader.GEP;
     IRBuilder<> BaseBuilder(FirstGEP);
     BaseBuilder.SetCurrentDebugLocation(FirstGEP->getDebugLoc());
+    Loop *L = LI->getLoopFor(FirstGEP->getParent());
+    assert(L && L->getLoopPreheader() &&
+           "candidate must belong to a loop with a preheader");
+    Instruction *PreheaderTerminator = L->getLoopPreheader()->getTerminator();
 
     // Computing the varying portion once avoids a vector multiply-add for every
     // unrolled part. The remaining uniform offsets can be calculated scalarly.
@@ -1379,6 +1383,8 @@ bool SeparateConstOffsetFromGEP::shareScalableVectorGEPBase(BasicBlock &BB) {
       VectorGEPCandidate &Current = Candidates[CandidateIndex];
       GetElementPtrInst *GEP = Current.GEP;
 
+      IRBuilder<> OffsetBuilder(PreheaderTerminator);
+      OffsetBuilder.SetCurrentDebugLocation(GEP->getDebugLoc());
       IRBuilder<> Builder(GEP);
       Builder.SetCurrentDebugLocation(GEP->getDebugLoc());
 
@@ -1387,13 +1393,13 @@ bool SeparateConstOffsetFromGEP::shareScalableVectorGEPBase(BasicBlock &BB) {
       Value *Offset = ConstantInt::get(OffsetType, 0);
 
       for (const VectorGEPOffsetTerm &Term : Current.OffsetTerms) {
-        Offset =
-            Term.IsSub
-                ? Builder.CreateSub(Offset, Term.Scalar, "vector.gep.offset")
-                : Builder.CreateAdd(Offset, Term.Scalar, "vector.gep.offset");
+        Offset = Term.IsSub ? OffsetBuilder.CreateSub(Offset, Term.Scalar,
+                                                      "vector.gep.offset")
+                            : OffsetBuilder.CreateAdd(Offset, Term.Scalar,
+                                                      "vector.gep.offset");
       }
 
-      Value *ByteOffset = Builder.CreateMul(
+      Value *ByteOffset = OffsetBuilder.CreateMul(
           Offset, ConstantInt::get(OffsetType, Current.Stride),
           "vector.gep.byte.offset");
       Value *NewGEP =
diff --git a/llvm/test/CodeGen/AArch64/O3-pipeline.ll b/llvm/test/CodeGen/AArch64/O3-pipeline.ll
index 9635d8bcee02c..5f6c174b31061 100644
--- a/llvm/test/CodeGen/AArch64/O3-pipeline.ll
+++ b/llvm/test/CodeGen/AArch64/O3-pipeline.ll
@@ -39,9 +39,11 @@
 ; CHECK-NEXT:       Scalar Evolution Analysis
 ; CHECK-NEXT:       Loop Data Prefetch
 ; CHECK-NEXT:       Falkor HW Prefetch Fix
+; CHECK-NEXT:       Share bases of scalable-vector GEPs
 ; CHECK-NEXT:       Module Verifier
 ; CHECK-NEXT:       Basic Alias Analysis (stateless AA impl)
 ; CHECK-NEXT:       Canonicalize natural loops
+; CHECK-NEXT:       Scalar Evolution Analysis
 ; CHECK-NEXT:       Loop Pass Manager
 ; CHECK-NEXT:         Canonicalize Freeze Instructions in Loops
 ; CHECK-NEXT:         Induction Variable Users
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
index eb7650c913d0c..5bf7eb4c834a2 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
@@ -7,14 +7,13 @@ define void @scalable_vector_geps(i64 %n, ptr %base, ptr %out0, ptr %out1, ptr %
 ; CHECK-LABEL: scalable_vector_geps:
 ; CHECK:       // %bb.0: // %entry
 ; CHECK-NEXT:    cntd x8
-; CHECK-NEXT:    cnth x9
+; CHECK-NEXT:    index z1.d, x0, #-1
 ; CHECK-NEXT:    mov z3.d, #56 // =0x38
-; CHECK-NEXT:    mov z2.d, x1
 ; CHECK-NEXT:    ptrue p0.d
-; CHECK-NEXT:    mov z0.d, x9
 ; CHECK-NEXT:    mvn x9, x8
-; CHECK-NEXT:    index z1.d, x0, #-1
+; CHECK-NEXT:    mov z0.d, x8
 ; CHECK-NEXT:    lsl x10, x9, #6
+; CHECK-NEXT:    mov z2.d, x1
 ; CHECK-NEXT:    sub x10, x10, x9, lsl #3
 ; CHECK-NEXT:    sub x9, x9, x8
 ; CHECK-NEXT:    lsl x11, x9, #6
@@ -28,18 +27,21 @@ define void @scalable_vector_geps(i64 %n, ptr %base, ptr %out0, ptr %out1, ptr %
 ; CHECK-NEXT:    .p2align 5, , 16
 ; CHECK-NEXT:  .LBB0_1: // %loop
 ; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
-; CHECK-NEXT:    movprfx z7, z2
-; CHECK-NEXT:    mla z7.d, p0/m, z1.d, z3.d
-; CHECK-NEXT:    sub z1.d, z1.d, z0.d
-; CHECK-NEXT:    movprfx z16, z7
+; CHECK-NEXT:    sub z7.d, z1.d, z0.d
+; CHECK-NEXT:    mad z1.d, p0/m, z3.d, z2.d
+; CHECK-NEXT:    sub z7.d, z7.d, z0.d
+; CHECK-NEXT:    sub z7.d, z7.d, z0.d
+; CHECK-NEXT:    movprfx z16, z1
 ; CHECK-NEXT:    sub z16.d, z16.d, #56 // =0x38
-; CHECK-NEXT:    add z17.d, z7.d, z4.d
-; CHECK-NEXT:    add z18.d, z7.d, z5.d
-; CHECK-NEXT:    add z7.d, z7.d, z6.d
+; CHECK-NEXT:    add z17.d, z1.d, z4.d
+; CHECK-NEXT:    add z18.d, z1.d, z5.d
+; CHECK-NEXT:    add z1.d, z1.d, z6.d
+; CHECK-NEXT:    sub z7.d, z7.d, z0.d
 ; CHECK-NEXT:    str z16, [x2]
 ; CHECK-NEXT:    str z17, [x3]
 ; CHECK-NEXT:    str z18, [x4]
-; CHECK-NEXT:    str z7, [x5]
+; CHECK-NEXT:    str z1, [x5]
+; CHECK-NEXT:    mov z1.d, z7.d
 ; CHECK-NEXT:    tbnz w6, #0, .LBB0_1
 ; CHECK-NEXT:  // %bb.2: // %exit
 ; CHECK-NEXT:    ret
diff --git a/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll b/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll
index 35246c73b634c..5c8aabddcef4b 100644
--- a/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll
+++ b/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll
@@ -9,14 +9,14 @@ define void @scalable_gep_common_base(ptr %base, ptr %out0, ptr %out1, i64 %offs
 ; CHECK-NEXT:  [[ENTRY:.*]]:
 ; CHECK-NEXT:    [[INSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[OFFSET]], i64 0
 ; CHECK-NEXT:    [[SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[INSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[VECTOR_GEP_OFFSET:%.*]] = sub i64 -1, [[OFFSET]]
+; CHECK-NEXT:    [[VECTOR_GEP_BYTE_OFFSET:%.*]] = mul i64 [[VECTOR_GEP_OFFSET]], 56
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[ENTRY]] ], [ [[NEXT:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[NEXT]] = sub <vscale x 2 x i64> [[IV]], [[SPLAT]]
 ; CHECK-NEXT:    [[VECTOR_GEP_BASE:%.*]] = getelementptr [56 x i8], ptr [[BASE]], <vscale x 2 x i64> [[IV]]
 ; CHECK-NEXT:    [[GEP0:%.*]] = getelementptr i8, <vscale x 2 x ptr> [[VECTOR_GEP_BASE]], i64 -56
-; CHECK-NEXT:    [[VECTOR_GEP_OFFSET:%.*]] = sub i64 -1, [[OFFSET]]
-; CHECK-NEXT:    [[VECTOR_GEP_BYTE_OFFSET:%.*]] = mul i64 [[VECTOR_GEP_OFFSET]], 56
 ; CHECK-NEXT:    [[GEP1:%.*]] = getelementptr i8, <vscale x 2 x ptr> [[VECTOR_GEP_BASE]], i64 [[VECTOR_GEP_BYTE_OFFSET]]
 ; CHECK-NEXT:    store <vscale x 2 x ptr> [[GEP0]], ptr [[OUT0]], align 16
 ; CHECK-NEXT:    store <vscale x 2 x ptr> [[GEP1]], ptr [[OUT1]], align 16



More information about the llvm-commits mailing list