[llvm] [SeparateConstOffsetFromGEP] Share scalable vector GEP bases (PR #214769)
Jacob Crawley via llvm-commits
llvm-commits at lists.llvm.org
Tue Aug 11 04:54:32 PDT 2026
https://github.com/jacob-crawley updated https://github.com/llvm/llvm-project/pull/214769
>From 4595e371a1c1c25c1a6d451c4b936d80adf5a6ef Mon Sep 17 00:00:00 2001
From: Jacob Crawley <jacob.crawley at arm.com>
Date: Fri, 7 Aug 2026 15:08:21 +0000
Subject: [PATCH 1/4] [SeparateConstOffsetFromGEP] Share scalable vector GEP
bases
Recognize scalable vector GEPs with a common varying index and
loop-invariant splat offsets. Compute the varying GEP once and express
each remaining offset as a scalar byte offset, avoiding repeated vector
address calculations in unrolled loops.
---
.../Scalar/SeparateConstOffsetFromGEP.cpp | 228 ++++++++++++++++++
llvm/test/CodeGen/AArch64/sve-vector-gep.ll | 34 +++
.../scalable-vector-gep-common-base.ll | 67 +++++
3 files changed, 329 insertions(+)
create mode 100644 llvm/test/CodeGen/AArch64/sve-vector-gep.ll
create mode 100644 llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll
diff --git a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
index 4870b8c888279..481e57cf55981 100644
--- a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
+++ b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
@@ -138,6 +138,7 @@
#include "llvm/Analysis/TargetLibraryInfo.h"
#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/Analysis/ValueTracking.h"
+#include "llvm/Analysis/VectorUtils.h"
#include "llvm/IR/BasicBlock.h"
#include "llvm/IR/Constant.h"
#include "llvm/IR/Constants.h"
@@ -156,6 +157,7 @@
#include "llvm/IR/Type.h"
#include "llvm/IR/User.h"
#include "llvm/IR/Value.h"
+#include "llvm/IR/ValueHandle.h"
#include "llvm/InitializePasses.h"
#include "llvm/Pass.h"
#include "llvm/Support/Casting.h"
@@ -394,6 +396,24 @@ class SeparateConstOffsetFromGEP {
return {B, A};
}
+ struct VectorGEPOffsetTerm {
+ Value *Scalar;
+ bool IsSub;
+ };
+
+ struct VectorGEPCandidate {
+ GetElementPtrInst *GEP;
+ Value *VaryingIndex;
+ SmallVector<VectorGEPOffsetTerm, 4> OffsetTerms;
+ uint64_t Stride;
+ };
+
+ bool collectVectorGEPCandidate(GetElementPtrInst *GEP,
+ VectorGEPCandidate &Candidate);
+ static bool haveSameVectorGEPOffset(const VectorGEPCandidate &LHS,
+ const VectorGEPCandidate &RHS);
+ bool shareScalableVectorGEPBase(BasicBlock &BB);
+
/// Tries to split the given GEP into a variadic base and a constant offset,
/// and returns true if the splitting succeeds.
bool splitGEP(GetElementPtrInst *GEP);
@@ -1172,6 +1192,211 @@ bool SeparateConstOffsetFromGEP::reorderGEP(GetElementPtrInst *GEP,
return true;
}
+bool SeparateConstOffsetFromGEP::haveSameVectorGEPOffset(
+ const VectorGEPCandidate &LHS, const VectorGEPCandidate &RHS) {
+ if (LHS.OffsetTerms.size() != RHS.OffsetTerms.size())
+ return false;
+
+ for (unsigned I = 0; I != LHS.OffsetTerms.size(); ++I) {
+ const VectorGEPOffsetTerm < = LHS.OffsetTerms[I];
+ const VectorGEPOffsetTerm &RT = RHS.OffsetTerms[I];
+
+ if (LT.Scalar != RT.Scalar || LT.IsSub != RT.IsSub)
+ return false;
+ }
+
+ return true;
+}
+
+/// Collect a scalable vector GEP whose index is a varying value plus or minus
+/// loop invariant splats. The invariant terms become scalar byte offsets.
+bool SeparateConstOffsetFromGEP::collectVectorGEPCandidate(
+ GetElementPtrInst *GEP, VectorGEPCandidate &Candidate) {
+ if (GEP->getNumIndices() != 1 || !GEP->getPointerOperandType()->isPointerTy())
+ return false;
+
+ auto *GEPType = dyn_cast<VectorType>(GEP->getType());
+ if (!GEPType || !GEPType->getElementCount().isScalable())
+ return false;
+
+ if (GEP->isInBounds() || GEP->hasNoUnsignedWrap() ||
+ GEP->hasNoUnsignedSignedWrap())
+ return false;
+
+ Value *Index = GEP->getOperand(1);
+ auto *IndexType = dyn_cast<VectorType>(Index->getType());
+ if (!IndexType || !IndexType->getElementCount().isScalable() ||
+ !IndexType->getElementType()->isIntegerTy())
+ return false;
+
+ Type *PointerIndexType = DL->getIndexType(GEP->getPointerOperandType());
+ if (IndexType->getElementType() != PointerIndexType)
+ return false;
+
+ TypeSize ElementSize = DL->getTypeAllocSize(GEP->getSourceElementType());
+ if (ElementSize.isScalable())
+ return false;
+
+ Loop *L = LI->getLoopFor(GEP->getParent());
+ if (!L)
+ return false;
+
+ SmallVector<VectorGEPOffsetTerm, 4> OffsetTerms;
+ Value *VaryingIndex = Index;
+
+ // Peel loop-invariant splats from an add/sub chain.
+ while (auto *BO = dyn_cast<BinaryOperator>(VaryingIndex)) {
+ unsigned Opcode = BO->getOpcode();
+ if (Opcode != Instruction::Add && Opcode != Instruction::Sub)
+ break;
+
+ Value *NextIndex = nullptr;
+ Value *ScalarOffset = nullptr;
+ bool IsSub = false;
+
+ auto GetInvariantSplat = [&](Value *V) -> Value * {
+ Value *Splat = getSplatValue(V);
+ if (!Splat || Splat->getType() != IndexType->getElementType() ||
+ !L->isLoopInvariant(Splat))
+ return nullptr;
+ return Splat;
+ };
+
+ unsigned SplatOperand = 1;
+ Value *Splat = GetInvariantSplat(BO->getOperand(SplatOperand));
+
+ if (!Splat && Opcode == Instruction::Add) {
+ SplatOperand = 0;
+ Splat = GetInvariantSplat(BO->getOperand(SplatOperand));
+ }
+
+ if (Splat) {
+ NextIndex = BO->getOperand(1 - SplatOperand);
+ ScalarOffset = Splat;
+ IsSub = Opcode == Instruction::Sub;
+ }
+
+ if (!NextIndex)
+ break;
+
+ OffsetTerms.push_back({ScalarOffset, IsSub});
+ VaryingIndex = NextIndex;
+ }
+
+ if (OffsetTerms.empty())
+ return false;
+
+ Candidate = {GEP, VaryingIndex, std::move(OffsetTerms),
+ ElementSize.getFixedValue()};
+
+ return true;
+}
+
+/// Find compatible scalable vector GEPs in \p BB that share a base pointer and
+/// varying index. Rewrite them to share one scaled vector GEP, with each
+/// original GEP represented by a scalar byte offset from the common address.
+bool SeparateConstOffsetFromGEP::shareScalableVectorGEPBase(BasicBlock &BB) {
+ SmallVector<VectorGEPCandidate, 8> Candidates;
+
+ for (Instruction &I : BB) {
+ auto *GEP = dyn_cast<GetElementPtrInst>(&I);
+ if (!GEP)
+ continue;
+
+ VectorGEPCandidate Candidate;
+ if (collectVectorGEPCandidate(GEP, Candidate))
+ Candidates.push_back(std::move(Candidate));
+ }
+
+ SmallVector<bool, 8> Rewritten(Candidates.size(), false);
+ SmallVector<WeakTrackingVH, 8> DeadIndices;
+ bool Changed = false;
+
+ for (unsigned I = 0; I != Candidates.size(); ++I) {
+ if (Rewritten[I])
+ continue;
+
+ SmallVector<unsigned, 4> Group;
+ Group.push_back(I);
+
+ const VectorGEPCandidate &Leader = Candidates[I];
+ for (unsigned J = I + 1; J != Candidates.size(); ++J) {
+ if (Rewritten[J])
+ continue;
+
+ const VectorGEPCandidate &Other = Candidates[J];
+ if (Other.GEP->getPointerOperand() == Leader.GEP->getPointerOperand() &&
+ Other.GEP->getSourceElementType() ==
+ Leader.GEP->getSourceElementType() &&
+ Other.VaryingIndex == Leader.VaryingIndex)
+ Group.push_back(J);
+ }
+
+ if (Group.size() < 2)
+ continue;
+
+ bool HasDistinctOffset = false;
+ for (unsigned K = 1; K != Group.size(); ++K) {
+ if (!haveSameVectorGEPOffset(Leader, Candidates[Group[K]])) {
+ HasDistinctOffset = true;
+ break;
+ }
+ }
+
+ if (!HasDistinctOffset)
+ continue;
+
+ GetElementPtrInst *FirstGEP = Leader.GEP;
+ IRBuilder<> BaseBuilder(FirstGEP);
+ BaseBuilder.SetCurrentDebugLocation(FirstGEP->getDebugLoc());
+
+ // Computing the varying portion once avoids a vector multiply-add for every
+ // unrolled part. The remaining uniform offsets can be calculated scalarly.
+ Value *CommonBase = BaseBuilder.CreateGEP(
+ FirstGEP->getSourceElementType(), FirstGEP->getPointerOperand(),
+ Leader.VaryingIndex, "vector.gep.base");
+
+ for (unsigned CandidateIndex : Group) {
+ VectorGEPCandidate &Current = Candidates[CandidateIndex];
+ GetElementPtrInst *GEP = Current.GEP;
+
+ IRBuilder<> Builder(GEP);
+ Builder.SetCurrentDebugLocation(GEP->getDebugLoc());
+
+ Type *OffsetType =
+ cast<VectorType>(GEP->getOperand(1)->getType())->getElementType();
+ Value *Offset = ConstantInt::get(OffsetType, 0);
+
+ for (const VectorGEPOffsetTerm &Term : Current.OffsetTerms) {
+ Offset =
+ Term.IsSub
+ ? Builder.CreateSub(Offset, Term.Scalar, "vector.gep.offset")
+ : Builder.CreateAdd(Offset, Term.Scalar, "vector.gep.offset");
+ }
+
+ Value *ByteOffset = Builder.CreateMul(
+ Offset, ConstantInt::get(OffsetType, Current.Stride),
+ "vector.gep.byte.offset");
+ Value *NewGEP =
+ Builder.CreatePtrAdd(CommonBase, ByteOffset, GEP->getName());
+
+ if (auto *NewI = dyn_cast<Instruction>(NewGEP)) {
+ NewI->copyMetadata(*GEP);
+ NewI->takeName(GEP);
+ }
+
+ DeadIndices.emplace_back(GEP->getOperand(1));
+ GEP->replaceAllUsesWith(NewGEP);
+ GEP->eraseFromParent();
+ Rewritten[CandidateIndex] = true;
+ Changed = true;
+ }
+ }
+
+ RecursivelyDeleteTriviallyDeadInstructionsPermissive(DeadIndices, TLI);
+ return Changed;
+}
+
bool SeparateConstOffsetFromGEP::splitGEP(GetElementPtrInst *GEP) {
// Skip vector GEPs.
if (GEP->getType()->isVectorTy())
@@ -1414,6 +1639,9 @@ bool SeparateConstOffsetFromGEP::run(Function &F) {
Changed |= splitGEP(GEP);
// No need to split GEP ConstantExprs because all its indices are constant
// already.
+
+ if (LowerGEP)
+ Changed |= shareScalableVectorGEPBase(*B);
}
Changed |= reuniteExts(F);
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
new file mode 100644
index 0000000000000..10ff1ed76b7e0
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
@@ -0,0 +1,34 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --filter "movprfx" --filter "m(ad|la).*z" --filter "(add|sub).*z" --version 6
+; RUN: opt -S -passes=loop-vectorize -force-vector-width=2 -force-vector-interleave=4 -scalable-vectorization=on < %s | llc -O3 -aarch64-enable-gep-opt=true | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @scalable_vector_geps(i64 %n, ptr %base, ptr %out) #0 {
+; CHECK-LABEL: scalable_vector_geps:
+; CHECK: movprfx z7, z0
+; CHECK: sub z7.d, z7.d, #1 // =0x1
+; CHECK: movprfx z7, z2
+; CHECK: mla z7.d, p0/m, z0.d, z3.d
+; CHECK: sub z0.d, z0.d, z1.d
+; CHECK: movprfx z16, z7
+; CHECK: sub z16.d, z16.d, #56 // =0x38
+; CHECK: add z17.d, z7.d, z4.d
+; CHECK: add z18.d, z7.d, z5.d
+; CHECK: add z7.d, z7.d, z6.d
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ %n, %entry ], [ %next, %loop ]
+ %next = add nsw i64 %iv, -1
+ %src = getelementptr [56 x i8], ptr %base, i64 %next
+ %dst = getelementptr inbounds nuw [8 x i8], ptr %out, i64 %next
+ store ptr %src, ptr %dst, align 8
+ %continue = icmp samesign ugt i64 %iv, 1
+ br i1 %continue, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+attributes #0 = { nounwind "target-cpu"="neoverse-v2" }
diff --git a/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll b/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll
new file mode 100644
index 0000000000000..35246c73b634c
--- /dev/null
+++ b/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll
@@ -0,0 +1,67 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes='separate-const-offset-from-gep<lower-gep>' < %s | FileCheck %s
+
+target triple = "aarch64-linux-gnu"
+
+define void @scalable_gep_common_base(ptr %base, ptr %out0, ptr %out1, i64 %offset, i1 %cond) {
+; CHECK-LABEL: define void @scalable_gep_common_base(
+; CHECK-SAME: ptr [[BASE:%.*]], ptr [[OUT0:%.*]], ptr [[OUT1:%.*]], i64 [[OFFSET:%.*]], i1 [[COND:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[INSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[OFFSET]], i64 0
+; CHECK-NEXT: [[SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[INSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[ENTRY]] ], [ [[NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[NEXT]] = sub <vscale x 2 x i64> [[IV]], [[SPLAT]]
+; CHECK-NEXT: [[VECTOR_GEP_BASE:%.*]] = getelementptr [56 x i8], ptr [[BASE]], <vscale x 2 x i64> [[IV]]
+; CHECK-NEXT: [[GEP0:%.*]] = getelementptr i8, <vscale x 2 x ptr> [[VECTOR_GEP_BASE]], i64 -56
+; CHECK-NEXT: [[VECTOR_GEP_OFFSET:%.*]] = sub i64 -1, [[OFFSET]]
+; CHECK-NEXT: [[VECTOR_GEP_BYTE_OFFSET:%.*]] = mul i64 [[VECTOR_GEP_OFFSET]], 56
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr i8, <vscale x 2 x ptr> [[VECTOR_GEP_BASE]], i64 [[VECTOR_GEP_BYTE_OFFSET]]
+; CHECK-NEXT: store <vscale x 2 x ptr> [[GEP0]], ptr [[OUT0]], align 16
+; CHECK-NEXT: store <vscale x 2 x ptr> [[GEP1]], ptr [[OUT1]], align 16
+; CHECK-NEXT: br i1 [[COND]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %insert = insertelement <vscale x 2 x i64> poison, i64 %offset, i64 0
+ %splat = shufflevector <vscale x 2 x i64> %insert,
+ <vscale x 2 x i64> poison,
+ <vscale x 2 x i32> zeroinitializer
+ br label %loop
+
+loop:
+ %iv = phi <vscale x 2 x i64> [ zeroinitializer, %entry ], [ %next, %loop ]
+ %index0 = add <vscale x 2 x i64> %iv, splat (i64 -1)
+ %next = sub <vscale x 2 x i64> %iv, %splat
+ %index1 = add <vscale x 2 x i64> %next, splat (i64 -1)
+ %gep0 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index0
+ %gep1 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index1
+ store <vscale x 2 x ptr> %gep0, ptr %out0
+ store <vscale x 2 x ptr> %gep1, ptr %out1
+
+ br i1 %cond, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+define void @identical_scalable_geps(ptr %base, ptr %out0, ptr %out1,
+; CHECK-LABEL: define void @identical_scalable_geps(
+; CHECK-SAME: ptr [[BASE:%.*]], ptr [[OUT0:%.*]], ptr [[OUT1:%.*]], <vscale x 2 x i64> [[IV:%.*]]) {
+; CHECK-NEXT: [[INDEX:%.*]] = add <vscale x 2 x i64> [[IV]], splat (i64 -1)
+; CHECK-NEXT: [[GEP0:%.*]] = getelementptr [56 x i8], ptr [[BASE]], <vscale x 2 x i64> [[INDEX]]
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr [56 x i8], ptr [[BASE]], <vscale x 2 x i64> [[INDEX]]
+; CHECK-NEXT: store <vscale x 2 x ptr> [[GEP0]], ptr [[OUT0]], align 16
+; CHECK-NEXT: store <vscale x 2 x ptr> [[GEP1]], ptr [[OUT1]], align 16
+; CHECK-NEXT: ret void
+;
+ <vscale x 2 x i64> %iv) {
+ %index = add <vscale x 2 x i64> %iv, splat (i64 -1)
+ %gep0 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index
+ %gep1 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index
+ store <vscale x 2 x ptr> %gep0, ptr %out0
+ store <vscale x 2 x ptr> %gep1, ptr %out1
+ ret void
+}
>From b6829931cf51a416542be800ea8d0788883a1dd6 Mon Sep 17 00:00:00 2001
From: Jacob Crawley <jacob.crawley at arm.com>
Date: Mon, 10 Aug 2026 14:02:23 +0000
Subject: [PATCH 2/4] Update codegen test
---
llvm/test/CodeGen/AArch64/sve-vector-gep.ll | 87 ++++++++++++++++-----
1 file changed, 67 insertions(+), 20 deletions(-)
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
index 10ff1ed76b7e0..19e1460bcd0a7 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
@@ -1,31 +1,78 @@
-; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --filter "movprfx" --filter "m(ad|la).*z" --filter "(add|sub).*z" --version 6
-; RUN: opt -S -passes=loop-vectorize -force-vector-width=2 -force-vector-interleave=4 -scalable-vectorization=on < %s | llc -O3 -aarch64-enable-gep-opt=true | FileCheck %s
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -O3 -aarch64-enable-gep-opt=true < %s | FileCheck %s
target triple = "aarch64-unknown-linux-gnu"
-define void @scalable_vector_geps(i64 %n, ptr %base, ptr %out) #0 {
+define void @scalable_vector_geps(i64 %n, ptr %base, ptr %out0, ptr %out1, ptr %out2, ptr %out3, i1 %cond) #0 {
; CHECK-LABEL: scalable_vector_geps:
-; CHECK: movprfx z7, z0
-; CHECK: sub z7.d, z7.d, #1 // =0x1
-; CHECK: movprfx z7, z2
-; CHECK: mla z7.d, p0/m, z0.d, z3.d
-; CHECK: sub z0.d, z0.d, z1.d
-; CHECK: movprfx z16, z7
-; CHECK: sub z16.d, z16.d, #56 // =0x38
-; CHECK: add z17.d, z7.d, z4.d
-; CHECK: add z18.d, z7.d, z5.d
-; CHECK: add z7.d, z7.d, z6.d
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: cntd x8
+; CHECK-NEXT: cnth x9
+; CHECK-NEXT: mov z3.d, #56 // =0x38
+; CHECK-NEXT: mov z2.d, x1
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: mov z0.d, x9
+; CHECK-NEXT: mvn x9, x8
+; CHECK-NEXT: index z1.d, x0, #-1
+; CHECK-NEXT: lsl x10, x9, #6
+; CHECK-NEXT: sub x10, x10, x9, lsl #3
+; CHECK-NEXT: sub x9, x9, x8
+; CHECK-NEXT: lsl x11, x9, #6
+; CHECK-NEXT: sub x8, x9, x8
+; CHECK-NEXT: mov z4.d, x10
+; CHECK-NEXT: sub x11, x11, x9, lsl #3
+; CHECK-NEXT: lsl x9, x8, #6
+; CHECK-NEXT: sub x8, x9, x8, lsl #3
+; CHECK-NEXT: mov z5.d, x11
+; CHECK-NEXT: mov z6.d, x8
+; CHECK-NEXT: .p2align 5, , 16
+; CHECK-NEXT: .LBB0_1: // %loop
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: movprfx z7, z2
+; CHECK-NEXT: mla z7.d, p0/m, z1.d, z3.d
+; CHECK-NEXT: sub z1.d, z1.d, z0.d
+; CHECK-NEXT: movprfx z16, z7
+; CHECK-NEXT: sub z16.d, z16.d, #56 // =0x38
+; CHECK-NEXT: add z17.d, z7.d, z4.d
+; CHECK-NEXT: add z18.d, z7.d, z5.d
+; CHECK-NEXT: add z7.d, z7.d, z6.d
+; CHECK-NEXT: str z16, [x2]
+; CHECK-NEXT: str z17, [x3]
+; CHECK-NEXT: str z18, [x4]
+; CHECK-NEXT: str z7, [x5]
+; CHECK-NEXT: tbnz w6, #0, .LBB0_1
+; CHECK-NEXT: // %bb.2: // %exit
+; CHECK-NEXT: ret
entry:
+ %vscale = call i64 @llvm.vscale.i64()
+ %step = shl nuw i64 %vscale, 1
+ %insert.step = insertelement <vscale x 2 x i64> poison, i64 %step, i64 0
+ %splat.step = shufflevector <vscale x 2 x i64> %insert.step, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %insert.n = insertelement <vscale x 2 x i64> poison, i64 %n, i64 0
+ %splat.n = shufflevector <vscale x 2 x i64> %insert.n, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %initial = sub <vscale x 2 x i64> %splat.n, %lane
br label %loop
loop:
- %iv = phi i64 [ %n, %entry ], [ %next, %loop ]
- %next = add nsw i64 %iv, -1
- %src = getelementptr [56 x i8], ptr %base, i64 %next
- %dst = getelementptr inbounds nuw [8 x i8], ptr %out, i64 %next
- store ptr %src, ptr %dst, align 8
- %continue = icmp samesign ugt i64 %iv, 1
- br i1 %continue, label %loop, label %exit
+ %iv0 = phi <vscale x 2 x i64> [ %initial, %entry ], [ %iv4, %loop ]
+ %iv1 = sub <vscale x 2 x i64> %iv0, %splat.step
+ %iv2 = sub <vscale x 2 x i64> %iv1, %splat.step
+ %iv3 = sub <vscale x 2 x i64> %iv2, %splat.step
+ %iv4 = sub <vscale x 2 x i64> %iv3, %splat.step
+ %index0 = add <vscale x 2 x i64> %iv0, splat (i64 -1)
+ %index1 = add <vscale x 2 x i64> %iv1, splat (i64 -1)
+ %index2 = add <vscale x 2 x i64> %iv2, splat (i64 -1)
+ %index3 = add <vscale x 2 x i64> %iv3, splat (i64 -1)
+ %gep0 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index0
+ %gep1 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index1
+ %gep2 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index2
+ %gep3 = getelementptr [56 x i8], ptr %base, <vscale x 2 x i64> %index3
+ store <vscale x 2 x ptr> %gep0, ptr %out0
+ store <vscale x 2 x ptr> %gep1, ptr %out1
+ store <vscale x 2 x ptr> %gep2, ptr %out2
+ store <vscale x 2 x ptr> %gep3, ptr %out3
+ br i1 %cond, label %loop, label %exit
exit:
ret void
>From 47c3298c078a2f6b3f7bea62429406c3179bb816 Mon Sep 17 00:00:00 2001
From: Jacob Crawley <jacob.crawley at arm.com>
Date: Tue, 11 Aug 2026 09:36:43 +0000
Subject: [PATCH 3/4] Apply scalable only pass by default
---
llvm/include/llvm/Transforms/Scalar.h | 9 ++++
.../Target/AArch64/AArch64TargetMachine.cpp | 9 ++++
.../Scalar/SeparateConstOffsetFromGEP.cpp | 48 ++++++++++++++-----
llvm/test/CodeGen/AArch64/sve-vector-gep.ll | 2 +-
4 files changed, 54 insertions(+), 14 deletions(-)
diff --git a/llvm/include/llvm/Transforms/Scalar.h b/llvm/include/llvm/Transforms/Scalar.h
index 24a127b1ffb17..4fc8dbfe4f558 100644
--- a/llvm/include/llvm/Transforms/Scalar.h
+++ b/llvm/include/llvm/Transforms/Scalar.h
@@ -171,6 +171,15 @@ LLVM_ABI FunctionPass *createPartiallyInlineLibCallsPass();
LLVM_ABI FunctionPass *
createSeparateConstOffsetFromGEPPass(bool LowerGEP = false);
+//===----------------------------------------------------------------------===//
+//
+// ShareScalableVectorGEPBase - Share the varying vector
+// base of compatible scalable-vector GEPs without
+// applying the scalar GEP transformations performed by
+// SeparateConstOffsetFromGEP.
+//
+LLVM_ABI FunctionPass *createShareScalableVectorGEPBasePass();
+
//===----------------------------------------------------------------------===//
//
// SpeculativeExecution - Aggressively hoist instructions to enable
diff --git a/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp b/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
index dee61a4f2ac8c..7fe4b177de3af 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
@@ -133,6 +133,11 @@ static cl::opt<bool>
cl::desc("Enable optimizations on complex GEPs"),
cl::init(false));
+static cl::opt<bool> EnableScalableVectorGEPOpt(
+ "aarch64-enable-scalable-vector-gep-opt", cl::Hidden,
+ cl::desc("Share common bases between scalable-vector GEPs"),
+ cl::init(true));
+
static cl::opt<bool>
EnableSelectOpt("aarch64-select-opt", cl::Hidden,
cl::desc("Enable select to branch optimizations"),
@@ -683,6 +688,10 @@ void AArch64PassConfig::addIRPasses() {
// Do loop invariant code motion in case part of the lowered result is
// invariant.
addPass(createLICMPass());
+ } else if (TM->getOptLevel() >= CodeGenOptLevel::Default &&
+ EnableScalableVectorGEPOpt) {
+ addPass(createShareScalableVectorGEPBasePass());
+ addPass(createLICMPass());
}
TargetPassConfig::addIRPasses();
diff --git a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
index 481e57cf55981..2a51b2dcbb7f3 100644
--- a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
+++ b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
@@ -353,12 +353,20 @@ class SeparateConstOffsetFromGEPLegacyPass : public FunctionPass {
public:
static char ID;
- SeparateConstOffsetFromGEPLegacyPass(bool LowerGEP = false)
- : FunctionPass(ID), LowerGEP(LowerGEP) {
+ SeparateConstOffsetFromGEPLegacyPass(
+ bool LowerGEP = false, bool ShareScalableVectorGEPBaseOnly = false)
+ : FunctionPass(ID), LowerGEP(LowerGEP),
+ ShareScalableVectorGEPBaseOnly(ShareScalableVectorGEPBaseOnly) {
initializeSeparateConstOffsetFromGEPLegacyPassPass(
*PassRegistry::getPassRegistry());
}
+ StringRef getPassName() const override {
+ if (ShareScalableVectorGEPBaseOnly)
+ return "Share bases of scalable-vector GEPs";
+ return FunctionPass::getPassName();
+ }
+
void getAnalysisUsage(AnalysisUsage &AU) const override {
AU.addRequired<DominatorTreeWrapperPass>();
AU.addRequired<TargetTransformInfoWrapperPass>();
@@ -371,6 +379,7 @@ class SeparateConstOffsetFromGEPLegacyPass : public FunctionPass {
private:
bool LowerGEP;
+ bool ShareScalableVectorGEPBaseOnly;
};
/// A pass that tries to split every GEP in the function into a variadic
@@ -380,8 +389,10 @@ class SeparateConstOffsetFromGEP {
public:
SeparateConstOffsetFromGEP(
DominatorTree *DT, LoopInfo *LI, TargetLibraryInfo *TLI,
- function_ref<TargetTransformInfo &(Function &)> GetTTI, bool LowerGEP)
- : DT(DT), LI(LI), TLI(TLI), GetTTI(GetTTI), LowerGEP(LowerGEP) {}
+ function_ref<TargetTransformInfo &(Function &)> GetTTI, bool LowerGEP,
+ bool ShareScalableVectorGEPBaseOnly = false)
+ : DT(DT), LI(LI), TLI(TLI), GetTTI(GetTTI), LowerGEP(LowerGEP),
+ ShareScalableVectorGEPBaseOnly(ShareScalableVectorGEPBaseOnly) {}
bool run(Function &F);
@@ -501,6 +512,10 @@ class SeparateConstOffsetFromGEP {
/// multiple GEPs with a single index.
bool LowerGEP;
+ /// Run only scalable-vector base sharing, without splitting or lowering
+ /// scalar GEPs.
+ bool ShareScalableVectorGEPBaseOnly;
+
DenseMap<ExprKey, SmallVector<Instruction *, 2>> DominatingAdds;
DenseMap<ExprKey, SmallVector<Instruction *, 2>> DominatingSubs;
};
@@ -527,6 +542,10 @@ FunctionPass *llvm::createSeparateConstOffsetFromGEPPass(bool LowerGEP) {
return new SeparateConstOffsetFromGEPLegacyPass(LowerGEP);
}
+FunctionPass *llvm::createShareScalableVectorGEPBasePass() {
+ return new SeparateConstOffsetFromGEPLegacyPass(false, true);
+}
+
// Checks if it is safe to reorder an add/sext result used in a GEP.
//
// An inbounds GEP does not guarantee that the index is non-negative.
@@ -1618,7 +1637,8 @@ bool SeparateConstOffsetFromGEPLegacyPass::runOnFunction(Function &F) {
auto GetTTI = [this](Function &F) -> TargetTransformInfo & {
return this->getAnalysis<TargetTransformInfoWrapperPass>().getTTI(F);
};
- SeparateConstOffsetFromGEP Impl(DT, LI, TLI, GetTTI, LowerGEP);
+ SeparateConstOffsetFromGEP Impl(DT, LI, TLI, GetTTI, LowerGEP,
+ ShareScalableVectorGEPBaseOnly);
return Impl.run(F);
}
@@ -1634,17 +1654,19 @@ bool SeparateConstOffsetFromGEP::run(Function &F) {
if (!DT->isReachableFromEntry(B))
continue;
- for (Instruction &I : llvm::make_early_inc_range(*B))
- if (GetElementPtrInst *GEP = dyn_cast<GetElementPtrInst>(&I))
- Changed |= splitGEP(GEP);
- // No need to split GEP ConstantExprs because all its indices are constant
- // already.
-
- if (LowerGEP)
+ if (!ShareScalableVectorGEPBaseOnly) {
+ for (Instruction &I : llvm::make_early_inc_range(*B))
+ if (GetElementPtrInst *GEP = dyn_cast<GetElementPtrInst>(&I))
+ Changed |= splitGEP(GEP);
+ // No need to split GEP ConstantExprs because all its indices are
+ // constant already.
+ }
+ if (LowerGEP || ShareScalableVectorGEPBaseOnly)
Changed |= shareScalableVectorGEPBase(*B);
}
- Changed |= reuniteExts(F);
+ if (!ShareScalableVectorGEPBaseOnly)
+ Changed |= reuniteExts(F);
if (VerifyNoDeadCode)
verifyNoDeadCode(F);
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
index 19e1460bcd0a7..eb7650c913d0c 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -O3 -aarch64-enable-gep-opt=true < %s | FileCheck %s
+; RUN: llc -O3 < %s | FileCheck %s
target triple = "aarch64-unknown-linux-gnu"
>From cded048df0e07af75e7c9d2ea1ca2ea13571d528 Mon Sep 17 00:00:00 2001
From: Jacob Crawley <jacob.crawley at arm.com>
Date: Tue, 11 Aug 2026 11:54:12 +0000
Subject: [PATCH 4/4] rm LICM from pass
---
.../Target/AArch64/AArch64TargetMachine.cpp | 1 -
.../Scalar/SeparateConstOffsetFromGEP.cpp | 16 ++++++++----
llvm/test/CodeGen/AArch64/O3-pipeline.ll | 2 ++
llvm/test/CodeGen/AArch64/sve-vector-gep.ll | 26 ++++++++++---------
.../scalable-vector-gep-common-base.ll | 4 +--
5 files changed, 29 insertions(+), 20 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp b/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
index 7fe4b177de3af..652d17d6256c8 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetMachine.cpp
@@ -691,7 +691,6 @@ void AArch64PassConfig::addIRPasses() {
} else if (TM->getOptLevel() >= CodeGenOptLevel::Default &&
EnableScalableVectorGEPOpt) {
addPass(createShareScalableVectorGEPBasePass());
- addPass(createLICMPass());
}
TargetPassConfig::addIRPasses();
diff --git a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
index 2a51b2dcbb7f3..cb3ff127c355e 100644
--- a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
+++ b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
@@ -1368,6 +1368,10 @@ bool SeparateConstOffsetFromGEP::shareScalableVectorGEPBase(BasicBlock &BB) {
GetElementPtrInst *FirstGEP = Leader.GEP;
IRBuilder<> BaseBuilder(FirstGEP);
BaseBuilder.SetCurrentDebugLocation(FirstGEP->getDebugLoc());
+ Loop *L = LI->getLoopFor(FirstGEP->getParent());
+ assert(L && L->getLoopPreheader() &&
+ "candidate must belong to a loop with a preheader");
+ Instruction *PreheaderTerminator = L->getLoopPreheader()->getTerminator();
// Computing the varying portion once avoids a vector multiply-add for every
// unrolled part. The remaining uniform offsets can be calculated scalarly.
@@ -1379,6 +1383,8 @@ bool SeparateConstOffsetFromGEP::shareScalableVectorGEPBase(BasicBlock &BB) {
VectorGEPCandidate &Current = Candidates[CandidateIndex];
GetElementPtrInst *GEP = Current.GEP;
+ IRBuilder<> OffsetBuilder(PreheaderTerminator);
+ OffsetBuilder.SetCurrentDebugLocation(GEP->getDebugLoc());
IRBuilder<> Builder(GEP);
Builder.SetCurrentDebugLocation(GEP->getDebugLoc());
@@ -1387,13 +1393,13 @@ bool SeparateConstOffsetFromGEP::shareScalableVectorGEPBase(BasicBlock &BB) {
Value *Offset = ConstantInt::get(OffsetType, 0);
for (const VectorGEPOffsetTerm &Term : Current.OffsetTerms) {
- Offset =
- Term.IsSub
- ? Builder.CreateSub(Offset, Term.Scalar, "vector.gep.offset")
- : Builder.CreateAdd(Offset, Term.Scalar, "vector.gep.offset");
+ Offset = Term.IsSub ? OffsetBuilder.CreateSub(Offset, Term.Scalar,
+ "vector.gep.offset")
+ : OffsetBuilder.CreateAdd(Offset, Term.Scalar,
+ "vector.gep.offset");
}
- Value *ByteOffset = Builder.CreateMul(
+ Value *ByteOffset = OffsetBuilder.CreateMul(
Offset, ConstantInt::get(OffsetType, Current.Stride),
"vector.gep.byte.offset");
Value *NewGEP =
diff --git a/llvm/test/CodeGen/AArch64/O3-pipeline.ll b/llvm/test/CodeGen/AArch64/O3-pipeline.ll
index 9635d8bcee02c..5f6c174b31061 100644
--- a/llvm/test/CodeGen/AArch64/O3-pipeline.ll
+++ b/llvm/test/CodeGen/AArch64/O3-pipeline.ll
@@ -39,9 +39,11 @@
; CHECK-NEXT: Scalar Evolution Analysis
; CHECK-NEXT: Loop Data Prefetch
; CHECK-NEXT: Falkor HW Prefetch Fix
+; CHECK-NEXT: Share bases of scalable-vector GEPs
; CHECK-NEXT: Module Verifier
; CHECK-NEXT: Basic Alias Analysis (stateless AA impl)
; CHECK-NEXT: Canonicalize natural loops
+; CHECK-NEXT: Scalar Evolution Analysis
; CHECK-NEXT: Loop Pass Manager
; CHECK-NEXT: Canonicalize Freeze Instructions in Loops
; CHECK-NEXT: Induction Variable Users
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
index eb7650c913d0c..5bf7eb4c834a2 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-gep.ll
@@ -7,14 +7,13 @@ define void @scalable_vector_geps(i64 %n, ptr %base, ptr %out0, ptr %out1, ptr %
; CHECK-LABEL: scalable_vector_geps:
; CHECK: // %bb.0: // %entry
; CHECK-NEXT: cntd x8
-; CHECK-NEXT: cnth x9
+; CHECK-NEXT: index z1.d, x0, #-1
; CHECK-NEXT: mov z3.d, #56 // =0x38
-; CHECK-NEXT: mov z2.d, x1
; CHECK-NEXT: ptrue p0.d
-; CHECK-NEXT: mov z0.d, x9
; CHECK-NEXT: mvn x9, x8
-; CHECK-NEXT: index z1.d, x0, #-1
+; CHECK-NEXT: mov z0.d, x8
; CHECK-NEXT: lsl x10, x9, #6
+; CHECK-NEXT: mov z2.d, x1
; CHECK-NEXT: sub x10, x10, x9, lsl #3
; CHECK-NEXT: sub x9, x9, x8
; CHECK-NEXT: lsl x11, x9, #6
@@ -28,18 +27,21 @@ define void @scalable_vector_geps(i64 %n, ptr %base, ptr %out0, ptr %out1, ptr %
; CHECK-NEXT: .p2align 5, , 16
; CHECK-NEXT: .LBB0_1: // %loop
; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
-; CHECK-NEXT: movprfx z7, z2
-; CHECK-NEXT: mla z7.d, p0/m, z1.d, z3.d
-; CHECK-NEXT: sub z1.d, z1.d, z0.d
-; CHECK-NEXT: movprfx z16, z7
+; CHECK-NEXT: sub z7.d, z1.d, z0.d
+; CHECK-NEXT: mad z1.d, p0/m, z3.d, z2.d
+; CHECK-NEXT: sub z7.d, z7.d, z0.d
+; CHECK-NEXT: sub z7.d, z7.d, z0.d
+; CHECK-NEXT: movprfx z16, z1
; CHECK-NEXT: sub z16.d, z16.d, #56 // =0x38
-; CHECK-NEXT: add z17.d, z7.d, z4.d
-; CHECK-NEXT: add z18.d, z7.d, z5.d
-; CHECK-NEXT: add z7.d, z7.d, z6.d
+; CHECK-NEXT: add z17.d, z1.d, z4.d
+; CHECK-NEXT: add z18.d, z1.d, z5.d
+; CHECK-NEXT: add z1.d, z1.d, z6.d
+; CHECK-NEXT: sub z7.d, z7.d, z0.d
; CHECK-NEXT: str z16, [x2]
; CHECK-NEXT: str z17, [x3]
; CHECK-NEXT: str z18, [x4]
-; CHECK-NEXT: str z7, [x5]
+; CHECK-NEXT: str z1, [x5]
+; CHECK-NEXT: mov z1.d, z7.d
; CHECK-NEXT: tbnz w6, #0, .LBB0_1
; CHECK-NEXT: // %bb.2: // %exit
; CHECK-NEXT: ret
diff --git a/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll b/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll
index 35246c73b634c..5c8aabddcef4b 100644
--- a/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll
+++ b/llvm/test/Transforms/SeparateConstOffsetFromGEP/AArch64/scalable-vector-gep-common-base.ll
@@ -9,14 +9,14 @@ define void @scalable_gep_common_base(ptr %base, ptr %out0, ptr %out1, i64 %offs
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[INSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[OFFSET]], i64 0
; CHECK-NEXT: [[SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[INSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[VECTOR_GEP_OFFSET:%.*]] = sub i64 -1, [[OFFSET]]
+; CHECK-NEXT: [[VECTOR_GEP_BYTE_OFFSET:%.*]] = mul i64 [[VECTOR_GEP_OFFSET]], 56
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, %[[ENTRY]] ], [ [[NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[NEXT]] = sub <vscale x 2 x i64> [[IV]], [[SPLAT]]
; CHECK-NEXT: [[VECTOR_GEP_BASE:%.*]] = getelementptr [56 x i8], ptr [[BASE]], <vscale x 2 x i64> [[IV]]
; CHECK-NEXT: [[GEP0:%.*]] = getelementptr i8, <vscale x 2 x ptr> [[VECTOR_GEP_BASE]], i64 -56
-; CHECK-NEXT: [[VECTOR_GEP_OFFSET:%.*]] = sub i64 -1, [[OFFSET]]
-; CHECK-NEXT: [[VECTOR_GEP_BYTE_OFFSET:%.*]] = mul i64 [[VECTOR_GEP_OFFSET]], 56
; CHECK-NEXT: [[GEP1:%.*]] = getelementptr i8, <vscale x 2 x ptr> [[VECTOR_GEP_BASE]], i64 [[VECTOR_GEP_BYTE_OFFSET]]
; CHECK-NEXT: store <vscale x 2 x ptr> [[GEP0]], ptr [[OUT0]], align 16
; CHECK-NEXT: store <vscale x 2 x ptr> [[GEP1]], ptr [[OUT1]], align 16
More information about the llvm-commits
mailing list