[llvm] [SLP] Avoid seeding related affine loop address computations (PR #226220)

Tim Besard via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 28 06:28:25 PDT 2026


https://github.com/maleadt updated https://github.com/llvm/llvm-project/pull/226220

>From 5b7b9e6594d66bd80affe9e5039db33549cb67a5 Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Thu, 24 Sep 2026 12:55:55 +0200
Subject: [PATCH 1/4] [SLP] Avoid seeding related affine loop address
 computations

When the loop vectorizer scalarizes strided accesses, each lane computes
its own address from the induction variable, a loop-invariant stride
and a loop-invariant offset, e.g.
base + ((off + stride * (iv + k)) << 2). These addresses are affine
recurrences that loop strength reduction turns into pointer increments.
With the loop-aware cost model (#150450), SLP now vectorizes the index
arithmetic instead: the multiplies, adds and shifts are done on
<4 x i64> in the loop, and every lane is extracted again to feed the
scalar loads. The cost model compares this against the scalar index
arithmetic, which LSR removes, so the vector form looks profitable
while the loop gets longer. For strided_gather_offset in the new test,
the loop grows from 13 to 30 instructions on Haswell, and from 13 to 23
on skylake-avx512, which has a vector i64 multiply.

vectorizeGEPIndices already drops pairs of getelementptrs with a
constant difference, because one address can be computed from the
other. Generalize this to loop-invariant differences: skip a bundle
when each lane is a short single-use chain of index arithmetic in a
loop, ending as the index of a getelementptr that is only used by
scalar loads and stores, and the addresses are affine recurrences of
that loop that differ by loop-invariant amounts. Check this in
tryToVectorizeList, because the once-used seeds vectorize the same
arithmetic when only the getelementptr seeds are skipped. Loads that
are vectorized as a gather together with their addresses are not
affected.

Assisted-by: Claude Code, Codex
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    |  71 ++++++++
 .../SLPVectorizer/AArch64/getelementptr.ll    |  27 +--
 .../X86/strength-reducible-address.ll         | 155 ++++++++----------
 3 files changed, 149 insertions(+), 104 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index c9e649c551b92..50db2ba206b87 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -30445,6 +30445,71 @@ void SLPVectorizerPass::collectSeedInstructions(BasicBlock *BB) {
   }
 }
 
+/// Returns true if \p Ptr is only used by scalar loads and stores, directly or
+/// through getelementptrs with constant offsets.
+static bool onlyFeedsScalarAccesses(const Value *Ptr, unsigned Depth = 0) {
+  constexpr unsigned MaxDepth = 2;
+  return !Ptr->use_empty() && all_of(Ptr->users(), [&](const User *U) {
+    if (const auto *Load = dyn_cast<LoadInst>(U))
+      return !Load->getType()->isVectorTy();
+    if (const auto *Store = dyn_cast<StoreInst>(U))
+      return Store->getPointerOperand() == Ptr &&
+             !Store->getValueOperand()->getType()->isVectorTy();
+    if (const auto *GEP = dyn_cast<GetElementPtrInst>(U))
+      return Depth < MaxDepth && GEP->getPointerOperand() == Ptr &&
+             GEP->hasAllConstantIndices() &&
+             onlyFeedsScalarAccesses(GEP, Depth + 1);
+    return false;
+  });
+}
+
+/// Leave related affine address recurrences feeding scalar accesses for loop
+/// strength reduction. Vectorizing their indices can retain expensive
+/// arithmetic and require an extract for each lane instead of a scalar pointer
+/// increment.
+static bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL,
+                                           ScalarEvolution &SE, LoopInfo &LI) {
+  constexpr unsigned MaxIndexChainLength = 3;
+  const Loop *L = nullptr;
+  SmallVector<GetElementPtrInst *> GEPs;
+  for (Value *V : VL) {
+    Value *Cur = V;
+    GetElementPtrInst *GEP = nullptr;
+    for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
+      if (!isa<BinaryOperator, CastInst>(Cur) || !Cur->hasOneUse())
+        return false;
+      User *U = Cur->user_back();
+      if ((GEP = dyn_cast<GetElementPtrInst>(U)))
+        break;
+      Cur = U;
+    }
+    if (!GEP || GEP->getPointerOperand() == Cur ||
+        !onlyFeedsScalarAccesses(GEP))
+      return false;
+    // LSR only removes the arithmetic computed in the loop itself.
+    const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
+    if (!GEPLoop || (L && L != GEPLoop) ||
+        LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
+      return false;
+    L = GEPLoop;
+    GEPs.push_back(GEP);
+  }
+  const SCEV *FirstAddr = nullptr;
+  for (GetElementPtrInst *GEP : GEPs) {
+    const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
+    if (!Addr || Addr->getLoop() != L || !Addr->isAffine())
+      return false;
+    if (!FirstAddr) {
+      FirstAddr = Addr;
+      continue;
+    }
+    const SCEV *Diff = SE.getMinusSCEV(Addr, FirstAddr);
+    if (isa<SCEVCouldNotCompute>(Diff) || !SE.isLoopInvariant(Diff, L))
+      return false;
+  }
+  return true;
+}
+
 bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
                                            bool MaxVFOnly,
                                            bool StandaloneSeeds) {
@@ -30576,6 +30641,12 @@ bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
         continue;
       }
 
+      if (isStrengthReducibleIndexBundle(Ops, *SE, *LI)) {
+        LLVM_DEBUG(dbgs() << "SLP: Not vectorizing strength-reducible address "
+                             "computations.\n");
+        continue;
+      }
+
       LLVM_DEBUG(dbgs() << "SLP: Analyzing " << ActualVF << " operations "
                         << "\n");
 
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
index d7f5ad7861034..ee7b4ab5a411d 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
@@ -8,6 +8,8 @@
 ; getelementptrs when they are known to have a constant difference. Such pairs
 ; are likely not good candidates for vectorization since one can be computed
 ; from the other. We use an unprofitable threshold to force vectorization.
+; The same holds for addresses that are affine recurrences of the loop with
+; loop-invariant differences, as in getelementptr_4x32.
 ;
 ; int getelementptr(int *g, int n, int w, int x, int y, int z) {
 ;   int sum = 0;
@@ -30,25 +32,12 @@
 ; YAML-NEXT:   - String:          ' and with tree size '
 ; YAML-NEXT:   - TreeSize:        '1'
 
-; YAML:      --- !Passed
-; YAML-NEXT: Pass:            slp-vectorizer
-; YAML-NEXT: Name:            VectorizedList
-; YAML-NEXT: Function:        getelementptr_4x32
-; YAML-NEXT: Args:
-; YAML-NEXT:   - String:          'SLP vectorized with cost '
-; YAML-NEXT:   - Cost:            '10'
-; YAML-NEXT:   - String:          ' and with tree size '
-; YAML-NEXT:   - TreeSize:        '3'
-
 define i32 @getelementptr_4x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y, i32 %z) {
 ; CHECK-LABEL: @getelementptr_4x32(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[CMP31:%.*]] = icmp sgt i32 [[N:%.*]], 0
 ; CHECK-NEXT:    br i1 [[CMP31]], label [[FOR_BODY_PREHEADER:%.*]], label [[FOR_COND_CLEANUP:%.*]]
 ; CHECK:       for.body.preheader:
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x i32> <i32 0, i32 poison>, i32 [[X:%.*]], i64 1
-; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <2 x i32> poison, i32 [[Y:%.*]], i64 0
-; CHECK-NEXT:    [[TMP5:%.*]] = insertelement <2 x i32> [[TMP4]], i32 [[Z:%.*]], i64 1
 ; CHECK-NEXT:    br label [[FOR_BODY:%.*]]
 ; CHECK:       for.cond.cleanup.loopexit:
 ; CHECK-NEXT:    br label [[FOR_COND_CLEANUP]]
@@ -63,20 +52,16 @@ define i32 @getelementptr_4x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y
 ; CHECK-NEXT:    [[SUM_32:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[OP_RDX:%.*]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x i32> [ zeroinitializer, [[FOR_BODY_PREHEADER]] ], [ [[SLPRDX_ACC1]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[T4:%.*]] = shl nsw i32 [[SUM_32]], 1
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x i32> poison, i32 [[T4]], i64 0
-; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP3:%.*]] = add nsw <2 x i32> [[TMP2]], [[TMP0]]
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <2 x i32> [[TMP3]], i64 0
+; CHECK-NEXT:    [[TMP12:%.*]] = add nsw i32 [[T4]], 0
 ; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[G:%.*]], i32 [[TMP12]]
 ; CHECK-NEXT:    [[T6:%.*]] = load i32, ptr [[ARRAYIDX]], align 4
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <2 x i32> [[TMP3]], i64 1
+; CHECK-NEXT:    [[TMP11:%.*]] = add nsw i32 [[T4]], [[X:%.*]]
 ; CHECK-NEXT:    [[ARRAYIDX5:%.*]] = getelementptr inbounds i32, ptr [[G]], i32 [[TMP11]]
 ; CHECK-NEXT:    [[T8:%.*]] = load i32, ptr [[ARRAYIDX5]], align 4
-; CHECK-NEXT:    [[TMP16:%.*]] = add nsw <2 x i32> [[TMP2]], [[TMP5]]
-; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <2 x i32> [[TMP16]], i64 0
+; CHECK-NEXT:    [[TMP13:%.*]] = add nsw i32 [[T4]], [[Y:%.*]]
 ; CHECK-NEXT:    [[ARRAYIDX10:%.*]] = getelementptr inbounds i32, ptr [[G]], i32 [[TMP13]]
 ; CHECK-NEXT:    [[T10:%.*]] = load i32, ptr [[ARRAYIDX10]], align 4
-; CHECK-NEXT:    [[TMP14:%.*]] = extractelement <2 x i32> [[TMP16]], i64 1
+; CHECK-NEXT:    [[TMP14:%.*]] = add nsw i32 [[T4]], [[Z:%.*]]
 ; CHECK-NEXT:    [[ARRAYIDX15:%.*]] = getelementptr inbounds i32, ptr [[G]], i32 [[TMP14]]
 ; CHECK-NEXT:    [[T12:%.*]] = load i32, ptr [[ARRAYIDX15]], align 4
 ; CHECK-NEXT:    [[TMP17:%.*]] = insertelement <4 x i32> poison, i32 [[T6]], i64 0
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll b/llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll
index 553b472868edb..86e00db13015a 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll
@@ -12,32 +12,33 @@ define i32 @strided_gather(ptr %base, ptr %dims, i64 %n) {
 ; AVX2-LABEL: define i32 @strided_gather(
 ; AVX2-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
 ; AVX2-NEXT:  [[ENTRY:.*]]:
-; AVX2-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; AVX2-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; AVX2-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; AVX2-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
-; AVX2-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
-; AVX2-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX2-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
-; AVX2-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
 ; AVX2-NEXT:    br label %[[LOOP:.*]]
 ; AVX2:       [[LOOP]]:
 ; AVX2-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
 ; AVX2-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
-; AVX2-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
-; AVX2-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX2-NEXT:    [[TMP6:%.*]] = add <4 x i64> [[TMP5]], <i64 1, i64 2, i64 3, i64 4>
+; AVX2-NEXT:    [[I1:%.*]] = add i64 [[IV]], 1
+; AVX2-NEXT:    [[I2:%.*]] = add i64 [[IV]], 2
+; AVX2-NEXT:    [[I3:%.*]] = add i64 [[IV]], 3
 ; AVX2-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
-; AVX2-NEXT:    [[TMP7:%.*]] = mul <4 x i64> [[TMP3]], [[TMP6]]
-; AVX2-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP1]], [[TMP7]]
-; AVX2-NEXT:    [[TMP9:%.*]] = shl <4 x i64> [[TMP8]], splat (i64 2)
-; AVX2-NEXT:    [[TMP10:%.*]] = extractelement <4 x i64> [[TMP9]], i64 0
-; AVX2-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP10]]
-; AVX2-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 1
-; AVX2-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP11]]
-; AVX2-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[TMP9]], i64 2
-; AVX2-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP12]]
-; AVX2-NEXT:    [[TMP13:%.*]] = extractelement <4 x i64> [[TMP9]], i64 3
-; AVX2-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP13]]
+; AVX2-NEXT:    [[M0:%.*]] = mul i64 [[STRIDE]], [[I1]]
+; AVX2-NEXT:    [[M1:%.*]] = mul i64 [[STRIDE]], [[I2]]
+; AVX2-NEXT:    [[M2:%.*]] = mul i64 [[STRIDE]], [[I3]]
+; AVX2-NEXT:    [[M3:%.*]] = mul i64 [[STRIDE]], [[IV_NEXT]]
+; AVX2-NEXT:    [[A0:%.*]] = add i64 [[OFF]], [[M0]]
+; AVX2-NEXT:    [[A1:%.*]] = add i64 [[OFF]], [[M1]]
+; AVX2-NEXT:    [[A2:%.*]] = add i64 [[OFF]], [[M2]]
+; AVX2-NEXT:    [[A3:%.*]] = add i64 [[OFF]], [[M3]]
+; AVX2-NEXT:    [[S0:%.*]] = shl i64 [[A0]], 2
+; AVX2-NEXT:    [[S1:%.*]] = shl i64 [[A1]], 2
+; AVX2-NEXT:    [[S2:%.*]] = shl i64 [[A2]], 2
+; AVX2-NEXT:    [[S3:%.*]] = shl i64 [[A3]], 2
+; AVX2-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S0]]
+; AVX2-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S1]]
+; AVX2-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S2]]
+; AVX2-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S3]]
 ; AVX2-NEXT:    [[L0:%.*]] = load i32, ptr [[P0]], align 4
 ; AVX2-NEXT:    [[L1:%.*]] = load i32, ptr [[P1]], align 4
 ; AVX2-NEXT:    [[L2:%.*]] = load i32, ptr [[P2]], align 4
@@ -139,32 +140,33 @@ define i32 @strided_gather_offset(ptr %base, ptr %dims, i64 %n) {
 ; CHECK-LABEL: define i32 @strided_gather_offset(
 ; CHECK-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
-; CHECK-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; CHECK-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; CHECK-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; CHECK-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
-; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP6:%.*]] = add <4 x i64> [[TMP5]], <i64 1, i64 2, i64 3, i64 4>
+; CHECK-NEXT:    [[I1:%.*]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[I2:%.*]] = add i64 [[IV]], 2
+; CHECK-NEXT:    [[I3:%.*]] = add i64 [[IV]], 3
 ; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
-; CHECK-NEXT:    [[TMP7:%.*]] = mul <4 x i64> [[TMP3]], [[TMP6]]
-; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP1]], [[TMP7]]
-; CHECK-NEXT:    [[TMP9:%.*]] = shl <4 x i64> [[TMP8]], splat (i64 2)
-; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <4 x i64> [[TMP9]], i64 0
-; CHECK-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP10]]
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 1
-; CHECK-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP11]]
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[TMP9]], i64 2
-; CHECK-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP12]]
-; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x i64> [[TMP9]], i64 3
-; CHECK-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP13]]
+; CHECK-NEXT:    [[M0:%.*]] = mul i64 [[STRIDE]], [[I1]]
+; CHECK-NEXT:    [[M1:%.*]] = mul i64 [[STRIDE]], [[I2]]
+; CHECK-NEXT:    [[M2:%.*]] = mul i64 [[STRIDE]], [[I3]]
+; CHECK-NEXT:    [[M3:%.*]] = mul i64 [[STRIDE]], [[IV_NEXT]]
+; CHECK-NEXT:    [[A0:%.*]] = add i64 [[OFF]], [[M0]]
+; CHECK-NEXT:    [[A1:%.*]] = add i64 [[OFF]], [[M1]]
+; CHECK-NEXT:    [[A2:%.*]] = add i64 [[OFF]], [[M2]]
+; CHECK-NEXT:    [[A3:%.*]] = add i64 [[OFF]], [[M3]]
+; CHECK-NEXT:    [[S0:%.*]] = shl i64 [[A0]], 2
+; CHECK-NEXT:    [[S1:%.*]] = shl i64 [[A1]], 2
+; CHECK-NEXT:    [[S2:%.*]] = shl i64 [[A2]], 2
+; CHECK-NEXT:    [[S3:%.*]] = shl i64 [[A3]], 2
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S0]]
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S1]]
+; CHECK-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S2]]
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S3]]
 ; CHECK-NEXT:    [[Q0:%.*]] = getelementptr i8, ptr [[P0]], i64 -4
 ; CHECK-NEXT:    [[L0:%.*]] = load i32, ptr [[Q0]], align 4
 ; CHECK-NEXT:    [[Q1:%.*]] = getelementptr i8, ptr [[P1]], i64 -4
@@ -240,31 +242,32 @@ define void @strided_update(ptr %base, ptr %dims, i64 %n) {
 ; CHECK-LABEL: define void @strided_update(
 ; CHECK-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
-; CHECK-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; CHECK-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; CHECK-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; CHECK-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
-; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP6:%.*]] = add <4 x i64> [[TMP5]], <i64 1, i64 2, i64 3, i64 4>
+; CHECK-NEXT:    [[I1:%.*]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[I2:%.*]] = add i64 [[IV]], 2
+; CHECK-NEXT:    [[I3:%.*]] = add i64 [[IV]], 3
 ; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
-; CHECK-NEXT:    [[TMP7:%.*]] = mul <4 x i64> [[TMP3]], [[TMP6]]
-; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP1]], [[TMP7]]
-; CHECK-NEXT:    [[TMP9:%.*]] = shl <4 x i64> [[TMP8]], splat (i64 2)
-; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <4 x i64> [[TMP9]], i64 0
-; CHECK-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP10]]
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 1
-; CHECK-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP11]]
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[TMP9]], i64 2
-; CHECK-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP12]]
-; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x i64> [[TMP9]], i64 3
-; CHECK-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP13]]
+; CHECK-NEXT:    [[M0:%.*]] = mul i64 [[STRIDE]], [[I1]]
+; CHECK-NEXT:    [[M1:%.*]] = mul i64 [[STRIDE]], [[I2]]
+; CHECK-NEXT:    [[M2:%.*]] = mul i64 [[STRIDE]], [[I3]]
+; CHECK-NEXT:    [[M3:%.*]] = mul i64 [[STRIDE]], [[IV_NEXT]]
+; CHECK-NEXT:    [[A0:%.*]] = add i64 [[OFF]], [[M0]]
+; CHECK-NEXT:    [[A1:%.*]] = add i64 [[OFF]], [[M1]]
+; CHECK-NEXT:    [[A2:%.*]] = add i64 [[OFF]], [[M2]]
+; CHECK-NEXT:    [[A3:%.*]] = add i64 [[OFF]], [[M3]]
+; CHECK-NEXT:    [[S0:%.*]] = shl i64 [[A0]], 2
+; CHECK-NEXT:    [[S1:%.*]] = shl i64 [[A1]], 2
+; CHECK-NEXT:    [[S2:%.*]] = shl i64 [[A2]], 2
+; CHECK-NEXT:    [[S3:%.*]] = shl i64 [[A3]], 2
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S0]]
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S1]]
+; CHECK-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S2]]
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S3]]
 ; CHECK-NEXT:    [[L0:%.*]] = load i32, ptr [[P0]], align 4
 ; CHECK-NEXT:    [[U0:%.*]] = add i32 [[L0]], 1
 ; CHECK-NEXT:    store i32 [[U0]], ptr [[P0]], align 4
@@ -560,37 +563,23 @@ define i32 @preheader_offsets(ptr %base, ptr %offs, i64 %n) {
 ; AVX2-LABEL: define i32 @preheader_offsets(
 ; AVX2-SAME: ptr [[BASE:%.*]], ptr [[OFFS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
 ; AVX2-NEXT:  [[ENTRY:.*]]:
-; AVX2-NEXT:    [[X1_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 8
-; AVX2-NEXT:    [[X2_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 16
-; AVX2-NEXT:    [[X3_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 24
-; AVX2-NEXT:    [[X0:%.*]] = load i64, ptr [[OFFS]], align 8
-; AVX2-NEXT:    [[X1:%.*]] = load i64, ptr [[X1_PTR]], align 8
-; AVX2-NEXT:    [[X2:%.*]] = load i64, ptr [[X2_PTR]], align 8
-; AVX2-NEXT:    [[X3:%.*]] = load i64, ptr [[X3_PTR]], align 8
 ; AVX2-NEXT:    [[Y0_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 32
-; AVX2-NEXT:    [[Y1_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 40
-; AVX2-NEXT:    [[Y2_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 48
-; AVX2-NEXT:    [[Y3_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 56
-; AVX2-NEXT:    [[Y0:%.*]] = load i64, ptr [[Y0_PTR]], align 8
-; AVX2-NEXT:    [[Y1:%.*]] = load i64, ptr [[Y1_PTR]], align 8
-; AVX2-NEXT:    [[Y2:%.*]] = load i64, ptr [[Y2_PTR]], align 8
-; AVX2-NEXT:    [[Y3:%.*]] = load i64, ptr [[Y3_PTR]], align 8
-; AVX2-NEXT:    [[T0:%.*]] = add i64 [[X0]], [[Y0]]
-; AVX2-NEXT:    [[T1:%.*]] = add i64 [[X1]], [[Y1]]
-; AVX2-NEXT:    [[T2:%.*]] = add i64 [[X2]], [[Y2]]
-; AVX2-NEXT:    [[T3:%.*]] = add i64 [[X3]], [[Y3]]
-; AVX2-NEXT:    [[O0:%.*]] = shl i64 [[T0]], 2
-; AVX2-NEXT:    [[O1:%.*]] = shl i64 [[T1]], 2
-; AVX2-NEXT:    [[O2:%.*]] = shl i64 [[T2]], 2
-; AVX2-NEXT:    [[O3:%.*]] = shl i64 [[T3]], 2
+; AVX2-NEXT:    [[TMP0:%.*]] = load <4 x i64>, ptr [[OFFS]], align 8
+; AVX2-NEXT:    [[TMP1:%.*]] = load <4 x i64>, ptr [[Y0_PTR]], align 8
+; AVX2-NEXT:    [[TMP2:%.*]] = add <4 x i64> [[TMP0]], [[TMP1]]
+; AVX2-NEXT:    [[TMP3:%.*]] = shl <4 x i64> [[TMP2]], splat (i64 2)
+; AVX2-NEXT:    [[TMP4:%.*]] = extractelement <4 x i64> [[TMP3]], i64 0
+; AVX2-NEXT:    [[TMP5:%.*]] = extractelement <4 x i64> [[TMP3]], i64 1
+; AVX2-NEXT:    [[TMP6:%.*]] = extractelement <4 x i64> [[TMP3]], i64 2
+; AVX2-NEXT:    [[TMP7:%.*]] = extractelement <4 x i64> [[TMP3]], i64 3
 ; AVX2-NEXT:    br label %[[LOOP:.*]]
 ; AVX2:       [[LOOP]]:
 ; AVX2-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
 ; AVX2-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
-; AVX2-NEXT:    [[A0:%.*]] = add i64 [[IV]], [[O0]]
-; AVX2-NEXT:    [[A1:%.*]] = add i64 [[IV]], [[O1]]
-; AVX2-NEXT:    [[A2:%.*]] = add i64 [[IV]], [[O2]]
-; AVX2-NEXT:    [[A3:%.*]] = add i64 [[IV]], [[O3]]
+; AVX2-NEXT:    [[A0:%.*]] = add i64 [[IV]], [[TMP4]]
+; AVX2-NEXT:    [[A1:%.*]] = add i64 [[IV]], [[TMP5]]
+; AVX2-NEXT:    [[A2:%.*]] = add i64 [[IV]], [[TMP6]]
+; AVX2-NEXT:    [[A3:%.*]] = add i64 [[IV]], [[TMP7]]
 ; AVX2-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[A0]]
 ; AVX2-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[A1]]
 ; AVX2-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[A2]]

>From b79aa848b8d0919ec8127056623a85a3107a5daf Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Thu, 24 Sep 2026 20:40:14 +0200
Subject: [PATCH 2/4] Address review comments

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 86 ++++---------------
 .../Vectorize/SLPVectorizer/SLPUtils.cpp      | 86 +++++++++++++++++++
 .../Vectorize/SLPVectorizer/SLPUtils.h        | 17 ++++
 3 files changed, 121 insertions(+), 68 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 50db2ba206b87..bd77f9277c0c7 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -29324,6 +29324,9 @@ bool SLPVectorizerPass::runImpl(Function &F, ScalarEvolution *SE_,
           R.isScalarFallbackBlock(BB))
         continue;
       R.clearReductionData();
+      // GEPs is collected per block and is also used to keep its index
+      // computations out of the standalone-seed attempt.
+      collectSeedInstructions(BB);
       Changed |= vectorizeOnceUsedSeeds(BB, R);
     }
   }
@@ -30445,71 +30448,6 @@ void SLPVectorizerPass::collectSeedInstructions(BasicBlock *BB) {
   }
 }
 
-/// Returns true if \p Ptr is only used by scalar loads and stores, directly or
-/// through getelementptrs with constant offsets.
-static bool onlyFeedsScalarAccesses(const Value *Ptr, unsigned Depth = 0) {
-  constexpr unsigned MaxDepth = 2;
-  return !Ptr->use_empty() && all_of(Ptr->users(), [&](const User *U) {
-    if (const auto *Load = dyn_cast<LoadInst>(U))
-      return !Load->getType()->isVectorTy();
-    if (const auto *Store = dyn_cast<StoreInst>(U))
-      return Store->getPointerOperand() == Ptr &&
-             !Store->getValueOperand()->getType()->isVectorTy();
-    if (const auto *GEP = dyn_cast<GetElementPtrInst>(U))
-      return Depth < MaxDepth && GEP->getPointerOperand() == Ptr &&
-             GEP->hasAllConstantIndices() &&
-             onlyFeedsScalarAccesses(GEP, Depth + 1);
-    return false;
-  });
-}
-
-/// Leave related affine address recurrences feeding scalar accesses for loop
-/// strength reduction. Vectorizing their indices can retain expensive
-/// arithmetic and require an extract for each lane instead of a scalar pointer
-/// increment.
-static bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL,
-                                           ScalarEvolution &SE, LoopInfo &LI) {
-  constexpr unsigned MaxIndexChainLength = 3;
-  const Loop *L = nullptr;
-  SmallVector<GetElementPtrInst *> GEPs;
-  for (Value *V : VL) {
-    Value *Cur = V;
-    GetElementPtrInst *GEP = nullptr;
-    for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
-      if (!isa<BinaryOperator, CastInst>(Cur) || !Cur->hasOneUse())
-        return false;
-      User *U = Cur->user_back();
-      if ((GEP = dyn_cast<GetElementPtrInst>(U)))
-        break;
-      Cur = U;
-    }
-    if (!GEP || GEP->getPointerOperand() == Cur ||
-        !onlyFeedsScalarAccesses(GEP))
-      return false;
-    // LSR only removes the arithmetic computed in the loop itself.
-    const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
-    if (!GEPLoop || (L && L != GEPLoop) ||
-        LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
-      return false;
-    L = GEPLoop;
-    GEPs.push_back(GEP);
-  }
-  const SCEV *FirstAddr = nullptr;
-  for (GetElementPtrInst *GEP : GEPs) {
-    const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
-    if (!Addr || Addr->getLoop() != L || !Addr->isAffine())
-      return false;
-    if (!FirstAddr) {
-      FirstAddr = Addr;
-      continue;
-    }
-    const SCEV *Diff = SE.getMinusSCEV(Addr, FirstAddr);
-    if (isa<SCEVCouldNotCompute>(Diff) || !SE.isLoopInvariant(Diff, L))
-      return false;
-  }
-  return true;
-}
-
 bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
                                            bool MaxVFOnly,
                                            bool StandaloneSeeds) {
@@ -30641,9 +30579,15 @@ bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
         continue;
       }
 
-      if (isStrengthReducibleIndexBundle(Ops, *SE, *LI)) {
-        LLVM_DEBUG(dbgs() << "SLP: Not vectorizing strength-reducible address "
-                             "computations.\n");
+      auto IsCollectedGEP = [&](GetElementPtrInst *GEP) {
+        auto It = GEPs.find(GEP->getPointerOperand());
+        return It != GEPs.end() && It->second.size() >= 2 &&
+               is_contained(It->second, GEP);
+      };
+      if (StandaloneSeeds &&
+          isGEPCandidateIndexBundle(Ops, *LI, SLPReVec, IsCollectedGEP)) {
+        LLVM_DEBUG(dbgs() << "SLP: Leaving collected GEP index computations "
+                             "to vectorizeGEPIndices.\n");
         continue;
       }
 
@@ -36480,6 +36424,12 @@ bool SLPVectorizerPass::vectorizeGEPIndices(BasicBlock *BB, BoUpSLP &R) {
         Bundle[BundleIndex++] = GEPIdx;
       }
 
+      if (isStrengthReducibleIndexBundle(Bundle, *SE, *LI, SLPReVec)) {
+        LLVM_DEBUG(dbgs() << "SLP: Not vectorizing strength-reducible address "
+                             "computations.\n");
+        continue;
+      }
+
       // Try and vectorize the indices. We are currently only interested in
       // gather-like cases of the form:
       //
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index d639e0d9c7d63..e3c4f4e2dbfd9 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -7,11 +7,15 @@
 //===----------------------------------------------------------------------===//
 
 #include "SLPUtils.h"
+#include "SLPTypeUtils.h"
 
 #include "llvm/ADT/APInt.h"
 #include "llvm/ADT/STLExtras.h"
 #include "llvm/ADT/Sequence.h"
 #include "llvm/Analysis/AssumptionCache.h"
+#include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/ScalarEvolution.h"
+#include "llvm/Analysis/ScalarEvolutionExpressions.h"
 #include "llvm/Analysis/ValueTracking.h"
 #include "llvm/Analysis/VectorUtils.h"
 #include "llvm/IR/Constants.h"
@@ -1003,6 +1007,88 @@ bool isOnceUsedSeed(const Instruction *I) {
       I);
 }
 
+/// Returns true if \p Ptr is only used by scalar loads and stores, directly or
+/// through getelementptrs with constant offsets.
+static bool onlyFeedsScalarAccesses(Value *Ptr, bool ReVec,
+                                    unsigned Depth = 0) {
+  constexpr unsigned MaxDepth = 2;
+  if (Ptr->use_empty() || Ptr->hasNUsesOrMore(UsesLimit))
+    return false;
+  return all_of(Ptr->users(), [&](User *U) {
+    if (isa<LoadInst>(U))
+      return !getValueType(U, ReVec)->isVectorTy();
+    if (auto *Store = dyn_cast<StoreInst>(U))
+      return Store->getPointerOperand() == Ptr &&
+             !getValueType(Store, ReVec)->isVectorTy();
+    if (auto *GEP = dyn_cast<GetElementPtrInst>(U))
+      return Depth < MaxDepth && GEP->getPointerOperand() == Ptr &&
+             GEP->hasAllConstantIndices() &&
+             onlyFeedsScalarAccesses(GEP, ReVec, Depth + 1);
+    return false;
+  });
+}
+
+static bool collectScalarAccessGEPs(ArrayRef<Value *> VL, LoopInfo &LI,
+                                    bool ReVec,
+                                    SmallVectorImpl<GetElementPtrInst *> &GEPs,
+                                    const Loop *&L) {
+  constexpr unsigned MaxIndexChainLength = 3;
+  for (Value *V : VL) {
+    Value *Cur = V;
+    GetElementPtrInst *GEP = nullptr;
+    for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
+      if (!isa<BinaryOperator, CastInst>(Cur) || !Cur->hasOneUse())
+        return false;
+      User *U = Cur->user_back();
+      if ((GEP = dyn_cast<GetElementPtrInst>(U)))
+        break;
+      Cur = U;
+    }
+    if (!GEP || GEP->getPointerOperand() == Cur ||
+        !onlyFeedsScalarAccesses(GEP, ReVec))
+      return false;
+    // LSR only removes the arithmetic computed in the loop itself.
+    const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
+    if (!GEPLoop || (L && L != GEPLoop) ||
+        LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
+      return false;
+    L = GEPLoop;
+    GEPs.push_back(GEP);
+  }
+  return true;
+}
+
+bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
+                                    LoopInfo &LI, bool ReVec) {
+  const Loop *L = nullptr;
+  SmallVector<GetElementPtrInst *> GEPs;
+  if (!collectScalarAccessGEPs(VL, LI, ReVec, GEPs, L))
+    return false;
+  const SCEV *FirstAddr = nullptr;
+  for (GetElementPtrInst *GEP : GEPs) {
+    const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
+    if (!Addr || Addr->getLoop() != L || !Addr->isAffine())
+      return false;
+    if (!FirstAddr) {
+      FirstAddr = Addr;
+      continue;
+    }
+    const SCEV *Diff = SE.getMinusSCEV(Addr, FirstAddr);
+    if (isa<SCEVCouldNotCompute>(Diff) || !SE.isLoopInvariant(Diff, L))
+      return false;
+  }
+  return true;
+}
+
+bool isGEPCandidateIndexBundle(
+    ArrayRef<Value *> VL, LoopInfo &LI, bool ReVec,
+    function_ref<bool(GetElementPtrInst *)> IsCandidate) {
+  const Loop *L = nullptr;
+  SmallVector<GetElementPtrInst *> GEPs;
+  return collectScalarAccessGEPs(VL, LI, ReVec, GEPs, L) &&
+         all_of(GEPs, IsCandidate);
+}
+
 Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable) {
   auto *Wide = dyn_cast<FPExtInst>(V);
   if (!Wide || !Wide->hasOneUse())
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index 4ac6880d0f17d..06e38eec92d0d 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -36,8 +36,11 @@ namespace llvm {
 class AssumptionCache;
 class Constant;
 class DataLayout;
+class GetElementPtrInst;
 class Instruction;
 class IRBuilderBase;
+class LoopInfo;
+class ScalarEvolution;
 class TargetLibraryInfo;
 class Type;
 class Value;
@@ -379,6 +382,20 @@ Intrinsic::ID getMaskedDivRemIntrinsic(unsigned Opcode);
 /// dedicated attempt.
 bool isOnceUsedSeed(const Instruction *I);
 
+/// Leave related affine address recurrences feeding scalar accesses for loop
+/// strength reduction. Vectorizing their indices can retain expensive
+/// arithmetic and require an extract for each lane instead of a scalar pointer
+/// increment.
+bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
+                                    LoopInfo &LI, bool ReVec);
+
+/// Returns true if each value in \p VL is an in-loop index computation ending
+/// at a getelementptr accepted by \p IsCandidate and used only by scalar
+/// accesses.
+bool isGEPCandidateIndexBundle(
+    ArrayRef<Value *> VL, LoopInfo &LI, bool ReVec,
+    function_ref<bool(GetElementPtrInst *)> IsCandidate);
+
 /// If \p V is a single-use fpext of a single-use fptrunc forming a round-trip
 /// back to the type of \p V, returns the fptrunc; the round-trip source is its
 /// operand, always an instruction of the same type as \p V. If

>From e180b116fc14370d1a859df6d216bc93cd466b87 Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Fri, 25 Sep 2026 09:25:15 +0200
Subject: [PATCH 3/4] Address review comments (round 2)

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 19 +++----
 .../Vectorize/SLPVectorizer/SLPUtils.cpp      | 57 ++++++++++++-------
 .../Vectorize/SLPVectorizer/SLPUtils.h        |  8 +--
 3 files changed, 46 insertions(+), 38 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index bd77f9277c0c7..43a4363393f8b 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -30579,18 +30579,6 @@ bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
         continue;
       }
 
-      auto IsCollectedGEP = [&](GetElementPtrInst *GEP) {
-        auto It = GEPs.find(GEP->getPointerOperand());
-        return It != GEPs.end() && It->second.size() >= 2 &&
-               is_contained(It->second, GEP);
-      };
-      if (StandaloneSeeds &&
-          isGEPCandidateIndexBundle(Ops, *LI, SLPReVec, IsCollectedGEP)) {
-        LLVM_DEBUG(dbgs() << "SLP: Leaving collected GEP index computations "
-                             "to vectorizeGEPIndices.\n");
-        continue;
-      }
-
       LLVM_DEBUG(dbgs() << "SLP: Analyzing " << ActualVF << " operations "
                         << "\n");
 
@@ -36306,6 +36294,13 @@ bool SLPVectorizerPass::vectorizeOnceUsedSeeds(BasicBlock *BB, BoUpSLP &R) {
         !isOnceUsedSeed(&I) || isNonVectorizableInst(&I, TLI) ||
         R.hasResolvedUser(&I))
       continue;
+    // Index chains of collected GEPs are handled by vectorizeGEPIndices.
+    if (isGEPCandidateIndexBundle({&I}, *SE, *LI, SLPReVec, [&](auto *GEP) {
+          auto It = GEPs.find(GEP->getPointerOperand());
+          return It != GEPs.end() && It->second.size() >= 2 &&
+                 is_contained(It->second, GEP);
+        }))
+      continue;
     // The poor-throughput ops are seeded on their own, with the different
     // grouping.
     if (VectorizePoorThroughput &&
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index e3c4f4e2dbfd9..7c552d35e6497 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -1015,11 +1015,12 @@ static bool onlyFeedsScalarAccesses(Value *Ptr, bool ReVec,
   if (Ptr->use_empty() || Ptr->hasNUsesOrMore(UsesLimit))
     return false;
   return all_of(Ptr->users(), [&](User *U) {
-    if (isa<LoadInst>(U))
-      return !getValueType(U, ReVec)->isVectorTy();
-    if (auto *Store = dyn_cast<StoreInst>(U))
-      return Store->getPointerOperand() == Ptr &&
-             !getValueType(Store, ReVec)->isVectorTy();
+    if (getValueType(U, ReVec)->isVectorTy())
+      return false;
+    Value *Op;
+    if (isa<LoadInst>(U) ||
+        (match(U, m_Store(m_Value(Op), m_Specific(Ptr))) && Op != Ptr))
+      return true;
     if (auto *GEP = dyn_cast<GetElementPtrInst>(U))
       return Depth < MaxDepth && GEP->getPointerOperand() == Ptr &&
              GEP->hasAllConstantIndices() &&
@@ -1028,17 +1029,20 @@ static bool onlyFeedsScalarAccesses(Value *Ptr, bool ReVec,
   });
 }
 
-static bool collectScalarAccessGEPs(ArrayRef<Value *> VL, LoopInfo &LI,
-                                    bool ReVec,
-                                    SmallVectorImpl<GetElementPtrInst *> &GEPs,
-                                    const Loop *&L) {
+/// Collects in \p GEPs the getelementptrs that the in-loop index computations
+/// \p VL end at, if they are only used by scalar accesses. Returns the loop
+/// containing all of them, or nullptr otherwise.
+static const Loop *
+collectScalarAccessGEPs(ArrayRef<Value *> VL, const LoopInfo &LI, bool ReVec,
+                        SmallVectorImpl<GetElementPtrInst *> &GEPs) {
   constexpr unsigned MaxIndexChainLength = 3;
+  const Loop *L = nullptr;
   for (Value *V : VL) {
     Value *Cur = V;
     GetElementPtrInst *GEP = nullptr;
     for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
       if (!isa<BinaryOperator, CastInst>(Cur) || !Cur->hasOneUse())
-        return false;
+        return nullptr;
       User *U = Cur->user_back();
       if ((GEP = dyn_cast<GetElementPtrInst>(U)))
         break;
@@ -1046,28 +1050,36 @@ static bool collectScalarAccessGEPs(ArrayRef<Value *> VL, LoopInfo &LI,
     }
     if (!GEP || GEP->getPointerOperand() == Cur ||
         !onlyFeedsScalarAccesses(GEP, ReVec))
-      return false;
+      return nullptr;
     // LSR only removes the arithmetic computed in the loop itself.
     const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
     if (!GEPLoop || (L && L != GEPLoop) ||
         LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
-      return false;
+      return nullptr;
     L = GEPLoop;
     GEPs.push_back(GEP);
   }
-  return true;
+  return L;
+}
+
+/// Returns the address computed by \p GEP if it is an affine recurrence of
+/// \p L, or nullptr otherwise.
+static const SCEVAddRecExpr *
+getAffineAddress(GetElementPtrInst *GEP, const Loop *L, ScalarEvolution &SE) {
+  const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
+  return Addr && Addr->getLoop() == L && Addr->isAffine() ? Addr : nullptr;
 }
 
 bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
-                                    LoopInfo &LI, bool ReVec) {
-  const Loop *L = nullptr;
+                                    const LoopInfo &LI, bool ReVec) {
   SmallVector<GetElementPtrInst *> GEPs;
-  if (!collectScalarAccessGEPs(VL, LI, ReVec, GEPs, L))
+  const Loop *L = collectScalarAccessGEPs(VL, LI, ReVec, GEPs);
+  if (!L)
     return false;
   const SCEV *FirstAddr = nullptr;
   for (GetElementPtrInst *GEP : GEPs) {
-    const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
-    if (!Addr || Addr->getLoop() != L || !Addr->isAffine())
+    const SCEV *Addr = getAffineAddress(GEP, L, SE);
+    if (!Addr)
       return false;
     if (!FirstAddr) {
       FirstAddr = Addr;
@@ -1081,12 +1093,13 @@ bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
 }
 
 bool isGEPCandidateIndexBundle(
-    ArrayRef<Value *> VL, LoopInfo &LI, bool ReVec,
+    ArrayRef<Value *> VL, ScalarEvolution &SE, const LoopInfo &LI, bool ReVec,
     function_ref<bool(GetElementPtrInst *)> IsCandidate) {
-  const Loop *L = nullptr;
   SmallVector<GetElementPtrInst *> GEPs;
-  return collectScalarAccessGEPs(VL, LI, ReVec, GEPs, L) &&
-         all_of(GEPs, IsCandidate);
+  const Loop *L = collectScalarAccessGEPs(VL, LI, ReVec, GEPs);
+  return L && all_of(GEPs, [&](GetElementPtrInst *GEP) {
+           return IsCandidate(GEP) && getAffineAddress(GEP, L, SE);
+         });
 }
 
 Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable) {
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index 06e38eec92d0d..f578230bfe292 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -387,13 +387,13 @@ bool isOnceUsedSeed(const Instruction *I);
 /// arithmetic and require an extract for each lane instead of a scalar pointer
 /// increment.
 bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
-                                    LoopInfo &LI, bool ReVec);
+                                    const LoopInfo &LI, bool ReVec);
 
 /// Returns true if each value in \p VL is an in-loop index computation ending
-/// at a getelementptr accepted by \p IsCandidate and used only by scalar
-/// accesses.
+/// at a getelementptr accepted by \p IsCandidate, used only by scalar accesses
+/// and computing an affine recurrence of that loop.
 bool isGEPCandidateIndexBundle(
-    ArrayRef<Value *> VL, LoopInfo &LI, bool ReVec,
+    ArrayRef<Value *> VL, ScalarEvolution &SE, const LoopInfo &LI, bool ReVec,
     function_ref<bool(GetElementPtrInst *)> IsCandidate);
 
 /// If \p V is a single-use fpext of a single-use fptrunc forming a round-trip

>From 73269a5d8f76dd763fe3c2d27a540dd3964afbd8 Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Mon, 28 Sep 2026 09:38:22 +0200
Subject: [PATCH 4/4] Address review comments (round 3)

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    |  2 +-
 .../Vectorize/SLPVectorizer/SLPUtils.cpp      | 87 +++++++------------
 .../Vectorize/SLPVectorizer/SLPUtils.h        | 10 +--
 3 files changed, 36 insertions(+), 63 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 43a4363393f8b..3cc34c7ed3f3a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -36295,7 +36295,7 @@ bool SLPVectorizerPass::vectorizeOnceUsedSeeds(BasicBlock *BB, BoUpSLP &R) {
         R.hasResolvedUser(&I))
       continue;
     // Index chains of collected GEPs are handled by vectorizeGEPIndices.
-    if (isGEPCandidateIndexBundle({&I}, *SE, *LI, SLPReVec, [&](auto *GEP) {
+    if (!GEPs.empty() && isGEPCandidateIndex(&I, [&](auto *GEP) {
           auto It = GEPs.find(GEP->getPointerOperand());
           return It != GEPs.end() && It->second.size() >= 2 &&
                  is_contained(It->second, GEP);
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index 7c552d35e6497..a74ae8c113d95 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -1008,10 +1008,9 @@ bool isOnceUsedSeed(const Instruction *I) {
 }
 
 /// Returns true if \p Ptr is only used by scalar loads and stores, directly or
-/// through getelementptrs with constant offsets.
+/// through at most \p Depth levels of getelementptrs with constant offsets.
 static bool onlyFeedsScalarAccesses(Value *Ptr, bool ReVec,
-                                    unsigned Depth = 0) {
-  constexpr unsigned MaxDepth = 2;
+                                    unsigned Depth = 2) {
   if (Ptr->use_empty() || Ptr->hasNUsesOrMore(UsesLimit))
     return false;
   return all_of(Ptr->users(), [&](User *U) {
@@ -1022,64 +1021,44 @@ static bool onlyFeedsScalarAccesses(Value *Ptr, bool ReVec,
         (match(U, m_Store(m_Value(Op), m_Specific(Ptr))) && Op != Ptr))
       return true;
     if (auto *GEP = dyn_cast<GetElementPtrInst>(U))
-      return Depth < MaxDepth && GEP->getPointerOperand() == Ptr &&
+      return Depth > 0 && GEP->getPointerOperand() == Ptr &&
              GEP->hasAllConstantIndices() &&
-             onlyFeedsScalarAccesses(GEP, ReVec, Depth + 1);
+             onlyFeedsScalarAccesses(GEP, ReVec, Depth - 1);
     return false;
   });
 }
 
-/// Collects in \p GEPs the getelementptrs that the in-loop index computations
-/// \p VL end at, if they are only used by scalar accesses. Returns the loop
-/// containing all of them, or nullptr otherwise.
-static const Loop *
-collectScalarAccessGEPs(ArrayRef<Value *> VL, const LoopInfo &LI, bool ReVec,
-                        SmallVectorImpl<GetElementPtrInst *> &GEPs) {
+/// Returns the getelementptr whose index is computed by the short chain of
+/// single-use arithmetic starting at \p V, or nullptr otherwise.
+static GetElementPtrInst *getIndexChainGEP(Value *V) {
   constexpr unsigned MaxIndexChainLength = 3;
-  const Loop *L = nullptr;
-  for (Value *V : VL) {
-    Value *Cur = V;
-    GetElementPtrInst *GEP = nullptr;
-    for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
-      if (!isa<BinaryOperator, CastInst>(Cur) || !Cur->hasOneUse())
-        return nullptr;
-      User *U = Cur->user_back();
-      if ((GEP = dyn_cast<GetElementPtrInst>(U)))
-        break;
-      Cur = U;
-    }
-    if (!GEP || GEP->getPointerOperand() == Cur ||
-        !onlyFeedsScalarAccesses(GEP, ReVec))
-      return nullptr;
-    // LSR only removes the arithmetic computed in the loop itself.
-    const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
-    if (!GEPLoop || (L && L != GEPLoop) ||
-        LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
+  for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
+    if (!isa<BinaryOperator, CastInst>(V) || !V->hasOneUse())
       return nullptr;
-    L = GEPLoop;
-    GEPs.push_back(GEP);
+    User *U = V->user_back();
+    if (auto *GEP = dyn_cast<GetElementPtrInst>(U))
+      return GEP->getPointerOperand() != V ? GEP : nullptr;
+    V = U;
   }
-  return L;
-}
-
-/// Returns the address computed by \p GEP if it is an affine recurrence of
-/// \p L, or nullptr otherwise.
-static const SCEVAddRecExpr *
-getAffineAddress(GetElementPtrInst *GEP, const Loop *L, ScalarEvolution &SE) {
-  const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
-  return Addr && Addr->getLoop() == L && Addr->isAffine() ? Addr : nullptr;
+  return nullptr;
 }
 
 bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
                                     const LoopInfo &LI, bool ReVec) {
-  SmallVector<GetElementPtrInst *> GEPs;
-  const Loop *L = collectScalarAccessGEPs(VL, LI, ReVec, GEPs);
-  if (!L)
-    return false;
+  const Loop *L = nullptr;
   const SCEV *FirstAddr = nullptr;
-  for (GetElementPtrInst *GEP : GEPs) {
-    const SCEV *Addr = getAffineAddress(GEP, L, SE);
-    if (!Addr)
+  for (Value *V : VL) {
+    GetElementPtrInst *GEP = getIndexChainGEP(V);
+    if (!GEP || !onlyFeedsScalarAccesses(GEP, ReVec))
+      return false;
+    // LSR only removes the arithmetic computed in the loop itself.
+    const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
+    if (!GEPLoop || (L && L != GEPLoop) ||
+        LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
+      return false;
+    L = GEPLoop;
+    const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
+    if (!Addr || Addr->getLoop() != L || !Addr->isAffine())
       return false;
     if (!FirstAddr) {
       FirstAddr = Addr;
@@ -1092,14 +1071,10 @@ bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
   return true;
 }
 
-bool isGEPCandidateIndexBundle(
-    ArrayRef<Value *> VL, ScalarEvolution &SE, const LoopInfo &LI, bool ReVec,
-    function_ref<bool(GetElementPtrInst *)> IsCandidate) {
-  SmallVector<GetElementPtrInst *> GEPs;
-  const Loop *L = collectScalarAccessGEPs(VL, LI, ReVec, GEPs);
-  return L && all_of(GEPs, [&](GetElementPtrInst *GEP) {
-           return IsCandidate(GEP) && getAffineAddress(GEP, L, SE);
-         });
+bool isGEPCandidateIndex(Instruction *I,
+                         function_ref<bool(GetElementPtrInst *)> IsCandidate) {
+  GetElementPtrInst *GEP = getIndexChainGEP(I);
+  return GEP && IsCandidate(GEP);
 }
 
 Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable) {
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index f578230bfe292..91bcc75ef4c0b 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -389,12 +389,10 @@ bool isOnceUsedSeed(const Instruction *I);
 bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
                                     const LoopInfo &LI, bool ReVec);
 
-/// Returns true if each value in \p VL is an in-loop index computation ending
-/// at a getelementptr accepted by \p IsCandidate, used only by scalar accesses
-/// and computing an affine recurrence of that loop.
-bool isGEPCandidateIndexBundle(
-    ArrayRef<Value *> VL, ScalarEvolution &SE, const LoopInfo &LI, bool ReVec,
-    function_ref<bool(GetElementPtrInst *)> IsCandidate);
+/// Returns true if \p I starts a short chain of single-use arithmetic that
+/// computes the index of a getelementptr accepted by \p IsCandidate.
+bool isGEPCandidateIndex(Instruction *I,
+                         function_ref<bool(GetElementPtrInst *)> IsCandidate);
 
 /// If \p V is a single-use fpext of a single-use fptrunc forming a round-trip
 /// back to the type of \p V, returns the fptrunc; the round-trip source is its



More information about the llvm-commits mailing list