[llvm] [SLP] Avoid seeding related affine loop address computations (PR #226220)

Tim Besard via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 28 01:06:48 PDT 2026


https://github.com/maleadt updated https://github.com/llvm/llvm-project/pull/226220

>From 2b1cf1719389b8a575fa9a90e9fe76f50193456a Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Thu, 24 Sep 2026 12:49:47 +0200
Subject: [PATCH 1/5] [SLP] Add tests for affine loop address computations
 (NFC)

Precommit tests for vectorizing the index arithmetic of related affine
address recurrences that feed scalar loads in a loop.
---
 .../X86/strength-reducible-address.ll         | 696 ++++++++++++++++++
 1 file changed, 696 insertions(+)
 create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll

diff --git a/llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll b/llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll
new file mode 100644
index 0000000000000..553b472868edb
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll
@@ -0,0 +1,696 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu -mcpu=haswell < %s | FileCheck %s --check-prefixes=CHECK,AVX2
+; RUN: opt -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu -mcpu=skylake-avx512 < %s | FileCheck %s --check-prefixes=CHECK,AVX512
+
+; The addresses are affine recurrences of the loop that differ by loop-invariant
+; amounts. Keep their index arithmetic scalar, so loop strength reduction can
+; replace it with pointer increments, instead of vectorizing it and extracting
+; every lane. With AVX-512, the loads are gathered together with their
+; addresses, which is not affected.
+
+define i32 @strided_gather(ptr %base, ptr %dims, i64 %n) {
+; AVX2-LABEL: define i32 @strided_gather(
+; AVX2-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; AVX2-NEXT:  [[ENTRY:.*]]:
+; AVX2-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
+; AVX2-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; AVX2-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
+; AVX2-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
+; AVX2-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX2-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
+; AVX2-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX2-NEXT:    br label %[[LOOP:.*]]
+; AVX2:       [[LOOP]]:
+; AVX2-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; AVX2-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
+; AVX2-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
+; AVX2-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX2-NEXT:    [[TMP6:%.*]] = add <4 x i64> [[TMP5]], <i64 1, i64 2, i64 3, i64 4>
+; AVX2-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
+; AVX2-NEXT:    [[TMP7:%.*]] = mul <4 x i64> [[TMP3]], [[TMP6]]
+; AVX2-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP1]], [[TMP7]]
+; AVX2-NEXT:    [[TMP9:%.*]] = shl <4 x i64> [[TMP8]], splat (i64 2)
+; AVX2-NEXT:    [[TMP10:%.*]] = extractelement <4 x i64> [[TMP9]], i64 0
+; AVX2-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP10]]
+; AVX2-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 1
+; AVX2-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP11]]
+; AVX2-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[TMP9]], i64 2
+; AVX2-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP12]]
+; AVX2-NEXT:    [[TMP13:%.*]] = extractelement <4 x i64> [[TMP9]], i64 3
+; AVX2-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP13]]
+; AVX2-NEXT:    [[L0:%.*]] = load i32, ptr [[P0]], align 4
+; AVX2-NEXT:    [[L1:%.*]] = load i32, ptr [[P1]], align 4
+; AVX2-NEXT:    [[L2:%.*]] = load i32, ptr [[P2]], align 4
+; AVX2-NEXT:    [[L3:%.*]] = load i32, ptr [[P3]], align 4
+; AVX2-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[L0]], i64 0
+; AVX2-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[L1]], i64 1
+; AVX2-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[L2]], i64 2
+; AVX2-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[L3]], i64 3
+; AVX2-NEXT:    [[R:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[V3]])
+; AVX2-NEXT:    [[ACC_NEXT]] = add i32 [[ACC]], [[R]]
+; AVX2-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; AVX2-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; AVX2:       [[EXIT]]:
+; AVX2-NEXT:    ret i32 [[ACC_NEXT]]
+;
+; AVX512-LABEL: define i32 @strided_gather(
+; AVX512-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; AVX512-NEXT:  [[ENTRY:.*]]:
+; AVX512-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
+; AVX512-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; AVX512-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
+; AVX512-NEXT:    [[TMP0:%.*]] = insertelement <4 x ptr> poison, ptr [[BASE]], i64 0
+; AVX512-NEXT:    [[TMP1:%.*]] = shufflevector <4 x ptr> [[TMP0]], <4 x ptr> poison, <4 x i32> zeroinitializer
+; AVX512-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
+; AVX512-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX512-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
+; AVX512-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX512-NEXT:    br label %[[LOOP:.*]]
+; AVX512:       [[LOOP]]:
+; AVX512-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; AVX512-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
+; AVX512-NEXT:    [[TMP6:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
+; AVX512-NEXT:    [[TMP7:%.*]] = shufflevector <4 x i64> [[TMP6]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX512-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP7]], <i64 1, i64 2, i64 3, i64 4>
+; AVX512-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
+; AVX512-NEXT:    [[TMP9:%.*]] = mul <4 x i64> [[TMP5]], [[TMP8]]
+; AVX512-NEXT:    [[TMP10:%.*]] = add <4 x i64> [[TMP3]], [[TMP9]]
+; AVX512-NEXT:    [[TMP11:%.*]] = shl <4 x i64> [[TMP10]], splat (i64 2)
+; AVX512-NEXT:    [[TMP12:%.*]] = getelementptr i8, <4 x ptr> [[TMP1]], <4 x i64> [[TMP11]]
+; AVX512-NEXT:    [[TMP13:%.*]] = call <4 x i32> @llvm.masked.gather.v4i32.v4p0(<4 x ptr> align 4 [[TMP12]], <4 x i1> splat (i1 true), <4 x i32> poison)
+; AVX512-NEXT:    [[R:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP13]])
+; AVX512-NEXT:    [[ACC_NEXT]] = add i32 [[ACC]], [[R]]
+; AVX512-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; AVX512-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; AVX512:       [[EXIT]]:
+; AVX512-NEXT:    ret i32 [[ACC_NEXT]]
+;
+entry:
+  %stride = load i64, ptr %dims, align 8
+  %off.ptr = getelementptr i8, ptr %dims, i64 8
+  %off = load i64, ptr %off.ptr, align 8
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i32 [ 0, %entry ], [ %acc.next, %loop ]
+  %i1 = add i64 %iv, 1
+  %i2 = add i64 %iv, 2
+  %i3 = add i64 %iv, 3
+  %iv.next = add i64 %iv, 4
+  %m0 = mul i64 %stride, %i1
+  %m1 = mul i64 %stride, %i2
+  %m2 = mul i64 %stride, %i3
+  %m3 = mul i64 %stride, %iv.next
+  %a0 = add i64 %off, %m0
+  %a1 = add i64 %off, %m1
+  %a2 = add i64 %off, %m2
+  %a3 = add i64 %off, %m3
+  %s0 = shl i64 %a0, 2
+  %s1 = shl i64 %a1, 2
+  %s2 = shl i64 %a2, 2
+  %s3 = shl i64 %a3, 2
+  %p0 = getelementptr i8, ptr %base, i64 %s0
+  %p1 = getelementptr i8, ptr %base, i64 %s1
+  %p2 = getelementptr i8, ptr %base, i64 %s2
+  %p3 = getelementptr i8, ptr %base, i64 %s3
+  %l0 = load i32, ptr %p0, align 4
+  %l1 = load i32, ptr %p1, align 4
+  %l2 = load i32, ptr %p2, align 4
+  %l3 = load i32, ptr %p3, align 4
+  %v0 = insertelement <4 x i32> poison, i32 %l0, i64 0
+  %v1 = insertelement <4 x i32> %v0, i32 %l1, i64 1
+  %v2 = insertelement <4 x i32> %v1, i32 %l2, i64 2
+  %v3 = insertelement <4 x i32> %v2, i32 %l3, i64 3
+  %r = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> %v3)
+  %acc.next = add i32 %acc, %r
+  %done = icmp eq i64 %iv.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret i32 %acc.next
+}
+
+; Constant-offset GEPs between the addresses and the loads do not prevent this.
+; The loads are not gathered with AVX-512 here, so the index arithmetic would
+; be vectorized despite the extracts.
+
+define i32 @strided_gather_offset(ptr %base, ptr %dims, i64 %n) {
+; CHECK-LABEL: define i32 @strided_gather_offset(
+; CHECK-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
+; CHECK-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; CHECK-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
+; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
+; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP6:%.*]] = add <4 x i64> [[TMP5]], <i64 1, i64 2, i64 3, i64 4>
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = mul <4 x i64> [[TMP3]], [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP1]], [[TMP7]]
+; CHECK-NEXT:    [[TMP9:%.*]] = shl <4 x i64> [[TMP8]], splat (i64 2)
+; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <4 x i64> [[TMP9]], i64 0
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP10]]
+; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 1
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP11]]
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[TMP9]], i64 2
+; CHECK-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP12]]
+; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x i64> [[TMP9]], i64 3
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP13]]
+; CHECK-NEXT:    [[Q0:%.*]] = getelementptr i8, ptr [[P0]], i64 -4
+; CHECK-NEXT:    [[L0:%.*]] = load i32, ptr [[Q0]], align 4
+; CHECK-NEXT:    [[Q1:%.*]] = getelementptr i8, ptr [[P1]], i64 -4
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[Q1]], align 4
+; CHECK-NEXT:    [[Q2:%.*]] = getelementptr i8, ptr [[P2]], i64 -4
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[Q2]], align 4
+; CHECK-NEXT:    [[Q3:%.*]] = getelementptr i8, ptr [[P3]], i64 -4
+; CHECK-NEXT:    [[L3:%.*]] = load i32, ptr [[Q3]], align 4
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[L0]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[L1]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[L2]], i64 2
+; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[L3]], i64 3
+; CHECK-NEXT:    [[R:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[V3]])
+; CHECK-NEXT:    [[ACC_NEXT]] = add i32 [[ACC]], [[R]]
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret i32 [[ACC_NEXT]]
+;
+entry:
+  %stride = load i64, ptr %dims, align 8
+  %off.ptr = getelementptr i8, ptr %dims, i64 8
+  %off = load i64, ptr %off.ptr, align 8
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i32 [ 0, %entry ], [ %acc.next, %loop ]
+  %i1 = add i64 %iv, 1
+  %i2 = add i64 %iv, 2
+  %i3 = add i64 %iv, 3
+  %iv.next = add i64 %iv, 4
+  %m0 = mul i64 %stride, %i1
+  %m1 = mul i64 %stride, %i2
+  %m2 = mul i64 %stride, %i3
+  %m3 = mul i64 %stride, %iv.next
+  %a0 = add i64 %off, %m0
+  %a1 = add i64 %off, %m1
+  %a2 = add i64 %off, %m2
+  %a3 = add i64 %off, %m3
+  %s0 = shl i64 %a0, 2
+  %s1 = shl i64 %a1, 2
+  %s2 = shl i64 %a2, 2
+  %s3 = shl i64 %a3, 2
+  %p0 = getelementptr i8, ptr %base, i64 %s0
+  %p1 = getelementptr i8, ptr %base, i64 %s1
+  %p2 = getelementptr i8, ptr %base, i64 %s2
+  %p3 = getelementptr i8, ptr %base, i64 %s3
+  %q0 = getelementptr i8, ptr %p0, i64 -4
+  %l0 = load i32, ptr %q0, align 4
+  %q1 = getelementptr i8, ptr %p1, i64 -4
+  %l1 = load i32, ptr %q1, align 4
+  %q2 = getelementptr i8, ptr %p2, i64 -4
+  %l2 = load i32, ptr %q2, align 4
+  %q3 = getelementptr i8, ptr %p3, i64 -4
+  %l3 = load i32, ptr %q3, align 4
+  %v0 = insertelement <4 x i32> poison, i32 %l0, i64 0
+  %v1 = insertelement <4 x i32> %v0, i32 %l1, i64 1
+  %v2 = insertelement <4 x i32> %v1, i32 %l2, i64 2
+  %v3 = insertelement <4 x i32> %v2, i32 %l3, i64 3
+  %r = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> %v3)
+  %acc.next = add i32 %acc, %r
+  %done = icmp eq i64 %iv.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret i32 %acc.next
+}
+
+; Stores, and a load and a store of the same address, are handled the same.
+
+define void @strided_update(ptr %base, ptr %dims, i64 %n) {
+; CHECK-LABEL: define void @strided_update(
+; CHECK-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
+; CHECK-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; CHECK-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
+; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
+; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP6:%.*]] = add <4 x i64> [[TMP5]], <i64 1, i64 2, i64 3, i64 4>
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = mul <4 x i64> [[TMP3]], [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP1]], [[TMP7]]
+; CHECK-NEXT:    [[TMP9:%.*]] = shl <4 x i64> [[TMP8]], splat (i64 2)
+; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <4 x i64> [[TMP9]], i64 0
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP10]]
+; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 1
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP11]]
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[TMP9]], i64 2
+; CHECK-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP12]]
+; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x i64> [[TMP9]], i64 3
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP13]]
+; CHECK-NEXT:    [[L0:%.*]] = load i32, ptr [[P0]], align 4
+; CHECK-NEXT:    [[U0:%.*]] = add i32 [[L0]], 1
+; CHECK-NEXT:    store i32 [[U0]], ptr [[P0]], align 4
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[P1]], align 4
+; CHECK-NEXT:    [[U1:%.*]] = add i32 [[L1]], 1
+; CHECK-NEXT:    store i32 [[U1]], ptr [[P1]], align 4
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[P2]], align 4
+; CHECK-NEXT:    [[U2:%.*]] = add i32 [[L2]], 1
+; CHECK-NEXT:    store i32 [[U2]], ptr [[P2]], align 4
+; CHECK-NEXT:    [[L3:%.*]] = load i32, ptr [[P3]], align 4
+; CHECK-NEXT:    [[U3:%.*]] = add i32 [[L3]], 1
+; CHECK-NEXT:    store i32 [[U3]], ptr [[P3]], align 4
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %stride = load i64, ptr %dims, align 8
+  %off.ptr = getelementptr i8, ptr %dims, i64 8
+  %off = load i64, ptr %off.ptr, align 8
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %i1 = add i64 %iv, 1
+  %i2 = add i64 %iv, 2
+  %i3 = add i64 %iv, 3
+  %iv.next = add i64 %iv, 4
+  %m0 = mul i64 %stride, %i1
+  %m1 = mul i64 %stride, %i2
+  %m2 = mul i64 %stride, %i3
+  %m3 = mul i64 %stride, %iv.next
+  %a0 = add i64 %off, %m0
+  %a1 = add i64 %off, %m1
+  %a2 = add i64 %off, %m2
+  %a3 = add i64 %off, %m3
+  %s0 = shl i64 %a0, 2
+  %s1 = shl i64 %a1, 2
+  %s2 = shl i64 %a2, 2
+  %s3 = shl i64 %a3, 2
+  %p0 = getelementptr i8, ptr %base, i64 %s0
+  %p1 = getelementptr i8, ptr %base, i64 %s1
+  %p2 = getelementptr i8, ptr %base, i64 %s2
+  %p3 = getelementptr i8, ptr %base, i64 %s3
+  %l0 = load i32, ptr %p0, align 4
+  %u0 = add i32 %l0, 1
+  store i32 %u0, ptr %p0, align 4
+  %l1 = load i32, ptr %p1, align 4
+  %u1 = add i32 %l1, 1
+  store i32 %u1, ptr %p1, align 4
+  %l2 = load i32, ptr %p2, align 4
+  %u2 = add i32 %l2, 1
+  store i32 %u2, ptr %p2, align 4
+  %l3 = load i32, ptr %p3, align 4
+  %u3 = add i32 %l3, 1
+  store i32 %u3, ptr %p3, align 4
+  %done = icmp eq i64 %iv.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+; The offset is loaded in the loop, so the addresses are not recurrences of
+; the loop and the index arithmetic is vectorized as before.
+
+define i32 @variant_offset(ptr %base, ptr %dims, i64 %n) {
+; AVX2-LABEL: define i32 @variant_offset(
+; AVX2-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; AVX2-NEXT:  [[ENTRY:.*]]:
+; AVX2-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; AVX2-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
+; AVX2-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
+; AVX2-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX2-NEXT:    br label %[[LOOP:.*]]
+; AVX2:       [[LOOP]]:
+; AVX2-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; AVX2-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
+; AVX2-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
+; AVX2-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
+; AVX2-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX2-NEXT:    [[TMP4:%.*]] = add <4 x i64> [[TMP3]], <i64 1, i64 2, i64 3, i64 4>
+; AVX2-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
+; AVX2-NEXT:    [[TMP5:%.*]] = mul <4 x i64> [[TMP1]], [[TMP4]]
+; AVX2-NEXT:    [[TMP6:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
+; AVX2-NEXT:    [[TMP7:%.*]] = shufflevector <4 x i64> [[TMP6]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX2-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP7]], [[TMP5]]
+; AVX2-NEXT:    [[TMP9:%.*]] = shl <4 x i64> [[TMP8]], splat (i64 2)
+; AVX2-NEXT:    [[TMP10:%.*]] = extractelement <4 x i64> [[TMP9]], i64 0
+; AVX2-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP10]]
+; AVX2-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 1
+; AVX2-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP11]]
+; AVX2-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[TMP9]], i64 2
+; AVX2-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP12]]
+; AVX2-NEXT:    [[TMP13:%.*]] = extractelement <4 x i64> [[TMP9]], i64 3
+; AVX2-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP13]]
+; AVX2-NEXT:    [[L0:%.*]] = load i32, ptr [[P0]], align 4
+; AVX2-NEXT:    [[L1:%.*]] = load i32, ptr [[P1]], align 4
+; AVX2-NEXT:    [[L2:%.*]] = load i32, ptr [[P2]], align 4
+; AVX2-NEXT:    [[L3:%.*]] = load i32, ptr [[P3]], align 4
+; AVX2-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[L0]], i64 0
+; AVX2-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[L1]], i64 1
+; AVX2-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[L2]], i64 2
+; AVX2-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[L3]], i64 3
+; AVX2-NEXT:    [[R:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[V3]])
+; AVX2-NEXT:    [[ACC_NEXT]] = add i32 [[ACC]], [[R]]
+; AVX2-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; AVX2-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; AVX2:       [[EXIT]]:
+; AVX2-NEXT:    ret i32 [[ACC_NEXT]]
+;
+; AVX512-LABEL: define i32 @variant_offset(
+; AVX512-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; AVX512-NEXT:  [[ENTRY:.*]]:
+; AVX512-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; AVX512-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
+; AVX512-NEXT:    [[TMP0:%.*]] = insertelement <4 x ptr> poison, ptr [[BASE]], i64 0
+; AVX512-NEXT:    [[TMP1:%.*]] = shufflevector <4 x ptr> [[TMP0]], <4 x ptr> poison, <4 x i32> zeroinitializer
+; AVX512-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
+; AVX512-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX512-NEXT:    br label %[[LOOP:.*]]
+; AVX512:       [[LOOP]]:
+; AVX512-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; AVX512-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
+; AVX512-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
+; AVX512-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
+; AVX512-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX512-NEXT:    [[TMP6:%.*]] = add <4 x i64> [[TMP5]], <i64 1, i64 2, i64 3, i64 4>
+; AVX512-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
+; AVX512-NEXT:    [[TMP7:%.*]] = mul <4 x i64> [[TMP3]], [[TMP6]]
+; AVX512-NEXT:    [[TMP8:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
+; AVX512-NEXT:    [[TMP9:%.*]] = shufflevector <4 x i64> [[TMP8]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX512-NEXT:    [[TMP10:%.*]] = add <4 x i64> [[TMP9]], [[TMP7]]
+; AVX512-NEXT:    [[TMP11:%.*]] = shl <4 x i64> [[TMP10]], splat (i64 2)
+; AVX512-NEXT:    [[TMP12:%.*]] = getelementptr i8, <4 x ptr> [[TMP1]], <4 x i64> [[TMP11]]
+; AVX512-NEXT:    [[TMP13:%.*]] = call <4 x i32> @llvm.masked.gather.v4i32.v4p0(<4 x ptr> align 4 [[TMP12]], <4 x i1> splat (i1 true), <4 x i32> poison)
+; AVX512-NEXT:    [[R:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP13]])
+; AVX512-NEXT:    [[ACC_NEXT]] = add i32 [[ACC]], [[R]]
+; AVX512-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; AVX512-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; AVX512:       [[EXIT]]:
+; AVX512-NEXT:    ret i32 [[ACC_NEXT]]
+;
+entry:
+  %stride = load i64, ptr %dims, align 8
+  %off.ptr = getelementptr i8, ptr %dims, i64 8
+
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i32 [ 0, %entry ], [ %acc.next, %loop ]
+  %off = load i64, ptr %off.ptr, align 8
+  %i1 = add i64 %iv, 1
+  %i2 = add i64 %iv, 2
+  %i3 = add i64 %iv, 3
+  %iv.next = add i64 %iv, 4
+  %m0 = mul i64 %stride, %i1
+  %m1 = mul i64 %stride, %i2
+  %m2 = mul i64 %stride, %i3
+  %m3 = mul i64 %stride, %iv.next
+  %a0 = add i64 %off, %m0
+  %a1 = add i64 %off, %m1
+  %a2 = add i64 %off, %m2
+  %a3 = add i64 %off, %m3
+  %s0 = shl i64 %a0, 2
+  %s1 = shl i64 %a1, 2
+  %s2 = shl i64 %a2, 2
+  %s3 = shl i64 %a3, 2
+  %p0 = getelementptr i8, ptr %base, i64 %s0
+  %p1 = getelementptr i8, ptr %base, i64 %s1
+  %p2 = getelementptr i8, ptr %base, i64 %s2
+  %p3 = getelementptr i8, ptr %base, i64 %s3
+  %l0 = load i32, ptr %p0, align 4
+  %l1 = load i32, ptr %p1, align 4
+  %l2 = load i32, ptr %p2, align 4
+  %l3 = load i32, ptr %p3, align 4
+  %v0 = insertelement <4 x i32> poison, i32 %l0, i64 0
+  %v1 = insertelement <4 x i32> %v0, i32 %l1, i64 1
+  %v2 = insertelement <4 x i32> %v1, i32 %l2, i64 2
+  %v3 = insertelement <4 x i32> %v2, i32 %l3, i64 3
+  %r = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> %v3)
+  %acc.next = add i32 %acc, %r
+  %done = icmp eq i64 %iv.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret i32 %acc.next
+}
+
+; One of the addresses is also passed to a call, so the index arithmetic is
+; vectorized as before.
+
+define i32 @escaping_address(ptr %base, ptr %dims, i64 %n) {
+; CHECK-LABEL: define i32 @escaping_address(
+; CHECK-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
+; CHECK-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; CHECK-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
+; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
+; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP6:%.*]] = add <4 x i64> [[TMP5]], <i64 1, i64 2, i64 3, i64 4>
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = mul <4 x i64> [[TMP3]], [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP1]], [[TMP7]]
+; CHECK-NEXT:    [[TMP9:%.*]] = shl <4 x i64> [[TMP8]], splat (i64 2)
+; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <4 x i64> [[TMP9]], i64 0
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP10]]
+; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 1
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP11]]
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[TMP9]], i64 2
+; CHECK-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP12]]
+; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x i64> [[TMP9]], i64 3
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP13]]
+; CHECK-NEXT:    [[L0:%.*]] = load i32, ptr [[P0]], align 4
+; CHECK-NEXT:    call void @use(ptr [[P0]])
+; CHECK-NEXT:    [[L1:%.*]] = load i32, ptr [[P1]], align 4
+; CHECK-NEXT:    [[L2:%.*]] = load i32, ptr [[P2]], align 4
+; CHECK-NEXT:    [[L3:%.*]] = load i32, ptr [[P3]], align 4
+; CHECK-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[L0]], i64 0
+; CHECK-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[L1]], i64 1
+; CHECK-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[L2]], i64 2
+; CHECK-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[L3]], i64 3
+; CHECK-NEXT:    [[R:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[V3]])
+; CHECK-NEXT:    [[ACC_NEXT]] = add i32 [[ACC]], [[R]]
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret i32 [[ACC_NEXT]]
+;
+entry:
+  %stride = load i64, ptr %dims, align 8
+  %off.ptr = getelementptr i8, ptr %dims, i64 8
+  %off = load i64, ptr %off.ptr, align 8
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i32 [ 0, %entry ], [ %acc.next, %loop ]
+  %i1 = add i64 %iv, 1
+  %i2 = add i64 %iv, 2
+  %i3 = add i64 %iv, 3
+  %iv.next = add i64 %iv, 4
+  %m0 = mul i64 %stride, %i1
+  %m1 = mul i64 %stride, %i2
+  %m2 = mul i64 %stride, %i3
+  %m3 = mul i64 %stride, %iv.next
+  %a0 = add i64 %off, %m0
+  %a1 = add i64 %off, %m1
+  %a2 = add i64 %off, %m2
+  %a3 = add i64 %off, %m3
+  %s0 = shl i64 %a0, 2
+  %s1 = shl i64 %a1, 2
+  %s2 = shl i64 %a2, 2
+  %s3 = shl i64 %a3, 2
+  %p0 = getelementptr i8, ptr %base, i64 %s0
+  %p1 = getelementptr i8, ptr %base, i64 %s1
+  %p2 = getelementptr i8, ptr %base, i64 %s2
+  %p3 = getelementptr i8, ptr %base, i64 %s3
+  %l0 = load i32, ptr %p0, align 4
+  call void @use(ptr %p0)
+  %l1 = load i32, ptr %p1, align 4
+  %l2 = load i32, ptr %p2, align 4
+  %l3 = load i32, ptr %p3, align 4
+  %v0 = insertelement <4 x i32> poison, i32 %l0, i64 0
+  %v1 = insertelement <4 x i32> %v0, i32 %l1, i64 1
+  %v2 = insertelement <4 x i32> %v1, i32 %l2, i64 2
+  %v3 = insertelement <4 x i32> %v2, i32 %l3, i64 3
+  %r = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> %v3)
+  %acc.next = add i32 %acc, %r
+  %done = icmp eq i64 %iv.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret i32 %acc.next
+}
+
+; The offsets are computed before the loop, where LSR does not remove them, so
+; they are left to the cost model. Skipping the in-loop bundle lets them be
+; vectorized as once-used seeds.
+
+define i32 @preheader_offsets(ptr %base, ptr %offs, i64 %n) {
+; AVX2-LABEL: define i32 @preheader_offsets(
+; AVX2-SAME: ptr [[BASE:%.*]], ptr [[OFFS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; AVX2-NEXT:  [[ENTRY:.*]]:
+; AVX2-NEXT:    [[X1_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 8
+; AVX2-NEXT:    [[X2_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 16
+; AVX2-NEXT:    [[X3_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 24
+; AVX2-NEXT:    [[X0:%.*]] = load i64, ptr [[OFFS]], align 8
+; AVX2-NEXT:    [[X1:%.*]] = load i64, ptr [[X1_PTR]], align 8
+; AVX2-NEXT:    [[X2:%.*]] = load i64, ptr [[X2_PTR]], align 8
+; AVX2-NEXT:    [[X3:%.*]] = load i64, ptr [[X3_PTR]], align 8
+; AVX2-NEXT:    [[Y0_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 32
+; AVX2-NEXT:    [[Y1_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 40
+; AVX2-NEXT:    [[Y2_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 48
+; AVX2-NEXT:    [[Y3_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 56
+; AVX2-NEXT:    [[Y0:%.*]] = load i64, ptr [[Y0_PTR]], align 8
+; AVX2-NEXT:    [[Y1:%.*]] = load i64, ptr [[Y1_PTR]], align 8
+; AVX2-NEXT:    [[Y2:%.*]] = load i64, ptr [[Y2_PTR]], align 8
+; AVX2-NEXT:    [[Y3:%.*]] = load i64, ptr [[Y3_PTR]], align 8
+; AVX2-NEXT:    [[T0:%.*]] = add i64 [[X0]], [[Y0]]
+; AVX2-NEXT:    [[T1:%.*]] = add i64 [[X1]], [[Y1]]
+; AVX2-NEXT:    [[T2:%.*]] = add i64 [[X2]], [[Y2]]
+; AVX2-NEXT:    [[T3:%.*]] = add i64 [[X3]], [[Y3]]
+; AVX2-NEXT:    [[O0:%.*]] = shl i64 [[T0]], 2
+; AVX2-NEXT:    [[O1:%.*]] = shl i64 [[T1]], 2
+; AVX2-NEXT:    [[O2:%.*]] = shl i64 [[T2]], 2
+; AVX2-NEXT:    [[O3:%.*]] = shl i64 [[T3]], 2
+; AVX2-NEXT:    br label %[[LOOP:.*]]
+; AVX2:       [[LOOP]]:
+; AVX2-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; AVX2-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
+; AVX2-NEXT:    [[A0:%.*]] = add i64 [[IV]], [[O0]]
+; AVX2-NEXT:    [[A1:%.*]] = add i64 [[IV]], [[O1]]
+; AVX2-NEXT:    [[A2:%.*]] = add i64 [[IV]], [[O2]]
+; AVX2-NEXT:    [[A3:%.*]] = add i64 [[IV]], [[O3]]
+; AVX2-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[A0]]
+; AVX2-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[A1]]
+; AVX2-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[A2]]
+; AVX2-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[A3]]
+; AVX2-NEXT:    [[L0:%.*]] = load i32, ptr [[P0]], align 4
+; AVX2-NEXT:    [[L1:%.*]] = load i32, ptr [[P1]], align 4
+; AVX2-NEXT:    [[L2:%.*]] = load i32, ptr [[P2]], align 4
+; AVX2-NEXT:    [[L3:%.*]] = load i32, ptr [[P3]], align 4
+; AVX2-NEXT:    [[V0:%.*]] = insertelement <4 x i32> poison, i32 [[L0]], i64 0
+; AVX2-NEXT:    [[V1:%.*]] = insertelement <4 x i32> [[V0]], i32 [[L1]], i64 1
+; AVX2-NEXT:    [[V2:%.*]] = insertelement <4 x i32> [[V1]], i32 [[L2]], i64 2
+; AVX2-NEXT:    [[V3:%.*]] = insertelement <4 x i32> [[V2]], i32 [[L3]], i64 3
+; AVX2-NEXT:    [[R:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[V3]])
+; AVX2-NEXT:    [[ACC_NEXT]] = add i32 [[ACC]], [[R]]
+; AVX2-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
+; AVX2-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; AVX2-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; AVX2:       [[EXIT]]:
+; AVX2-NEXT:    ret i32 [[ACC_NEXT]]
+;
+; AVX512-LABEL: define i32 @preheader_offsets(
+; AVX512-SAME: ptr [[BASE:%.*]], ptr [[OFFS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; AVX512-NEXT:  [[ENTRY:.*]]:
+; AVX512-NEXT:    [[Y0_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 32
+; AVX512-NEXT:    [[TMP0:%.*]] = load <4 x i64>, ptr [[OFFS]], align 8
+; AVX512-NEXT:    [[TMP1:%.*]] = load <4 x i64>, ptr [[Y0_PTR]], align 8
+; AVX512-NEXT:    [[TMP2:%.*]] = add <4 x i64> [[TMP0]], [[TMP1]]
+; AVX512-NEXT:    [[TMP3:%.*]] = shl <4 x i64> [[TMP2]], splat (i64 2)
+; AVX512-NEXT:    [[TMP4:%.*]] = insertelement <4 x ptr> poison, ptr [[BASE]], i64 0
+; AVX512-NEXT:    [[TMP5:%.*]] = shufflevector <4 x ptr> [[TMP4]], <4 x ptr> poison, <4 x i32> zeroinitializer
+; AVX512-NEXT:    br label %[[LOOP:.*]]
+; AVX512:       [[LOOP]]:
+; AVX512-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; AVX512-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
+; AVX512-NEXT:    [[TMP6:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
+; AVX512-NEXT:    [[TMP7:%.*]] = shufflevector <4 x i64> [[TMP6]], <4 x i64> poison, <4 x i32> zeroinitializer
+; AVX512-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP7]], [[TMP3]]
+; AVX512-NEXT:    [[TMP9:%.*]] = getelementptr i8, <4 x ptr> [[TMP5]], <4 x i64> [[TMP8]]
+; AVX512-NEXT:    [[TMP10:%.*]] = call <4 x i32> @llvm.masked.gather.v4i32.v4p0(<4 x ptr> align 4 [[TMP9]], <4 x i1> splat (i1 true), <4 x i32> poison)
+; AVX512-NEXT:    [[R:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP10]])
+; AVX512-NEXT:    [[ACC_NEXT]] = add i32 [[ACC]], [[R]]
+; AVX512-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
+; AVX512-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; AVX512-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; AVX512:       [[EXIT]]:
+; AVX512-NEXT:    ret i32 [[ACC_NEXT]]
+;
+entry:
+  %x1.ptr = getelementptr i8, ptr %offs, i64 8
+  %x2.ptr = getelementptr i8, ptr %offs, i64 16
+  %x3.ptr = getelementptr i8, ptr %offs, i64 24
+  %x0 = load i64, ptr %offs, align 8
+  %x1 = load i64, ptr %x1.ptr, align 8
+  %x2 = load i64, ptr %x2.ptr, align 8
+  %x3 = load i64, ptr %x3.ptr, align 8
+  %y0.ptr = getelementptr i8, ptr %offs, i64 32
+  %y1.ptr = getelementptr i8, ptr %offs, i64 40
+  %y2.ptr = getelementptr i8, ptr %offs, i64 48
+  %y3.ptr = getelementptr i8, ptr %offs, i64 56
+  %y0 = load i64, ptr %y0.ptr, align 8
+  %y1 = load i64, ptr %y1.ptr, align 8
+  %y2 = load i64, ptr %y2.ptr, align 8
+  %y3 = load i64, ptr %y3.ptr, align 8
+  %t0 = add i64 %x0, %y0
+  %t1 = add i64 %x1, %y1
+  %t2 = add i64 %x2, %y2
+  %t3 = add i64 %x3, %y3
+  %o0 = shl i64 %t0, 2
+  %o1 = shl i64 %t1, 2
+  %o2 = shl i64 %t2, 2
+  %o3 = shl i64 %t3, 2
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i32 [ 0, %entry ], [ %acc.next, %loop ]
+  %a0 = add i64 %iv, %o0
+  %a1 = add i64 %iv, %o1
+  %a2 = add i64 %iv, %o2
+  %a3 = add i64 %iv, %o3
+  %p0 = getelementptr i8, ptr %base, i64 %a0
+  %p1 = getelementptr i8, ptr %base, i64 %a1
+  %p2 = getelementptr i8, ptr %base, i64 %a2
+  %p3 = getelementptr i8, ptr %base, i64 %a3
+  %l0 = load i32, ptr %p0, align 4
+  %l1 = load i32, ptr %p1, align 4
+  %l2 = load i32, ptr %p2, align 4
+  %l3 = load i32, ptr %p3, align 4
+  %v0 = insertelement <4 x i32> poison, i32 %l0, i64 0
+  %v1 = insertelement <4 x i32> %v0, i32 %l1, i64 1
+  %v2 = insertelement <4 x i32> %v1, i32 %l2, i64 2
+  %v3 = insertelement <4 x i32> %v2, i32 %l3, i64 3
+  %r = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> %v3)
+  %acc.next = add i32 %acc, %r
+  %iv.next = add i64 %iv, 4
+  %done = icmp eq i64 %iv.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret i32 %acc.next
+}
+
+declare void @use(ptr)

>From 39edb908eab4cfa1a5c08d18e7ccb655e749139c Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Thu, 24 Sep 2026 12:55:55 +0200
Subject: [PATCH 2/5] [SLP] Avoid seeding related affine loop address
 computations

When the loop vectorizer scalarizes strided accesses, each lane computes
its own address from the induction variable, a loop-invariant stride
and a loop-invariant offset, e.g.
base + ((off + stride * (iv + k)) << 2). These addresses are affine
recurrences that loop strength reduction turns into pointer increments.
With the loop-aware cost model (#150450), SLP now vectorizes the index
arithmetic instead: the multiplies, adds and shifts are done on
<4 x i64> in the loop, and every lane is extracted again to feed the
scalar loads. The cost model compares this against the scalar index
arithmetic, which LSR removes, so the vector form looks profitable
while the loop gets longer. For strided_gather_offset in the new test,
the loop grows from 13 to 30 instructions on Haswell, and from 13 to 23
on skylake-avx512, which has a vector i64 multiply.

vectorizeGEPIndices already drops pairs of getelementptrs with a
constant difference, because one address can be computed from the
other. Generalize this to loop-invariant differences: skip a bundle
when each lane is a short single-use chain of index arithmetic in a
loop, ending as the index of a getelementptr that is only used by
scalar loads and stores, and the addresses are affine recurrences of
that loop that differ by loop-invariant amounts. Check this in
tryToVectorizeList, because the once-used seeds vectorize the same
arithmetic when only the getelementptr seeds are skipped. Loads that
are vectorized as a gather together with their addresses are not
affected.

Assisted-by: Claude Code, Codex
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    |  71 ++++++++
 .../SLPVectorizer/AArch64/getelementptr.ll    |  27 +--
 .../X86/strength-reducible-address.ll         | 155 ++++++++----------
 3 files changed, 149 insertions(+), 104 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 442cc9a0895b2..fb000ef645ffe 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -30227,6 +30227,71 @@ void SLPVectorizerPass::collectSeedInstructions(BasicBlock *BB) {
   }
 }
 
+/// Returns true if \p Ptr is only used by scalar loads and stores, directly or
+/// through getelementptrs with constant offsets.
+static bool onlyFeedsScalarAccesses(const Value *Ptr, unsigned Depth = 0) {
+  constexpr unsigned MaxDepth = 2;
+  return !Ptr->use_empty() && all_of(Ptr->users(), [&](const User *U) {
+    if (const auto *Load = dyn_cast<LoadInst>(U))
+      return !Load->getType()->isVectorTy();
+    if (const auto *Store = dyn_cast<StoreInst>(U))
+      return Store->getPointerOperand() == Ptr &&
+             !Store->getValueOperand()->getType()->isVectorTy();
+    if (const auto *GEP = dyn_cast<GetElementPtrInst>(U))
+      return Depth < MaxDepth && GEP->getPointerOperand() == Ptr &&
+             GEP->hasAllConstantIndices() &&
+             onlyFeedsScalarAccesses(GEP, Depth + 1);
+    return false;
+  });
+}
+
+/// Leave related affine address recurrences feeding scalar accesses for loop
+/// strength reduction. Vectorizing their indices can retain expensive
+/// arithmetic and require an extract for each lane instead of a scalar pointer
+/// increment.
+static bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL,
+                                           ScalarEvolution &SE, LoopInfo &LI) {
+  constexpr unsigned MaxIndexChainLength = 3;
+  const Loop *L = nullptr;
+  SmallVector<GetElementPtrInst *> GEPs;
+  for (Value *V : VL) {
+    Value *Cur = V;
+    GetElementPtrInst *GEP = nullptr;
+    for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
+      if (!isa<BinaryOperator, CastInst>(Cur) || !Cur->hasOneUse())
+        return false;
+      User *U = Cur->user_back();
+      if ((GEP = dyn_cast<GetElementPtrInst>(U)))
+        break;
+      Cur = U;
+    }
+    if (!GEP || GEP->getPointerOperand() == Cur ||
+        !onlyFeedsScalarAccesses(GEP))
+      return false;
+    // LSR only removes the arithmetic computed in the loop itself.
+    const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
+    if (!GEPLoop || (L && L != GEPLoop) ||
+        LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
+      return false;
+    L = GEPLoop;
+    GEPs.push_back(GEP);
+  }
+  const SCEV *FirstAddr = nullptr;
+  for (GetElementPtrInst *GEP : GEPs) {
+    const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
+    if (!Addr || Addr->getLoop() != L || !Addr->isAffine())
+      return false;
+    if (!FirstAddr) {
+      FirstAddr = Addr;
+      continue;
+    }
+    const SCEV *Diff = SE.getMinusSCEV(Addr, FirstAddr);
+    if (isa<SCEVCouldNotCompute>(Diff) || !SE.isLoopInvariant(Diff, L))
+      return false;
+  }
+  return true;
+}
+
 bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
                                            bool MaxVFOnly,
                                            bool StandaloneSeeds) {
@@ -30358,6 +30423,12 @@ bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
         continue;
       }
 
+      if (isStrengthReducibleIndexBundle(Ops, *SE, *LI)) {
+        LLVM_DEBUG(dbgs() << "SLP: Not vectorizing strength-reducible address "
+                             "computations.\n");
+        continue;
+      }
+
       LLVM_DEBUG(dbgs() << "SLP: Analyzing " << ActualVF << " operations "
                         << "\n");
 
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
index d7f5ad7861034..ee7b4ab5a411d 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
@@ -8,6 +8,8 @@
 ; getelementptrs when they are known to have a constant difference. Such pairs
 ; are likely not good candidates for vectorization since one can be computed
 ; from the other. We use an unprofitable threshold to force vectorization.
+; The same holds for addresses that are affine recurrences of the loop with
+; loop-invariant differences, as in getelementptr_4x32.
 ;
 ; int getelementptr(int *g, int n, int w, int x, int y, int z) {
 ;   int sum = 0;
@@ -30,25 +32,12 @@
 ; YAML-NEXT:   - String:          ' and with tree size '
 ; YAML-NEXT:   - TreeSize:        '1'
 
-; YAML:      --- !Passed
-; YAML-NEXT: Pass:            slp-vectorizer
-; YAML-NEXT: Name:            VectorizedList
-; YAML-NEXT: Function:        getelementptr_4x32
-; YAML-NEXT: Args:
-; YAML-NEXT:   - String:          'SLP vectorized with cost '
-; YAML-NEXT:   - Cost:            '10'
-; YAML-NEXT:   - String:          ' and with tree size '
-; YAML-NEXT:   - TreeSize:        '3'
-
 define i32 @getelementptr_4x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y, i32 %z) {
 ; CHECK-LABEL: @getelementptr_4x32(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[CMP31:%.*]] = icmp sgt i32 [[N:%.*]], 0
 ; CHECK-NEXT:    br i1 [[CMP31]], label [[FOR_BODY_PREHEADER:%.*]], label [[FOR_COND_CLEANUP:%.*]]
 ; CHECK:       for.body.preheader:
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x i32> <i32 0, i32 poison>, i32 [[X:%.*]], i64 1
-; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <2 x i32> poison, i32 [[Y:%.*]], i64 0
-; CHECK-NEXT:    [[TMP5:%.*]] = insertelement <2 x i32> [[TMP4]], i32 [[Z:%.*]], i64 1
 ; CHECK-NEXT:    br label [[FOR_BODY:%.*]]
 ; CHECK:       for.cond.cleanup.loopexit:
 ; CHECK-NEXT:    br label [[FOR_COND_CLEANUP]]
@@ -63,20 +52,16 @@ define i32 @getelementptr_4x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y
 ; CHECK-NEXT:    [[SUM_32:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[OP_RDX:%.*]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x i32> [ zeroinitializer, [[FOR_BODY_PREHEADER]] ], [ [[SLPRDX_ACC1]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[T4:%.*]] = shl nsw i32 [[SUM_32]], 1
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x i32> poison, i32 [[T4]], i64 0
-; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP3:%.*]] = add nsw <2 x i32> [[TMP2]], [[TMP0]]
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <2 x i32> [[TMP3]], i64 0
+; CHECK-NEXT:    [[TMP12:%.*]] = add nsw i32 [[T4]], 0
 ; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[G:%.*]], i32 [[TMP12]]
 ; CHECK-NEXT:    [[T6:%.*]] = load i32, ptr [[ARRAYIDX]], align 4
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <2 x i32> [[TMP3]], i64 1
+; CHECK-NEXT:    [[TMP11:%.*]] = add nsw i32 [[T4]], [[X:%.*]]
 ; CHECK-NEXT:    [[ARRAYIDX5:%.*]] = getelementptr inbounds i32, ptr [[G]], i32 [[TMP11]]
 ; CHECK-NEXT:    [[T8:%.*]] = load i32, ptr [[ARRAYIDX5]], align 4
-; CHECK-NEXT:    [[TMP16:%.*]] = add nsw <2 x i32> [[TMP2]], [[TMP5]]
-; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <2 x i32> [[TMP16]], i64 0
+; CHECK-NEXT:    [[TMP13:%.*]] = add nsw i32 [[T4]], [[Y:%.*]]
 ; CHECK-NEXT:    [[ARRAYIDX10:%.*]] = getelementptr inbounds i32, ptr [[G]], i32 [[TMP13]]
 ; CHECK-NEXT:    [[T10:%.*]] = load i32, ptr [[ARRAYIDX10]], align 4
-; CHECK-NEXT:    [[TMP14:%.*]] = extractelement <2 x i32> [[TMP16]], i64 1
+; CHECK-NEXT:    [[TMP14:%.*]] = add nsw i32 [[T4]], [[Z:%.*]]
 ; CHECK-NEXT:    [[ARRAYIDX15:%.*]] = getelementptr inbounds i32, ptr [[G]], i32 [[TMP14]]
 ; CHECK-NEXT:    [[T12:%.*]] = load i32, ptr [[ARRAYIDX15]], align 4
 ; CHECK-NEXT:    [[TMP17:%.*]] = insertelement <4 x i32> poison, i32 [[T6]], i64 0
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll b/llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll
index 553b472868edb..86e00db13015a 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/strength-reducible-address.ll
@@ -12,32 +12,33 @@ define i32 @strided_gather(ptr %base, ptr %dims, i64 %n) {
 ; AVX2-LABEL: define i32 @strided_gather(
 ; AVX2-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
 ; AVX2-NEXT:  [[ENTRY:.*]]:
-; AVX2-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; AVX2-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; AVX2-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; AVX2-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
-; AVX2-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
-; AVX2-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX2-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
-; AVX2-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
 ; AVX2-NEXT:    br label %[[LOOP:.*]]
 ; AVX2:       [[LOOP]]:
 ; AVX2-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
 ; AVX2-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
-; AVX2-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
-; AVX2-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
-; AVX2-NEXT:    [[TMP6:%.*]] = add <4 x i64> [[TMP5]], <i64 1, i64 2, i64 3, i64 4>
+; AVX2-NEXT:    [[I1:%.*]] = add i64 [[IV]], 1
+; AVX2-NEXT:    [[I2:%.*]] = add i64 [[IV]], 2
+; AVX2-NEXT:    [[I3:%.*]] = add i64 [[IV]], 3
 ; AVX2-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
-; AVX2-NEXT:    [[TMP7:%.*]] = mul <4 x i64> [[TMP3]], [[TMP6]]
-; AVX2-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP1]], [[TMP7]]
-; AVX2-NEXT:    [[TMP9:%.*]] = shl <4 x i64> [[TMP8]], splat (i64 2)
-; AVX2-NEXT:    [[TMP10:%.*]] = extractelement <4 x i64> [[TMP9]], i64 0
-; AVX2-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP10]]
-; AVX2-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 1
-; AVX2-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP11]]
-; AVX2-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[TMP9]], i64 2
-; AVX2-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP12]]
-; AVX2-NEXT:    [[TMP13:%.*]] = extractelement <4 x i64> [[TMP9]], i64 3
-; AVX2-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP13]]
+; AVX2-NEXT:    [[M0:%.*]] = mul i64 [[STRIDE]], [[I1]]
+; AVX2-NEXT:    [[M1:%.*]] = mul i64 [[STRIDE]], [[I2]]
+; AVX2-NEXT:    [[M2:%.*]] = mul i64 [[STRIDE]], [[I3]]
+; AVX2-NEXT:    [[M3:%.*]] = mul i64 [[STRIDE]], [[IV_NEXT]]
+; AVX2-NEXT:    [[A0:%.*]] = add i64 [[OFF]], [[M0]]
+; AVX2-NEXT:    [[A1:%.*]] = add i64 [[OFF]], [[M1]]
+; AVX2-NEXT:    [[A2:%.*]] = add i64 [[OFF]], [[M2]]
+; AVX2-NEXT:    [[A3:%.*]] = add i64 [[OFF]], [[M3]]
+; AVX2-NEXT:    [[S0:%.*]] = shl i64 [[A0]], 2
+; AVX2-NEXT:    [[S1:%.*]] = shl i64 [[A1]], 2
+; AVX2-NEXT:    [[S2:%.*]] = shl i64 [[A2]], 2
+; AVX2-NEXT:    [[S3:%.*]] = shl i64 [[A3]], 2
+; AVX2-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S0]]
+; AVX2-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S1]]
+; AVX2-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S2]]
+; AVX2-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S3]]
 ; AVX2-NEXT:    [[L0:%.*]] = load i32, ptr [[P0]], align 4
 ; AVX2-NEXT:    [[L1:%.*]] = load i32, ptr [[P1]], align 4
 ; AVX2-NEXT:    [[L2:%.*]] = load i32, ptr [[P2]], align 4
@@ -139,32 +140,33 @@ define i32 @strided_gather_offset(ptr %base, ptr %dims, i64 %n) {
 ; CHECK-LABEL: define i32 @strided_gather_offset(
 ; CHECK-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
-; CHECK-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; CHECK-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; CHECK-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; CHECK-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
-; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP6:%.*]] = add <4 x i64> [[TMP5]], <i64 1, i64 2, i64 3, i64 4>
+; CHECK-NEXT:    [[I1:%.*]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[I2:%.*]] = add i64 [[IV]], 2
+; CHECK-NEXT:    [[I3:%.*]] = add i64 [[IV]], 3
 ; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
-; CHECK-NEXT:    [[TMP7:%.*]] = mul <4 x i64> [[TMP3]], [[TMP6]]
-; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP1]], [[TMP7]]
-; CHECK-NEXT:    [[TMP9:%.*]] = shl <4 x i64> [[TMP8]], splat (i64 2)
-; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <4 x i64> [[TMP9]], i64 0
-; CHECK-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP10]]
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 1
-; CHECK-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP11]]
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[TMP9]], i64 2
-; CHECK-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP12]]
-; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x i64> [[TMP9]], i64 3
-; CHECK-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP13]]
+; CHECK-NEXT:    [[M0:%.*]] = mul i64 [[STRIDE]], [[I1]]
+; CHECK-NEXT:    [[M1:%.*]] = mul i64 [[STRIDE]], [[I2]]
+; CHECK-NEXT:    [[M2:%.*]] = mul i64 [[STRIDE]], [[I3]]
+; CHECK-NEXT:    [[M3:%.*]] = mul i64 [[STRIDE]], [[IV_NEXT]]
+; CHECK-NEXT:    [[A0:%.*]] = add i64 [[OFF]], [[M0]]
+; CHECK-NEXT:    [[A1:%.*]] = add i64 [[OFF]], [[M1]]
+; CHECK-NEXT:    [[A2:%.*]] = add i64 [[OFF]], [[M2]]
+; CHECK-NEXT:    [[A3:%.*]] = add i64 [[OFF]], [[M3]]
+; CHECK-NEXT:    [[S0:%.*]] = shl i64 [[A0]], 2
+; CHECK-NEXT:    [[S1:%.*]] = shl i64 [[A1]], 2
+; CHECK-NEXT:    [[S2:%.*]] = shl i64 [[A2]], 2
+; CHECK-NEXT:    [[S3:%.*]] = shl i64 [[A3]], 2
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S0]]
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S1]]
+; CHECK-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S2]]
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S3]]
 ; CHECK-NEXT:    [[Q0:%.*]] = getelementptr i8, ptr [[P0]], i64 -4
 ; CHECK-NEXT:    [[L0:%.*]] = load i32, ptr [[Q0]], align 4
 ; CHECK-NEXT:    [[Q1:%.*]] = getelementptr i8, ptr [[P1]], i64 -4
@@ -240,31 +242,32 @@ define void @strided_update(ptr %base, ptr %dims, i64 %n) {
 ; CHECK-LABEL: define void @strided_update(
 ; CHECK-SAME: ptr [[BASE:%.*]], ptr [[DIMS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
-; CHECK-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; CHECK-NEXT:    [[STRIDE:%.*]] = load i64, ptr [[DIMS]], align 8
+; CHECK-NEXT:    [[OFF_PTR:%.*]] = getelementptr i8, ptr [[DIMS]], i64 8
 ; CHECK-NEXT:    [[OFF:%.*]] = load i64, ptr [[OFF_PTR]], align 8
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x i64> poison, i64 [[OFF]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i64> poison, i64 [[STRIDE]], i64 0
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i64> poison, i64 [[IV]], i64 0
-; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i64> [[TMP4]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP6:%.*]] = add <4 x i64> [[TMP5]], <i64 1, i64 2, i64 3, i64 4>
+; CHECK-NEXT:    [[I1:%.*]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[I2:%.*]] = add i64 [[IV]], 2
+; CHECK-NEXT:    [[I3:%.*]] = add i64 [[IV]], 3
 ; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 4
-; CHECK-NEXT:    [[TMP7:%.*]] = mul <4 x i64> [[TMP3]], [[TMP6]]
-; CHECK-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[TMP1]], [[TMP7]]
-; CHECK-NEXT:    [[TMP9:%.*]] = shl <4 x i64> [[TMP8]], splat (i64 2)
-; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <4 x i64> [[TMP9]], i64 0
-; CHECK-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP10]]
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 1
-; CHECK-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP11]]
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[TMP9]], i64 2
-; CHECK-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP12]]
-; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x i64> [[TMP9]], i64 3
-; CHECK-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[TMP13]]
+; CHECK-NEXT:    [[M0:%.*]] = mul i64 [[STRIDE]], [[I1]]
+; CHECK-NEXT:    [[M1:%.*]] = mul i64 [[STRIDE]], [[I2]]
+; CHECK-NEXT:    [[M2:%.*]] = mul i64 [[STRIDE]], [[I3]]
+; CHECK-NEXT:    [[M3:%.*]] = mul i64 [[STRIDE]], [[IV_NEXT]]
+; CHECK-NEXT:    [[A0:%.*]] = add i64 [[OFF]], [[M0]]
+; CHECK-NEXT:    [[A1:%.*]] = add i64 [[OFF]], [[M1]]
+; CHECK-NEXT:    [[A2:%.*]] = add i64 [[OFF]], [[M2]]
+; CHECK-NEXT:    [[A3:%.*]] = add i64 [[OFF]], [[M3]]
+; CHECK-NEXT:    [[S0:%.*]] = shl i64 [[A0]], 2
+; CHECK-NEXT:    [[S1:%.*]] = shl i64 [[A1]], 2
+; CHECK-NEXT:    [[S2:%.*]] = shl i64 [[A2]], 2
+; CHECK-NEXT:    [[S3:%.*]] = shl i64 [[A3]], 2
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S0]]
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S1]]
+; CHECK-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S2]]
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[S3]]
 ; CHECK-NEXT:    [[L0:%.*]] = load i32, ptr [[P0]], align 4
 ; CHECK-NEXT:    [[U0:%.*]] = add i32 [[L0]], 1
 ; CHECK-NEXT:    store i32 [[U0]], ptr [[P0]], align 4
@@ -560,37 +563,23 @@ define i32 @preheader_offsets(ptr %base, ptr %offs, i64 %n) {
 ; AVX2-LABEL: define i32 @preheader_offsets(
 ; AVX2-SAME: ptr [[BASE:%.*]], ptr [[OFFS:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
 ; AVX2-NEXT:  [[ENTRY:.*]]:
-; AVX2-NEXT:    [[X1_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 8
-; AVX2-NEXT:    [[X2_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 16
-; AVX2-NEXT:    [[X3_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 24
-; AVX2-NEXT:    [[X0:%.*]] = load i64, ptr [[OFFS]], align 8
-; AVX2-NEXT:    [[X1:%.*]] = load i64, ptr [[X1_PTR]], align 8
-; AVX2-NEXT:    [[X2:%.*]] = load i64, ptr [[X2_PTR]], align 8
-; AVX2-NEXT:    [[X3:%.*]] = load i64, ptr [[X3_PTR]], align 8
 ; AVX2-NEXT:    [[Y0_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 32
-; AVX2-NEXT:    [[Y1_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 40
-; AVX2-NEXT:    [[Y2_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 48
-; AVX2-NEXT:    [[Y3_PTR:%.*]] = getelementptr i8, ptr [[OFFS]], i64 56
-; AVX2-NEXT:    [[Y0:%.*]] = load i64, ptr [[Y0_PTR]], align 8
-; AVX2-NEXT:    [[Y1:%.*]] = load i64, ptr [[Y1_PTR]], align 8
-; AVX2-NEXT:    [[Y2:%.*]] = load i64, ptr [[Y2_PTR]], align 8
-; AVX2-NEXT:    [[Y3:%.*]] = load i64, ptr [[Y3_PTR]], align 8
-; AVX2-NEXT:    [[T0:%.*]] = add i64 [[X0]], [[Y0]]
-; AVX2-NEXT:    [[T1:%.*]] = add i64 [[X1]], [[Y1]]
-; AVX2-NEXT:    [[T2:%.*]] = add i64 [[X2]], [[Y2]]
-; AVX2-NEXT:    [[T3:%.*]] = add i64 [[X3]], [[Y3]]
-; AVX2-NEXT:    [[O0:%.*]] = shl i64 [[T0]], 2
-; AVX2-NEXT:    [[O1:%.*]] = shl i64 [[T1]], 2
-; AVX2-NEXT:    [[O2:%.*]] = shl i64 [[T2]], 2
-; AVX2-NEXT:    [[O3:%.*]] = shl i64 [[T3]], 2
+; AVX2-NEXT:    [[TMP0:%.*]] = load <4 x i64>, ptr [[OFFS]], align 8
+; AVX2-NEXT:    [[TMP1:%.*]] = load <4 x i64>, ptr [[Y0_PTR]], align 8
+; AVX2-NEXT:    [[TMP2:%.*]] = add <4 x i64> [[TMP0]], [[TMP1]]
+; AVX2-NEXT:    [[TMP3:%.*]] = shl <4 x i64> [[TMP2]], splat (i64 2)
+; AVX2-NEXT:    [[TMP4:%.*]] = extractelement <4 x i64> [[TMP3]], i64 0
+; AVX2-NEXT:    [[TMP5:%.*]] = extractelement <4 x i64> [[TMP3]], i64 1
+; AVX2-NEXT:    [[TMP6:%.*]] = extractelement <4 x i64> [[TMP3]], i64 2
+; AVX2-NEXT:    [[TMP7:%.*]] = extractelement <4 x i64> [[TMP3]], i64 3
 ; AVX2-NEXT:    br label %[[LOOP:.*]]
 ; AVX2:       [[LOOP]]:
 ; AVX2-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
 ; AVX2-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
-; AVX2-NEXT:    [[A0:%.*]] = add i64 [[IV]], [[O0]]
-; AVX2-NEXT:    [[A1:%.*]] = add i64 [[IV]], [[O1]]
-; AVX2-NEXT:    [[A2:%.*]] = add i64 [[IV]], [[O2]]
-; AVX2-NEXT:    [[A3:%.*]] = add i64 [[IV]], [[O3]]
+; AVX2-NEXT:    [[A0:%.*]] = add i64 [[IV]], [[TMP4]]
+; AVX2-NEXT:    [[A1:%.*]] = add i64 [[IV]], [[TMP5]]
+; AVX2-NEXT:    [[A2:%.*]] = add i64 [[IV]], [[TMP6]]
+; AVX2-NEXT:    [[A3:%.*]] = add i64 [[IV]], [[TMP7]]
 ; AVX2-NEXT:    [[P0:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[A0]]
 ; AVX2-NEXT:    [[P1:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[A1]]
 ; AVX2-NEXT:    [[P2:%.*]] = getelementptr i8, ptr [[BASE]], i64 [[A2]]

>From 830935b79c40896169a09430e1ad422285e89b35 Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Thu, 24 Sep 2026 20:40:14 +0200
Subject: [PATCH 3/5] Address review comments

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 86 ++++---------------
 .../Vectorize/SLPVectorizer/SLPUtils.cpp      | 86 +++++++++++++++++++
 .../Vectorize/SLPVectorizer/SLPUtils.h        | 17 ++++
 3 files changed, 121 insertions(+), 68 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index fb000ef645ffe..30b175333841f 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -29106,6 +29106,9 @@ bool SLPVectorizerPass::runImpl(Function &F, ScalarEvolution *SE_,
           R.isScalarFallbackBlock(BB))
         continue;
       R.clearReductionData();
+      // GEPs is collected per block and is also used to keep its index
+      // computations out of the standalone-seed attempt.
+      collectSeedInstructions(BB);
       Changed |= vectorizeOnceUsedSeeds(BB, R);
     }
   }
@@ -30227,71 +30230,6 @@ void SLPVectorizerPass::collectSeedInstructions(BasicBlock *BB) {
   }
 }
 
-/// Returns true if \p Ptr is only used by scalar loads and stores, directly or
-/// through getelementptrs with constant offsets.
-static bool onlyFeedsScalarAccesses(const Value *Ptr, unsigned Depth = 0) {
-  constexpr unsigned MaxDepth = 2;
-  return !Ptr->use_empty() && all_of(Ptr->users(), [&](const User *U) {
-    if (const auto *Load = dyn_cast<LoadInst>(U))
-      return !Load->getType()->isVectorTy();
-    if (const auto *Store = dyn_cast<StoreInst>(U))
-      return Store->getPointerOperand() == Ptr &&
-             !Store->getValueOperand()->getType()->isVectorTy();
-    if (const auto *GEP = dyn_cast<GetElementPtrInst>(U))
-      return Depth < MaxDepth && GEP->getPointerOperand() == Ptr &&
-             GEP->hasAllConstantIndices() &&
-             onlyFeedsScalarAccesses(GEP, Depth + 1);
-    return false;
-  });
-}
-
-/// Leave related affine address recurrences feeding scalar accesses for loop
-/// strength reduction. Vectorizing their indices can retain expensive
-/// arithmetic and require an extract for each lane instead of a scalar pointer
-/// increment.
-static bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL,
-                                           ScalarEvolution &SE, LoopInfo &LI) {
-  constexpr unsigned MaxIndexChainLength = 3;
-  const Loop *L = nullptr;
-  SmallVector<GetElementPtrInst *> GEPs;
-  for (Value *V : VL) {
-    Value *Cur = V;
-    GetElementPtrInst *GEP = nullptr;
-    for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
-      if (!isa<BinaryOperator, CastInst>(Cur) || !Cur->hasOneUse())
-        return false;
-      User *U = Cur->user_back();
-      if ((GEP = dyn_cast<GetElementPtrInst>(U)))
-        break;
-      Cur = U;
-    }
-    if (!GEP || GEP->getPointerOperand() == Cur ||
-        !onlyFeedsScalarAccesses(GEP))
-      return false;
-    // LSR only removes the arithmetic computed in the loop itself.
-    const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
-    if (!GEPLoop || (L && L != GEPLoop) ||
-        LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
-      return false;
-    L = GEPLoop;
-    GEPs.push_back(GEP);
-  }
-  const SCEV *FirstAddr = nullptr;
-  for (GetElementPtrInst *GEP : GEPs) {
-    const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
-    if (!Addr || Addr->getLoop() != L || !Addr->isAffine())
-      return false;
-    if (!FirstAddr) {
-      FirstAddr = Addr;
-      continue;
-    }
-    const SCEV *Diff = SE.getMinusSCEV(Addr, FirstAddr);
-    if (isa<SCEVCouldNotCompute>(Diff) || !SE.isLoopInvariant(Diff, L))
-      return false;
-  }
-  return true;
-}
-
 bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
                                            bool MaxVFOnly,
                                            bool StandaloneSeeds) {
@@ -30423,9 +30361,15 @@ bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
         continue;
       }
 
-      if (isStrengthReducibleIndexBundle(Ops, *SE, *LI)) {
-        LLVM_DEBUG(dbgs() << "SLP: Not vectorizing strength-reducible address "
-                             "computations.\n");
+      auto IsCollectedGEP = [&](GetElementPtrInst *GEP) {
+        auto It = GEPs.find(GEP->getPointerOperand());
+        return It != GEPs.end() && It->second.size() >= 2 &&
+               is_contained(It->second, GEP);
+      };
+      if (StandaloneSeeds &&
+          isGEPCandidateIndexBundle(Ops, *LI, SLPReVec, IsCollectedGEP)) {
+        LLVM_DEBUG(dbgs() << "SLP: Leaving collected GEP index computations "
+                             "to vectorizeGEPIndices.\n");
         continue;
       }
 
@@ -36259,6 +36203,12 @@ bool SLPVectorizerPass::vectorizeGEPIndices(BasicBlock *BB, BoUpSLP &R) {
         Bundle[BundleIndex++] = GEPIdx;
       }
 
+      if (isStrengthReducibleIndexBundle(Bundle, *SE, *LI, SLPReVec)) {
+        LLVM_DEBUG(dbgs() << "SLP: Not vectorizing strength-reducible address "
+                             "computations.\n");
+        continue;
+      }
+
       // Try and vectorize the indices. We are currently only interested in
       // gather-like cases of the form:
       //
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index d639e0d9c7d63..e3c4f4e2dbfd9 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -7,11 +7,15 @@
 //===----------------------------------------------------------------------===//
 
 #include "SLPUtils.h"
+#include "SLPTypeUtils.h"
 
 #include "llvm/ADT/APInt.h"
 #include "llvm/ADT/STLExtras.h"
 #include "llvm/ADT/Sequence.h"
 #include "llvm/Analysis/AssumptionCache.h"
+#include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/ScalarEvolution.h"
+#include "llvm/Analysis/ScalarEvolutionExpressions.h"
 #include "llvm/Analysis/ValueTracking.h"
 #include "llvm/Analysis/VectorUtils.h"
 #include "llvm/IR/Constants.h"
@@ -1003,6 +1007,88 @@ bool isOnceUsedSeed(const Instruction *I) {
       I);
 }
 
+/// Returns true if \p Ptr is only used by scalar loads and stores, directly or
+/// through getelementptrs with constant offsets.
+static bool onlyFeedsScalarAccesses(Value *Ptr, bool ReVec,
+                                    unsigned Depth = 0) {
+  constexpr unsigned MaxDepth = 2;
+  if (Ptr->use_empty() || Ptr->hasNUsesOrMore(UsesLimit))
+    return false;
+  return all_of(Ptr->users(), [&](User *U) {
+    if (isa<LoadInst>(U))
+      return !getValueType(U, ReVec)->isVectorTy();
+    if (auto *Store = dyn_cast<StoreInst>(U))
+      return Store->getPointerOperand() == Ptr &&
+             !getValueType(Store, ReVec)->isVectorTy();
+    if (auto *GEP = dyn_cast<GetElementPtrInst>(U))
+      return Depth < MaxDepth && GEP->getPointerOperand() == Ptr &&
+             GEP->hasAllConstantIndices() &&
+             onlyFeedsScalarAccesses(GEP, ReVec, Depth + 1);
+    return false;
+  });
+}
+
+static bool collectScalarAccessGEPs(ArrayRef<Value *> VL, LoopInfo &LI,
+                                    bool ReVec,
+                                    SmallVectorImpl<GetElementPtrInst *> &GEPs,
+                                    const Loop *&L) {
+  constexpr unsigned MaxIndexChainLength = 3;
+  for (Value *V : VL) {
+    Value *Cur = V;
+    GetElementPtrInst *GEP = nullptr;
+    for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
+      if (!isa<BinaryOperator, CastInst>(Cur) || !Cur->hasOneUse())
+        return false;
+      User *U = Cur->user_back();
+      if ((GEP = dyn_cast<GetElementPtrInst>(U)))
+        break;
+      Cur = U;
+    }
+    if (!GEP || GEP->getPointerOperand() == Cur ||
+        !onlyFeedsScalarAccesses(GEP, ReVec))
+      return false;
+    // LSR only removes the arithmetic computed in the loop itself.
+    const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
+    if (!GEPLoop || (L && L != GEPLoop) ||
+        LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
+      return false;
+    L = GEPLoop;
+    GEPs.push_back(GEP);
+  }
+  return true;
+}
+
+bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
+                                    LoopInfo &LI, bool ReVec) {
+  const Loop *L = nullptr;
+  SmallVector<GetElementPtrInst *> GEPs;
+  if (!collectScalarAccessGEPs(VL, LI, ReVec, GEPs, L))
+    return false;
+  const SCEV *FirstAddr = nullptr;
+  for (GetElementPtrInst *GEP : GEPs) {
+    const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
+    if (!Addr || Addr->getLoop() != L || !Addr->isAffine())
+      return false;
+    if (!FirstAddr) {
+      FirstAddr = Addr;
+      continue;
+    }
+    const SCEV *Diff = SE.getMinusSCEV(Addr, FirstAddr);
+    if (isa<SCEVCouldNotCompute>(Diff) || !SE.isLoopInvariant(Diff, L))
+      return false;
+  }
+  return true;
+}
+
+bool isGEPCandidateIndexBundle(
+    ArrayRef<Value *> VL, LoopInfo &LI, bool ReVec,
+    function_ref<bool(GetElementPtrInst *)> IsCandidate) {
+  const Loop *L = nullptr;
+  SmallVector<GetElementPtrInst *> GEPs;
+  return collectScalarAccessGEPs(VL, LI, ReVec, GEPs, L) &&
+         all_of(GEPs, IsCandidate);
+}
+
 Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable) {
   auto *Wide = dyn_cast<FPExtInst>(V);
   if (!Wide || !Wide->hasOneUse())
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index 4ac6880d0f17d..06e38eec92d0d 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -36,8 +36,11 @@ namespace llvm {
 class AssumptionCache;
 class Constant;
 class DataLayout;
+class GetElementPtrInst;
 class Instruction;
 class IRBuilderBase;
+class LoopInfo;
+class ScalarEvolution;
 class TargetLibraryInfo;
 class Type;
 class Value;
@@ -379,6 +382,20 @@ Intrinsic::ID getMaskedDivRemIntrinsic(unsigned Opcode);
 /// dedicated attempt.
 bool isOnceUsedSeed(const Instruction *I);
 
+/// Leave related affine address recurrences feeding scalar accesses for loop
+/// strength reduction. Vectorizing their indices can retain expensive
+/// arithmetic and require an extract for each lane instead of a scalar pointer
+/// increment.
+bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
+                                    LoopInfo &LI, bool ReVec);
+
+/// Returns true if each value in \p VL is an in-loop index computation ending
+/// at a getelementptr accepted by \p IsCandidate and used only by scalar
+/// accesses.
+bool isGEPCandidateIndexBundle(
+    ArrayRef<Value *> VL, LoopInfo &LI, bool ReVec,
+    function_ref<bool(GetElementPtrInst *)> IsCandidate);
+
 /// If \p V is a single-use fpext of a single-use fptrunc forming a round-trip
 /// back to the type of \p V, returns the fptrunc; the round-trip source is its
 /// operand, always an instruction of the same type as \p V. If

>From 0c76e042d0ac699c73e7c23a35727bd72381b729 Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Fri, 25 Sep 2026 09:25:15 +0200
Subject: [PATCH 4/5] Address review comments (round 2)

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 19 +++----
 .../Vectorize/SLPVectorizer/SLPUtils.cpp      | 57 ++++++++++++-------
 .../Vectorize/SLPVectorizer/SLPUtils.h        |  8 +--
 3 files changed, 46 insertions(+), 38 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 30b175333841f..e3b0f4b556d84 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -30361,18 +30361,6 @@ bool SLPVectorizerPass::tryToVectorizeList(ArrayRef<Value *> VL, BoUpSLP &R,
         continue;
       }
 
-      auto IsCollectedGEP = [&](GetElementPtrInst *GEP) {
-        auto It = GEPs.find(GEP->getPointerOperand());
-        return It != GEPs.end() && It->second.size() >= 2 &&
-               is_contained(It->second, GEP);
-      };
-      if (StandaloneSeeds &&
-          isGEPCandidateIndexBundle(Ops, *LI, SLPReVec, IsCollectedGEP)) {
-        LLVM_DEBUG(dbgs() << "SLP: Leaving collected GEP index computations "
-                             "to vectorizeGEPIndices.\n");
-        continue;
-      }
-
       LLVM_DEBUG(dbgs() << "SLP: Analyzing " << ActualVF << " operations "
                         << "\n");
 
@@ -36085,6 +36073,13 @@ bool SLPVectorizerPass::vectorizeOnceUsedSeeds(BasicBlock *BB, BoUpSLP &R) {
         !isOnceUsedSeed(&I) || isNonVectorizableInst(&I, TLI) ||
         R.hasResolvedUser(&I))
       continue;
+    // Index chains of collected GEPs are handled by vectorizeGEPIndices.
+    if (isGEPCandidateIndexBundle({&I}, *SE, *LI, SLPReVec, [&](auto *GEP) {
+          auto It = GEPs.find(GEP->getPointerOperand());
+          return It != GEPs.end() && It->second.size() >= 2 &&
+                 is_contained(It->second, GEP);
+        }))
+      continue;
     // The poor-throughput ops are seeded on their own, with the different
     // grouping.
     if (VectorizePoorThroughput &&
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index e3c4f4e2dbfd9..7c552d35e6497 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -1015,11 +1015,12 @@ static bool onlyFeedsScalarAccesses(Value *Ptr, bool ReVec,
   if (Ptr->use_empty() || Ptr->hasNUsesOrMore(UsesLimit))
     return false;
   return all_of(Ptr->users(), [&](User *U) {
-    if (isa<LoadInst>(U))
-      return !getValueType(U, ReVec)->isVectorTy();
-    if (auto *Store = dyn_cast<StoreInst>(U))
-      return Store->getPointerOperand() == Ptr &&
-             !getValueType(Store, ReVec)->isVectorTy();
+    if (getValueType(U, ReVec)->isVectorTy())
+      return false;
+    Value *Op;
+    if (isa<LoadInst>(U) ||
+        (match(U, m_Store(m_Value(Op), m_Specific(Ptr))) && Op != Ptr))
+      return true;
     if (auto *GEP = dyn_cast<GetElementPtrInst>(U))
       return Depth < MaxDepth && GEP->getPointerOperand() == Ptr &&
              GEP->hasAllConstantIndices() &&
@@ -1028,17 +1029,20 @@ static bool onlyFeedsScalarAccesses(Value *Ptr, bool ReVec,
   });
 }
 
-static bool collectScalarAccessGEPs(ArrayRef<Value *> VL, LoopInfo &LI,
-                                    bool ReVec,
-                                    SmallVectorImpl<GetElementPtrInst *> &GEPs,
-                                    const Loop *&L) {
+/// Collects in \p GEPs the getelementptrs that the in-loop index computations
+/// \p VL end at, if they are only used by scalar accesses. Returns the loop
+/// containing all of them, or nullptr otherwise.
+static const Loop *
+collectScalarAccessGEPs(ArrayRef<Value *> VL, const LoopInfo &LI, bool ReVec,
+                        SmallVectorImpl<GetElementPtrInst *> &GEPs) {
   constexpr unsigned MaxIndexChainLength = 3;
+  const Loop *L = nullptr;
   for (Value *V : VL) {
     Value *Cur = V;
     GetElementPtrInst *GEP = nullptr;
     for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
       if (!isa<BinaryOperator, CastInst>(Cur) || !Cur->hasOneUse())
-        return false;
+        return nullptr;
       User *U = Cur->user_back();
       if ((GEP = dyn_cast<GetElementPtrInst>(U)))
         break;
@@ -1046,28 +1050,36 @@ static bool collectScalarAccessGEPs(ArrayRef<Value *> VL, LoopInfo &LI,
     }
     if (!GEP || GEP->getPointerOperand() == Cur ||
         !onlyFeedsScalarAccesses(GEP, ReVec))
-      return false;
+      return nullptr;
     // LSR only removes the arithmetic computed in the loop itself.
     const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
     if (!GEPLoop || (L && L != GEPLoop) ||
         LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
-      return false;
+      return nullptr;
     L = GEPLoop;
     GEPs.push_back(GEP);
   }
-  return true;
+  return L;
+}
+
+/// Returns the address computed by \p GEP if it is an affine recurrence of
+/// \p L, or nullptr otherwise.
+static const SCEVAddRecExpr *
+getAffineAddress(GetElementPtrInst *GEP, const Loop *L, ScalarEvolution &SE) {
+  const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
+  return Addr && Addr->getLoop() == L && Addr->isAffine() ? Addr : nullptr;
 }
 
 bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
-                                    LoopInfo &LI, bool ReVec) {
-  const Loop *L = nullptr;
+                                    const LoopInfo &LI, bool ReVec) {
   SmallVector<GetElementPtrInst *> GEPs;
-  if (!collectScalarAccessGEPs(VL, LI, ReVec, GEPs, L))
+  const Loop *L = collectScalarAccessGEPs(VL, LI, ReVec, GEPs);
+  if (!L)
     return false;
   const SCEV *FirstAddr = nullptr;
   for (GetElementPtrInst *GEP : GEPs) {
-    const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
-    if (!Addr || Addr->getLoop() != L || !Addr->isAffine())
+    const SCEV *Addr = getAffineAddress(GEP, L, SE);
+    if (!Addr)
       return false;
     if (!FirstAddr) {
       FirstAddr = Addr;
@@ -1081,12 +1093,13 @@ bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
 }
 
 bool isGEPCandidateIndexBundle(
-    ArrayRef<Value *> VL, LoopInfo &LI, bool ReVec,
+    ArrayRef<Value *> VL, ScalarEvolution &SE, const LoopInfo &LI, bool ReVec,
     function_ref<bool(GetElementPtrInst *)> IsCandidate) {
-  const Loop *L = nullptr;
   SmallVector<GetElementPtrInst *> GEPs;
-  return collectScalarAccessGEPs(VL, LI, ReVec, GEPs, L) &&
-         all_of(GEPs, IsCandidate);
+  const Loop *L = collectScalarAccessGEPs(VL, LI, ReVec, GEPs);
+  return L && all_of(GEPs, [&](GetElementPtrInst *GEP) {
+           return IsCandidate(GEP) && getAffineAddress(GEP, L, SE);
+         });
 }
 
 Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable) {
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index 06e38eec92d0d..f578230bfe292 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -387,13 +387,13 @@ bool isOnceUsedSeed(const Instruction *I);
 /// arithmetic and require an extract for each lane instead of a scalar pointer
 /// increment.
 bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
-                                    LoopInfo &LI, bool ReVec);
+                                    const LoopInfo &LI, bool ReVec);
 
 /// Returns true if each value in \p VL is an in-loop index computation ending
-/// at a getelementptr accepted by \p IsCandidate and used only by scalar
-/// accesses.
+/// at a getelementptr accepted by \p IsCandidate, used only by scalar accesses
+/// and computing an affine recurrence of that loop.
 bool isGEPCandidateIndexBundle(
-    ArrayRef<Value *> VL, LoopInfo &LI, bool ReVec,
+    ArrayRef<Value *> VL, ScalarEvolution &SE, const LoopInfo &LI, bool ReVec,
     function_ref<bool(GetElementPtrInst *)> IsCandidate);
 
 /// If \p V is a single-use fpext of a single-use fptrunc forming a round-trip

>From a58d5a97f308af7b7d8aee2087c0819b89b961d1 Mon Sep 17 00:00:00 2001
From: Tim Besard <tim.besard at gmail.com>
Date: Mon, 28 Sep 2026 09:38:22 +0200
Subject: [PATCH 5/5] Address review comments (round 3)

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    |  2 +-
 .../Vectorize/SLPVectorizer/SLPUtils.cpp      | 87 +++++++------------
 .../Vectorize/SLPVectorizer/SLPUtils.h        | 10 +--
 3 files changed, 36 insertions(+), 63 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index e3b0f4b556d84..05023efdac91f 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -36074,7 +36074,7 @@ bool SLPVectorizerPass::vectorizeOnceUsedSeeds(BasicBlock *BB, BoUpSLP &R) {
         R.hasResolvedUser(&I))
       continue;
     // Index chains of collected GEPs are handled by vectorizeGEPIndices.
-    if (isGEPCandidateIndexBundle({&I}, *SE, *LI, SLPReVec, [&](auto *GEP) {
+    if (!GEPs.empty() && isGEPCandidateIndex(&I, [&](auto *GEP) {
           auto It = GEPs.find(GEP->getPointerOperand());
           return It != GEPs.end() && It->second.size() >= 2 &&
                  is_contained(It->second, GEP);
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index 7c552d35e6497..a74ae8c113d95 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -1008,10 +1008,9 @@ bool isOnceUsedSeed(const Instruction *I) {
 }
 
 /// Returns true if \p Ptr is only used by scalar loads and stores, directly or
-/// through getelementptrs with constant offsets.
+/// through at most \p Depth levels of getelementptrs with constant offsets.
 static bool onlyFeedsScalarAccesses(Value *Ptr, bool ReVec,
-                                    unsigned Depth = 0) {
-  constexpr unsigned MaxDepth = 2;
+                                    unsigned Depth = 2) {
   if (Ptr->use_empty() || Ptr->hasNUsesOrMore(UsesLimit))
     return false;
   return all_of(Ptr->users(), [&](User *U) {
@@ -1022,64 +1021,44 @@ static bool onlyFeedsScalarAccesses(Value *Ptr, bool ReVec,
         (match(U, m_Store(m_Value(Op), m_Specific(Ptr))) && Op != Ptr))
       return true;
     if (auto *GEP = dyn_cast<GetElementPtrInst>(U))
-      return Depth < MaxDepth && GEP->getPointerOperand() == Ptr &&
+      return Depth > 0 && GEP->getPointerOperand() == Ptr &&
              GEP->hasAllConstantIndices() &&
-             onlyFeedsScalarAccesses(GEP, ReVec, Depth + 1);
+             onlyFeedsScalarAccesses(GEP, ReVec, Depth - 1);
     return false;
   });
 }
 
-/// Collects in \p GEPs the getelementptrs that the in-loop index computations
-/// \p VL end at, if they are only used by scalar accesses. Returns the loop
-/// containing all of them, or nullptr otherwise.
-static const Loop *
-collectScalarAccessGEPs(ArrayRef<Value *> VL, const LoopInfo &LI, bool ReVec,
-                        SmallVectorImpl<GetElementPtrInst *> &GEPs) {
+/// Returns the getelementptr whose index is computed by the short chain of
+/// single-use arithmetic starting at \p V, or nullptr otherwise.
+static GetElementPtrInst *getIndexChainGEP(Value *V) {
   constexpr unsigned MaxIndexChainLength = 3;
-  const Loop *L = nullptr;
-  for (Value *V : VL) {
-    Value *Cur = V;
-    GetElementPtrInst *GEP = nullptr;
-    for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
-      if (!isa<BinaryOperator, CastInst>(Cur) || !Cur->hasOneUse())
-        return nullptr;
-      User *U = Cur->user_back();
-      if ((GEP = dyn_cast<GetElementPtrInst>(U)))
-        break;
-      Cur = U;
-    }
-    if (!GEP || GEP->getPointerOperand() == Cur ||
-        !onlyFeedsScalarAccesses(GEP, ReVec))
-      return nullptr;
-    // LSR only removes the arithmetic computed in the loop itself.
-    const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
-    if (!GEPLoop || (L && L != GEPLoop) ||
-        LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
+  for ([[maybe_unused]] unsigned _ : seq<unsigned>(MaxIndexChainLength)) {
+    if (!isa<BinaryOperator, CastInst>(V) || !V->hasOneUse())
       return nullptr;
-    L = GEPLoop;
-    GEPs.push_back(GEP);
+    User *U = V->user_back();
+    if (auto *GEP = dyn_cast<GetElementPtrInst>(U))
+      return GEP->getPointerOperand() != V ? GEP : nullptr;
+    V = U;
   }
-  return L;
-}
-
-/// Returns the address computed by \p GEP if it is an affine recurrence of
-/// \p L, or nullptr otherwise.
-static const SCEVAddRecExpr *
-getAffineAddress(GetElementPtrInst *GEP, const Loop *L, ScalarEvolution &SE) {
-  const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
-  return Addr && Addr->getLoop() == L && Addr->isAffine() ? Addr : nullptr;
+  return nullptr;
 }
 
 bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
                                     const LoopInfo &LI, bool ReVec) {
-  SmallVector<GetElementPtrInst *> GEPs;
-  const Loop *L = collectScalarAccessGEPs(VL, LI, ReVec, GEPs);
-  if (!L)
-    return false;
+  const Loop *L = nullptr;
   const SCEV *FirstAddr = nullptr;
-  for (GetElementPtrInst *GEP : GEPs) {
-    const SCEV *Addr = getAffineAddress(GEP, L, SE);
-    if (!Addr)
+  for (Value *V : VL) {
+    GetElementPtrInst *GEP = getIndexChainGEP(V);
+    if (!GEP || !onlyFeedsScalarAccesses(GEP, ReVec))
+      return false;
+    // LSR only removes the arithmetic computed in the loop itself.
+    const Loop *GEPLoop = LI.getLoopFor(GEP->getParent());
+    if (!GEPLoop || (L && L != GEPLoop) ||
+        LI.getLoopFor(cast<Instruction>(V)->getParent()) != GEPLoop)
+      return false;
+    L = GEPLoop;
+    const auto *Addr = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(GEP));
+    if (!Addr || Addr->getLoop() != L || !Addr->isAffine())
       return false;
     if (!FirstAddr) {
       FirstAddr = Addr;
@@ -1092,14 +1071,10 @@ bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
   return true;
 }
 
-bool isGEPCandidateIndexBundle(
-    ArrayRef<Value *> VL, ScalarEvolution &SE, const LoopInfo &LI, bool ReVec,
-    function_ref<bool(GetElementPtrInst *)> IsCandidate) {
-  SmallVector<GetElementPtrInst *> GEPs;
-  const Loop *L = collectScalarAccessGEPs(VL, LI, ReVec, GEPs);
-  return L && all_of(GEPs, [&](GetElementPtrInst *GEP) {
-           return IsCandidate(GEP) && getAffineAddress(GEP, L, SE);
-         });
+bool isGEPCandidateIndex(Instruction *I,
+                         function_ref<bool(GetElementPtrInst *)> IsCandidate) {
+  GetElementPtrInst *GEP = getIndexChainGEP(I);
+  return GEP && IsCandidate(GEP);
 }
 
 Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable) {
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index f578230bfe292..91bcc75ef4c0b 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -389,12 +389,10 @@ bool isOnceUsedSeed(const Instruction *I);
 bool isStrengthReducibleIndexBundle(ArrayRef<Value *> VL, ScalarEvolution &SE,
                                     const LoopInfo &LI, bool ReVec);
 
-/// Returns true if each value in \p VL is an in-loop index computation ending
-/// at a getelementptr accepted by \p IsCandidate, used only by scalar accesses
-/// and computing an affine recurrence of that loop.
-bool isGEPCandidateIndexBundle(
-    ArrayRef<Value *> VL, ScalarEvolution &SE, const LoopInfo &LI, bool ReVec,
-    function_ref<bool(GetElementPtrInst *)> IsCandidate);
+/// Returns true if \p I starts a short chain of single-use arithmetic that
+/// computes the index of a getelementptr accepted by \p IsCandidate.
+bool isGEPCandidateIndex(Instruction *I,
+                         function_ref<bool(GetElementPtrInst *)> IsCandidate);
 
 /// If \p V is a single-use fpext of a single-use fptrunc forming a round-trip
 /// back to the type of \p V, returns the fptrunc; the round-trip source is its



More information about the llvm-commits mailing list