[llvm] [SLP][NFC]Add a test with non-profitable vectorization for GEPs external users, NFC (PR #216518)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Sat Aug 15 16:10:12 PDT 2026
https://github.com/alexey-bataev created https://github.com/llvm/llvm-project/pull/216518
None
>From 94f5932bfd0f18f107c8d7876d4e8aecd71ccd3a Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sat, 15 Aug 2026 16:10:00 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../reduction-vals-used-as-load-indices.ll | 202 ++++++++++++++++++
1 file changed, 202 insertions(+)
create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/reduction-vals-used-as-load-indices.ll
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reduction-vals-used-as-load-indices.ll b/llvm/test/Transforms/SLPVectorizer/X86/reduction-vals-used-as-load-indices.ll
new file mode 100644
index 0000000000000..dcb93453b5520
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reduction-vals-used-as-load-indices.ll
@@ -0,0 +1,202 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -S -passes=slp-vectorizer -mcpu=znver2 -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s
+
+define i32 @test(ptr %this, i32 %a, i32 %b) {
+; CHECK-LABEL: define i32 @test(
+; CHECK-SAME: ptr [[THIS:%.*]], i32 [[A:%.*]], i32 [[B:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ARRAY_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[THIS]], i64 24
+; CHECK-NEXT: [[TMP0:%.*]] = load ptr, ptr [[ARRAY_PTR]], align 8
+; CHECK-NEXT: [[TOBOOL_NOT:%.*]] = icmp eq ptr [[TMP0]], null
+; CHECK-NEXT: br i1 [[TOBOOL_NOT]], label %[[IF_THEN:.*]], label %[[IF_END:.*]]
+; CHECK: [[IF_THEN]]:
+; CHECK-NEXT: tail call void @deopt()
+; CHECK-NEXT: unreachable
+; CHECK: [[IF_END]]:
+; CHECK-NEXT: [[LENGTH:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 8
+; CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr [[LENGTH]], align 8
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[B]], [[A]]
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <8 x i32> poison, i32 [[ADD]], i64 0
+; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP2]], <8 x i32> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP4:%.*]] = lshr <8 x i32> [[TMP3]], <i32 0, i32 4, i32 8, i32 12, i32 16, i32 20, i32 24, i32 0>
+; CHECK-NEXT: [[TMP5:%.*]] = and <8 x i32> [[TMP4]], <i32 15, i32 15, i32 15, i32 15, i32 15, i32 15, i32 15, i32 28>
+; CHECK-NEXT: [[TMP6:%.*]] = lshr <8 x i32> [[TMP4]], <i32 15, i32 15, i32 15, i32 15, i32 15, i32 15, i32 15, i32 28>
+; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <8 x i32> [[TMP5]], <8 x i32> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 15>
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[ADD]], 15
+; CHECK-NEXT: [[TMP8:%.*]] = or disjoint <8 x i32> [[TMP7]], <i32 0, i32 16, i32 32, i32 48, i32 64, i32 80, i32 96, i32 112>
+; CHECK-NEXT: [[TMP9:%.*]] = insertelement <8 x i32> poison, i32 [[TMP1]], i64 0
+; CHECK-NEXT: [[TMP10:%.*]] = shufflevector <8 x i32> [[TMP9]], <8 x i32> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP11:%.*]] = icmp ult <8 x i32> [[TMP8]], [[TMP10]]
+; CHECK-NEXT: [[TMP12:%.*]] = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> [[TMP11]])
+; CHECK-NEXT: br i1 [[TMP12]], label %[[IF_END88:.*]], label %[[IF_THEN87:.*]]
+; CHECK: [[IF_THEN87]]:
+; CHECK-NEXT: tail call void @deopt()
+; CHECK-NEXT: unreachable
+; CHECK: [[IF_END88]]:
+; CHECK-NEXT: [[DATA:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 16
+; CHECK-NEXT: [[IDXPROM:%.*]] = zext nneg i32 [[AND]] to i64
+; CHECK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[DATA]], i64 [[IDXPROM]]
+; CHECK-NEXT: [[TMP13:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
+; CHECK-NEXT: [[CONV89:%.*]] = sext i8 [[TMP13]] to i32
+; CHECK-NEXT: [[TMP14:%.*]] = extractelement <8 x i32> [[TMP8]], i64 1
+; CHECK-NEXT: [[IDXPROM90:%.*]] = zext nneg i32 [[TMP14]] to i64
+; CHECK-NEXT: [[ARRAYIDX91:%.*]] = getelementptr inbounds nuw i8, ptr [[DATA]], i64 [[IDXPROM90]]
+; CHECK-NEXT: [[TMP15:%.*]] = load i8, ptr [[ARRAYIDX91]], align 1
+; CHECK-NEXT: [[CONV92:%.*]] = sext i8 [[TMP15]] to i32
+; CHECK-NEXT: [[SHL93:%.*]] = shl nsw i32 [[CONV92]], 4
+; CHECK-NEXT: [[ADD94:%.*]] = add nsw i32 [[SHL93]], [[CONV89]]
+; CHECK-NEXT: [[TMP16:%.*]] = extractelement <8 x i32> [[TMP8]], i64 2
+; CHECK-NEXT: [[IDXPROM95:%.*]] = zext nneg i32 [[TMP16]] to i64
+; CHECK-NEXT: [[ARRAYIDX96:%.*]] = getelementptr inbounds nuw i8, ptr [[DATA]], i64 [[IDXPROM95]]
+; CHECK-NEXT: [[TMP17:%.*]] = load i8, ptr [[ARRAYIDX96]], align 1
+; CHECK-NEXT: [[CONV97:%.*]] = sext i8 [[TMP17]] to i32
+; CHECK-NEXT: [[SHL98:%.*]] = shl nsw i32 [[CONV97]], 8
+; CHECK-NEXT: [[ADD99:%.*]] = add nsw i32 [[ADD94]], [[SHL98]]
+; CHECK-NEXT: [[TMP18:%.*]] = extractelement <8 x i32> [[TMP8]], i64 3
+; CHECK-NEXT: [[IDXPROM100:%.*]] = zext nneg i32 [[TMP18]] to i64
+; CHECK-NEXT: [[ARRAYIDX101:%.*]] = getelementptr inbounds nuw i8, ptr [[DATA]], i64 [[IDXPROM100]]
+; CHECK-NEXT: [[TMP19:%.*]] = load i8, ptr [[ARRAYIDX101]], align 1
+; CHECK-NEXT: [[CONV102:%.*]] = sext i8 [[TMP19]] to i32
+; CHECK-NEXT: [[SHL103:%.*]] = shl nsw i32 [[CONV102]], 12
+; CHECK-NEXT: [[ADD104:%.*]] = add nsw i32 [[ADD99]], [[SHL103]]
+; CHECK-NEXT: [[TMP20:%.*]] = extractelement <8 x i32> [[TMP8]], i64 4
+; CHECK-NEXT: [[IDXPROM105:%.*]] = zext nneg i32 [[TMP20]] to i64
+; CHECK-NEXT: [[ARRAYIDX106:%.*]] = getelementptr inbounds nuw i8, ptr [[DATA]], i64 [[IDXPROM105]]
+; CHECK-NEXT: [[TMP21:%.*]] = load i8, ptr [[ARRAYIDX106]], align 1
+; CHECK-NEXT: [[CONV107:%.*]] = sext i8 [[TMP21]] to i32
+; CHECK-NEXT: [[SHL108:%.*]] = shl nsw i32 [[CONV107]], 16
+; CHECK-NEXT: [[ADD109:%.*]] = add nsw i32 [[ADD104]], [[SHL108]]
+; CHECK-NEXT: [[TMP22:%.*]] = extractelement <8 x i32> [[TMP8]], i64 5
+; CHECK-NEXT: [[IDXPROM110:%.*]] = zext nneg i32 [[TMP22]] to i64
+; CHECK-NEXT: [[ARRAYIDX111:%.*]] = getelementptr inbounds nuw i8, ptr [[DATA]], i64 [[IDXPROM110]]
+; CHECK-NEXT: [[TMP23:%.*]] = load i8, ptr [[ARRAYIDX111]], align 1
+; CHECK-NEXT: [[CONV112:%.*]] = sext i8 [[TMP23]] to i32
+; CHECK-NEXT: [[SHL113:%.*]] = shl nsw i32 [[CONV112]], 20
+; CHECK-NEXT: [[ADD114:%.*]] = add nsw i32 [[ADD109]], [[SHL113]]
+; CHECK-NEXT: [[TMP24:%.*]] = extractelement <8 x i32> [[TMP8]], i64 6
+; CHECK-NEXT: [[IDXPROM115:%.*]] = zext nneg i32 [[TMP24]] to i64
+; CHECK-NEXT: [[ARRAYIDX116:%.*]] = getelementptr inbounds nuw i8, ptr [[DATA]], i64 [[IDXPROM115]]
+; CHECK-NEXT: [[TMP25:%.*]] = load i8, ptr [[ARRAYIDX116]], align 1
+; CHECK-NEXT: [[CONV117:%.*]] = sext i8 [[TMP25]] to i32
+; CHECK-NEXT: [[SHL118:%.*]] = shl nsw i32 [[CONV117]], 24
+; CHECK-NEXT: [[ADD119:%.*]] = add nsw i32 [[ADD114]], [[SHL118]]
+; CHECK-NEXT: [[TMP26:%.*]] = extractelement <8 x i32> [[TMP8]], i64 7
+; CHECK-NEXT: [[IDXPROM120:%.*]] = zext nneg i32 [[TMP26]] to i64
+; CHECK-NEXT: [[ARRAYIDX121:%.*]] = getelementptr inbounds nuw i8, ptr [[DATA]], i64 [[IDXPROM120]]
+; CHECK-NEXT: [[TMP27:%.*]] = load i8, ptr [[ARRAYIDX121]], align 1
+; CHECK-NEXT: [[CONV122:%.*]] = zext i8 [[TMP27]] to i32
+; CHECK-NEXT: [[SHL123:%.*]] = shl i32 [[CONV122]], 28
+; CHECK-NEXT: [[ADD124:%.*]] = add nsw i32 [[ADD119]], [[SHL123]]
+; CHECK-NEXT: [[OR127:%.*]] = tail call i32 @llvm.fshl.i32(i32 [[ADD124]], i32 [[ADD124]], i32 11)
+; CHECK-NEXT: ret i32 [[OR127]]
+;
+entry:
+ %array_ptr = getelementptr inbounds nuw i8, ptr %this, i64 24
+ %0 = load ptr, ptr %array_ptr, align 8
+ %tobool.not = icmp eq ptr %0, null
+ br i1 %tobool.not, label %if.then, label %if.end
+
+if.then:
+ tail call void @deopt()
+ unreachable
+
+if.end:
+ %add = add nsw i32 %b, %a
+ %length = getelementptr inbounds nuw i8, ptr %0, i64 8
+ %1 = load i32, ptr %length, align 8
+ %and = and i32 %add, 15
+ %shr = lshr i32 %add, 4
+ %and1 = and i32 %shr, 15
+ %or = or disjoint i32 %and1, 16
+ %shr2 = lshr i32 %add, 8
+ %and3 = and i32 %shr2, 15
+ %or4 = or disjoint i32 %and3, 32
+ %shr5 = lshr i32 %add, 12
+ %and6 = and i32 %shr5, 15
+ %or7 = or disjoint i32 %and6, 48
+ %shr8 = lshr i32 %add, 16
+ %and9 = and i32 %shr8, 15
+ %or10 = or disjoint i32 %and9, 64
+ %shr11 = lshr i32 %add, 20
+ %and12 = and i32 %shr11, 15
+ %or13 = or disjoint i32 %and12, 80
+ %shr14 = lshr i32 %add, 24
+ %and15 = and i32 %shr14, 15
+ %or16 = or disjoint i32 %and15, 96
+ %shr17 = lshr i32 %add, 28
+ %or18 = or disjoint i32 %shr17, 112
+ %cmp = icmp ult i32 %and, %1
+ %cmp19 = icmp ult i32 %or, %1
+ %and23167 = and i1 %cmp, %cmp19
+ %cmp26 = icmp ult i32 %or4, %1
+ %and33168 = and i1 %cmp26, %and23167
+ %cmp36 = icmp ult i32 %or7, %1
+ %and43169 = and i1 %cmp36, %and33168
+ %cmp46 = icmp ult i32 %or10, %1
+ %and53170 = and i1 %cmp46, %and43169
+ %cmp56 = icmp ult i32 %or13, %1
+ %and63171 = and i1 %cmp56, %and53170
+ %cmp66 = icmp ult i32 %or16, %1
+ %and73172 = and i1 %cmp66, %and63171
+ %cmp76 = icmp ult i32 %or18, %1
+ %and83173 = and i1 %cmp76, %and73172
+ br i1 %and83173, label %if.end88, label %if.then87
+
+if.then87:
+ tail call void @deopt()
+ unreachable
+
+if.end88:
+ %data = getelementptr inbounds nuw i8, ptr %0, i64 16
+ %idxprom = zext nneg i32 %and to i64
+ %arrayidx = getelementptr inbounds nuw i8, ptr %data, i64 %idxprom
+ %2 = load i8, ptr %arrayidx, align 1
+ %conv89 = sext i8 %2 to i32
+ %idxprom90 = zext nneg i32 %or to i64
+ %arrayidx91 = getelementptr inbounds nuw i8, ptr %data, i64 %idxprom90
+ %3 = load i8, ptr %arrayidx91, align 1
+ %conv92 = sext i8 %3 to i32
+ %shl93 = shl nsw i32 %conv92, 4
+ %add94 = add nsw i32 %shl93, %conv89
+ %idxprom95 = zext nneg i32 %or4 to i64
+ %arrayidx96 = getelementptr inbounds nuw i8, ptr %data, i64 %idxprom95
+ %4 = load i8, ptr %arrayidx96, align 1
+ %conv97 = sext i8 %4 to i32
+ %shl98 = shl nsw i32 %conv97, 8
+ %add99 = add nsw i32 %add94, %shl98
+ %idxprom100 = zext nneg i32 %or7 to i64
+ %arrayidx101 = getelementptr inbounds nuw i8, ptr %data, i64 %idxprom100
+ %5 = load i8, ptr %arrayidx101, align 1
+ %conv102 = sext i8 %5 to i32
+ %shl103 = shl nsw i32 %conv102, 12
+ %add104 = add nsw i32 %add99, %shl103
+ %idxprom105 = zext nneg i32 %or10 to i64
+ %arrayidx106 = getelementptr inbounds nuw i8, ptr %data, i64 %idxprom105
+ %6 = load i8, ptr %arrayidx106, align 1
+ %conv107 = sext i8 %6 to i32
+ %shl108 = shl nsw i32 %conv107, 16
+ %add109 = add nsw i32 %add104, %shl108
+ %idxprom110 = zext nneg i32 %or13 to i64
+ %arrayidx111 = getelementptr inbounds nuw i8, ptr %data, i64 %idxprom110
+ %7 = load i8, ptr %arrayidx111, align 1
+ %conv112 = sext i8 %7 to i32
+ %shl113 = shl nsw i32 %conv112, 20
+ %add114 = add nsw i32 %add109, %shl113
+ %idxprom115 = zext nneg i32 %or16 to i64
+ %arrayidx116 = getelementptr inbounds nuw i8, ptr %data, i64 %idxprom115
+ %8 = load i8, ptr %arrayidx116, align 1
+ %conv117 = sext i8 %8 to i32
+ %shl118 = shl nsw i32 %conv117, 24
+ %add119 = add nsw i32 %add114, %shl118
+ %idxprom120 = zext nneg i32 %or18 to i64
+ %arrayidx121 = getelementptr inbounds nuw i8, ptr %data, i64 %idxprom120
+ %9 = load i8, ptr %arrayidx121, align 1
+ %conv122 = zext i8 %9 to i32
+ %shl123 = shl i32 %conv122, 28
+ %add124 = add nsw i32 %add119, %shl123
+ %or127 = tail call i32 @llvm.fshl.i32(i32 %add124, i32 %add124, i32 11)
+ ret i32 %or127
+}
+
+declare void @deopt()
+declare i32 @llvm.fshl.i32(i32, i32, i32)
More information about the llvm-commits
mailing list