[llvm] [SLP][NFC]Add a test with the non-optimal vectorization, NFC (PR #228385)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 03:33:25 PDT 2026
https://github.com/alexey-bataev created https://github.com/llvm/llvm-project/pull/228385
None
>From 4d2c189453b051e23144c3da221d5da746dd3298 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Fri, 2 Oct 2026 03:33:10 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../X86/absorbing-copyable-lane.ll | 303 +++++++++++++
.../X86/wide-load-absorbed-lane.ll | 401 ++++++++++++++++++
2 files changed, 704 insertions(+)
create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll
create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/wide-load-absorbed-lane.ll
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll b/llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll
new file mode 100644
index 0000000000000..1d7933334c77d
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll
@@ -0,0 +1,303 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu -mattr=+avx2 | FileCheck %s --check-prefix=AVX2
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu -mattr=+sse4.2 | FileCheck %s --check-prefix=SSE4
+
+ at in = global [8 x i32] zeroinitializer, align 16
+ at out = global [8 x i32] zeroinitializer, align 16
+
+; out[i] = (in[i] + x) * i, unrolled: the multiplications by 0 and 1 are folded,
+; so the lane 0 is the absorbing constant and the lane 1 is the copyable add.
+define void @mul_by_index(i32 noundef %x) {
+; AVX2-LABEL: define void @mul_by_index(
+; AVX2-SAME: i32 noundef [[X:%.*]]) #[[ATTR0:[0-9]+]] {
+; AVX2-NEXT: [[ENTRY:.*:]]
+; AVX2-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT: [[TMP7:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT: [[TMP9:%.*]] = insertelement <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; AVX2-NEXT: [[TMP10:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP11:%.*]] = shufflevector <8 x i32> [[TMP9]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT: [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP7]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP11]], <8 x i32> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT: [[TMP8:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT: [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP8]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT: [[TMP4:%.*]] = add <8 x i32> [[TMP1]], [[TMP3]]
+; AVX2-NEXT: [[TMP5:%.*]] = mul <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, [[TMP4]]
+; AVX2-NEXT: store <8 x i32> [[TMP5]], ptr @out, align 16
+; AVX2-NEXT: ret void
+;
+; SSE4-LABEL: define void @mul_by_index(
+; SSE4-SAME: i32 noundef [[X:%.*]]) #[[ATTR0:[0-9]+]] {
+; SSE4-NEXT: [[ENTRY:.*:]]
+; SSE4-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT: [[TMP9:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT: [[TMP2:%.*]] = insertelement <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; SSE4-NEXT: [[TMP12:%.*]] = shufflevector <2 x i32> [[TMP9]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT: [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP12]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT: [[TMP13:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT: [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP13]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT: [[TMP4:%.*]] = add <4 x i32> [[TMP1]], [[TMP3]]
+; SSE4-NEXT: [[TMP5:%.*]] = mul <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[TMP4]]
+; SSE4-NEXT: store <4 x i32> [[TMP5]], ptr @out, align 16
+; SSE4-NEXT: [[TMP6:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT: [[TMP11:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SSE4-NEXT: [[TMP7:%.*]] = add <4 x i32> [[TMP6]], [[TMP11]]
+; SSE4-NEXT: [[TMP8:%.*]] = mul <4 x i32> [[TMP7]], <i32 4, i32 5, i32 6, i32 7>
+; SSE4-NEXT: store <4 x i32> [[TMP8]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+; SSE4-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %0 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %add.1 = add i32 %0, %x
+ store i32 %add.1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+ %add.2 = add i32 %1, %x
+ %mul.2 = shl i32 %add.2, 1
+ store i32 %mul.2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 8
+ %2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %add.3 = add i32 %2, %x
+ %mul.3 = mul i32 %add.3, 3
+ store i32 %mul.3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ %3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+ %add.4 = add i32 %3, %x
+ %mul.4 = shl i32 %add.4, 2
+ store i32 %mul.4, ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+ %4 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 20), align 4
+ %add.5 = add i32 %4, %x
+ %mul.5 = mul i32 %add.5, 5
+ store i32 %mul.5, ptr getelementptr inbounds nuw (i8, ptr @out, i64 20), align 4
+ %5 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 24), align 8
+ %add.6 = add i32 %5, %x
+ %mul.6 = mul i32 %add.6, 6
+ store i32 %mul.6, ptr getelementptr inbounds nuw (i8, ptr @out, i64 24), align 8
+ %6 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 28), align 4
+ %add.7 = add i32 %6, %x
+ %mul.7 = mul i32 %add.7, 7
+ store i32 %mul.7, ptr getelementptr inbounds nuw (i8, ptr @out, i64 28), align 4
+ ret void
+}
+
+; The wrapping flags of the add must not survive: the lane 0 of the add is not
+; computed from the original operands.
+define void @mul_by_index_nsw(i32 noundef %x) {
+; AVX2-LABEL: define void @mul_by_index_nsw(
+; AVX2-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; AVX2-NEXT: [[ENTRY:.*:]]
+; AVX2-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT: [[TMP1:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT: [[TMP3:%.*]] = insertelement <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; AVX2-NEXT: [[TMP10:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP11:%.*]] = shufflevector <8 x i32> [[TMP3]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT: [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP7:%.*]] = shufflevector <8 x i32> [[TMP11]], <8 x i32> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT: [[TMP8:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT: [[TMP9:%.*]] = shufflevector <8 x i32> [[TMP8]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT: [[TMP4:%.*]] = add nsw <8 x i32> [[TMP7]], [[TMP9]]
+; AVX2-NEXT: [[TMP5:%.*]] = mul <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, [[TMP4]]
+; AVX2-NEXT: store <8 x i32> [[TMP5]], ptr @out, align 16
+; AVX2-NEXT: ret void
+;
+; SSE4-LABEL: define void @mul_by_index_nsw(
+; SSE4-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; SSE4-NEXT: [[ENTRY:.*:]]
+; SSE4-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT: [[TMP1:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT: [[TMP2:%.*]] = insertelement <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; SSE4-NEXT: [[TMP9:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT: [[TMP11:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP9]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT: [[TMP12:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT: [[TMP13:%.*]] = shufflevector <4 x i32> [[TMP12]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT: [[TMP4:%.*]] = add nsw <4 x i32> [[TMP11]], [[TMP13]]
+; SSE4-NEXT: [[TMP5:%.*]] = mul <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[TMP4]]
+; SSE4-NEXT: store <4 x i32> [[TMP5]], ptr @out, align 16
+; SSE4-NEXT: [[TMP6:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT: [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SSE4-NEXT: [[TMP7:%.*]] = add nsw <4 x i32> [[TMP6]], [[TMP3]]
+; SSE4-NEXT: [[TMP8:%.*]] = mul <4 x i32> [[TMP7]], <i32 4, i32 5, i32 6, i32 7>
+; SSE4-NEXT: store <4 x i32> [[TMP8]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+; SSE4-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %0 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %add.1 = add nsw i32 %0, %x
+ store i32 %add.1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+ %add.2 = add nsw i32 %1, %x
+ %mul.2 = shl i32 %add.2, 1
+ store i32 %mul.2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 8
+ %2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %add.3 = add nsw i32 %2, %x
+ %mul.3 = mul i32 %add.3, 3
+ store i32 %mul.3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ %3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+ %add.4 = add nsw i32 %3, %x
+ %mul.4 = shl i32 %add.4, 2
+ store i32 %mul.4, ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+ %4 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 20), align 4
+ %add.5 = add nsw i32 %4, %x
+ %mul.5 = mul i32 %add.5, 5
+ store i32 %mul.5, ptr getelementptr inbounds nuw (i8, ptr @out, i64 20), align 4
+ %5 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 24), align 8
+ %add.6 = add nsw i32 %5, %x
+ %mul.6 = mul i32 %add.6, 6
+ store i32 %mul.6, ptr getelementptr inbounds nuw (i8, ptr @out, i64 24), align 8
+ %6 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 28), align 4
+ %add.7 = add nsw i32 %6, %x
+ %mul.7 = mul i32 %add.7, 7
+ store i32 %mul.7, ptr getelementptr inbounds nuw (i8, ptr @out, i64 28), align 4
+ ret void
+}
+
+; The same with the absorbing zero of the and.
+define void @and_with_zero_lane(i32 noundef %x) {
+; AVX2-LABEL: define void @and_with_zero_lane(
+; AVX2-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; AVX2-NEXT: [[ENTRY:.*:]]
+; AVX2-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT: [[TMP7:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT: [[TMP9:%.*]] = insertelement <8 x i32> <i32 -1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; AVX2-NEXT: [[TMP10:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP11:%.*]] = shufflevector <8 x i32> [[TMP9]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT: [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP7]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP11]], <8 x i32> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT: [[TMP8:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT: [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP8]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT: [[TMP4:%.*]] = add <8 x i32> [[TMP1]], [[TMP3]]
+; AVX2-NEXT: [[TMP5:%.*]] = and <8 x i32> <i32 0, i32 15, i32 255, i32 4095, i32 65535, i32 1048575, i32 16777215, i32 268435455>, [[TMP4]]
+; AVX2-NEXT: store <8 x i32> [[TMP5]], ptr @out, align 16
+; AVX2-NEXT: ret void
+;
+; SSE4-LABEL: define void @and_with_zero_lane(
+; SSE4-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; SSE4-NEXT: [[ENTRY:.*:]]
+; SSE4-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT: [[TMP9:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT: [[TMP2:%.*]] = insertelement <4 x i32> <i32 -1, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; SSE4-NEXT: [[TMP12:%.*]] = shufflevector <2 x i32> [[TMP9]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT: [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP12]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT: [[TMP13:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT: [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP13]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT: [[TMP4:%.*]] = add <4 x i32> [[TMP1]], [[TMP3]]
+; SSE4-NEXT: [[TMP5:%.*]] = and <4 x i32> <i32 0, i32 15, i32 255, i32 4095>, [[TMP4]]
+; SSE4-NEXT: store <4 x i32> [[TMP5]], ptr @out, align 16
+; SSE4-NEXT: [[TMP6:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT: [[TMP11:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SSE4-NEXT: [[TMP7:%.*]] = add <4 x i32> [[TMP6]], [[TMP11]]
+; SSE4-NEXT: [[TMP8:%.*]] = and <4 x i32> [[TMP7]], <i32 65535, i32 1048575, i32 16777215, i32 268435455>
+; SSE4-NEXT: store <4 x i32> [[TMP8]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+; SSE4-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %0 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %add.1 = add i32 %0, %x
+ %and.1 = and i32 %add.1, 15
+ store i32 %and.1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+ %add.2 = add i32 %1, %x
+ %and.2 = and i32 %add.2, 255
+ store i32 %and.2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 8
+ %2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %add.3 = add i32 %2, %x
+ %and.3 = and i32 %add.3, 4095
+ store i32 %and.3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ %3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+ %add.4 = add i32 %3, %x
+ %and.4 = and i32 %add.4, 65535
+ store i32 %and.4, ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+ %4 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 20), align 4
+ %add.5 = add i32 %4, %x
+ %and.5 = and i32 %add.5, 1048575
+ store i32 %and.5, ptr getelementptr inbounds nuw (i8, ptr @out, i64 20), align 4
+ %5 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 24), align 8
+ %add.6 = add i32 %5, %x
+ %and.6 = and i32 %add.6, 16777215
+ store i32 %and.6, ptr getelementptr inbounds nuw (i8, ptr @out, i64 24), align 8
+ %6 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 28), align 4
+ %add.7 = add i32 %6, %x
+ %and.7 = and i32 %add.7, 268435455
+ store i32 %and.7, ptr getelementptr inbounds nuw (i8, ptr @out, i64 28), align 4
+ ret void
+}
+
+; The scalar x may be poison, so the broadcast is frozen to keep the lane of the
+; folded multiplication non-poison.
+define void @mul_by_index_poison_x(i32 %x) {
+; AVX2-LABEL: define void @mul_by_index_poison_x(
+; AVX2-SAME: i32 [[X:%.*]]) #[[ATTR0]] {
+; AVX2-NEXT: [[ENTRY:.*:]]
+; AVX2-NEXT: [[L1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT: [[TMP8:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT: [[TMP2:%.*]] = insertelement <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[L1]], i64 1
+; AVX2-NEXT: [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP8]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP9:%.*]] = shufflevector <8 x i32> [[TMP2]], <8 x i32> [[TMP3]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT: [[TMP10:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP9]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT: [[TMP7:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT: [[TMP4:%.*]] = shufflevector <8 x i32> [[TMP7]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT: [[TMP5:%.*]] = add <8 x i32> [[TMP1]], [[TMP4]]
+; AVX2-NEXT: [[TMP6:%.*]] = mul <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, [[TMP5]]
+; AVX2-NEXT: store <8 x i32> [[TMP6]], ptr @out, align 16
+; AVX2-NEXT: ret void
+;
+; SSE4-LABEL: define void @mul_by_index_poison_x(
+; SSE4-SAME: i32 [[X:%.*]]) #[[ATTR0]] {
+; SSE4-NEXT: [[ENTRY:.*:]]
+; SSE4-NEXT: [[L1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison>, i32 [[L1]], i64 1
+; SSE4-NEXT: [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT: [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> [[TMP2]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT: [[TMP11:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT: [[TMP4:%.*]] = shufflevector <4 x i32> [[TMP11]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT: [[TMP5:%.*]] = add <4 x i32> [[TMP1]], [[TMP4]]
+; SSE4-NEXT: [[TMP6:%.*]] = mul <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[TMP5]]
+; SSE4-NEXT: store <4 x i32> [[TMP6]], ptr @out, align 16
+; SSE4-NEXT: [[TMP7:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT: [[TMP12:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT: [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP12]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SSE4-NEXT: [[TMP8:%.*]] = add <4 x i32> [[TMP7]], [[TMP3]]
+; SSE4-NEXT: [[TMP9:%.*]] = mul <4 x i32> [[TMP8]], <i32 4, i32 5, i32 6, i32 7>
+; SSE4-NEXT: store <4 x i32> [[TMP9]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+; SSE4-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %l1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %add.1 = add i32 %l1, %x
+ %r.1 = mul i32 %add.1, 1
+ store i32 %r.1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %l2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+ %add.2 = add i32 %l2, %x
+ %r.2 = mul i32 %add.2, 2
+ store i32 %r.2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 8
+ %l3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %add.3 = add i32 %l3, %x
+ %r.3 = mul i32 %add.3, 3
+ store i32 %r.3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ %l4 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+ %add.4 = add i32 %l4, %x
+ %r.4 = mul i32 %add.4, 4
+ store i32 %r.4, ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+ %l5 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 20), align 4
+ %add.5 = add i32 %l5, %x
+ %r.5 = mul i32 %add.5, 5
+ store i32 %r.5, ptr getelementptr inbounds nuw (i8, ptr @out, i64 20), align 4
+ %l6 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 24), align 8
+ %add.6 = add i32 %l6, %x
+ %r.6 = mul i32 %add.6, 6
+ store i32 %r.6, ptr getelementptr inbounds nuw (i8, ptr @out, i64 24), align 8
+ %l7 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 28), align 4
+ %add.7 = add i32 %l7, %x
+ %r.7 = mul i32 %add.7, 7
+ store i32 %r.7, ptr getelementptr inbounds nuw (i8, ptr @out, i64 28), align 4
+ ret void
+}
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/wide-load-absorbed-lane.ll b/llvm/test/Transforms/SLPVectorizer/X86/wide-load-absorbed-lane.ll
new file mode 100644
index 0000000000000..76e4547588392
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/wide-load-absorbed-lane.ll
@@ -0,0 +1,401 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu -mattr=+sse2 | FileCheck %s
+
+ at in = global [4 x i32] zeroinitializer, align 16
+ at out = global [4 x i32] zeroinitializer, align 16
+
+; out[i] = (in[i] + x) * c[i], unrolled. The multiplication by 0 is folded into
+; the store of 0, so the lane has no load. The loads of the other lanes are
+; consecutive, the memory of the lane is covered by a single wide load. The
+; memory may hold poison, so the wide load is frozen.
+define void @absorbed_first(i32 noundef %x) {
+; CHECK-LABEL: define void @absorbed_first(
+; CHECK-SAME: i32 noundef [[X:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: store i32 0, ptr @out, align 16
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; CHECK-NEXT: [[A1:%.*]] = add i32 [[L1]], [[X]]
+; CHECK-NEXT: [[M1:%.*]] = mul i32 [[A1]], 5
+; CHECK-NEXT: store i32 [[M1]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 4
+; CHECK-NEXT: [[A2:%.*]] = add i32 [[L2]], [[X]]
+; CHECK-NEXT: [[M2:%.*]] = mul i32 [[A2]], 7
+; CHECK-NEXT: store i32 [[M2]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+; CHECK-NEXT: [[L3:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+; CHECK-NEXT: [[A3:%.*]] = add i32 [[L3]], [[X]]
+; CHECK-NEXT: [[M3:%.*]] = mul i32 [[A3]], 9
+; CHECK-NEXT: store i32 [[M3]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %l1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %a1 = add i32 %l1, %x
+ %m1 = mul i32 %a1, 5
+ store i32 %m1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %l2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 4
+ %a2 = add i32 %l2, %x
+ %m2 = mul i32 %a2, 7
+ store i32 %m2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+ %l3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %a3 = add i32 %l3, %x
+ %m3 = mul i32 %a3, 9
+ store i32 %m3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ ret void
+}
+
+define void @absorbed_middle(i32 noundef %x) {
+; CHECK-LABEL: define void @absorbed_middle(
+; CHECK-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[L0:%.*]] = load i32, ptr @in, align 4
+; CHECK-NEXT: [[A0:%.*]] = add i32 [[L0]], [[X]]
+; CHECK-NEXT: [[M0:%.*]] = mul i32 [[A0]], 3
+; CHECK-NEXT: store i32 [[M0]], ptr @out, align 16
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; CHECK-NEXT: [[A1:%.*]] = add i32 [[L1]], [[X]]
+; CHECK-NEXT: [[M1:%.*]] = mul i32 [[A1]], 5
+; CHECK-NEXT: store i32 [[M1]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+; CHECK-NEXT: store i32 0, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+; CHECK-NEXT: [[L3:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+; CHECK-NEXT: [[A3:%.*]] = add i32 [[L3]], [[X]]
+; CHECK-NEXT: [[M3:%.*]] = mul i32 [[A3]], 9
+; CHECK-NEXT: store i32 [[M3]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %l0 = load i32, ptr @in, align 4
+ %a0 = add i32 %l0, %x
+ %m0 = mul i32 %a0, 3
+ store i32 %m0, ptr @out, align 16
+ %l1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %a1 = add i32 %l1, %x
+ %m1 = mul i32 %a1, 5
+ store i32 %m1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ store i32 0, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+ %l3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %a3 = add i32 %l3, %x
+ %m3 = mul i32 %a3, 9
+ store i32 %m3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ ret void
+}
+
+define void @absorbed_last(i32 noundef %x) {
+; CHECK-LABEL: define void @absorbed_last(
+; CHECK-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[L0:%.*]] = load i32, ptr @in, align 4
+; CHECK-NEXT: [[A0:%.*]] = add i32 [[L0]], [[X]]
+; CHECK-NEXT: [[M0:%.*]] = mul i32 [[A0]], 3
+; CHECK-NEXT: store i32 [[M0]], ptr @out, align 16
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; CHECK-NEXT: [[A1:%.*]] = add i32 [[L1]], [[X]]
+; CHECK-NEXT: [[M1:%.*]] = mul i32 [[A1]], 5
+; CHECK-NEXT: store i32 [[M1]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 4
+; CHECK-NEXT: [[A2:%.*]] = add i32 [[L2]], [[X]]
+; CHECK-NEXT: [[M2:%.*]] = mul i32 [[A2]], 7
+; CHECK-NEXT: store i32 [[M2]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+; CHECK-NEXT: store i32 0, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %l0 = load i32, ptr @in, align 4
+ %a0 = add i32 %l0, %x
+ %m0 = mul i32 %a0, 3
+ store i32 %m0, ptr @out, align 16
+ %l1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %a1 = add i32 %l1, %x
+ %m1 = mul i32 %a1, 5
+ store i32 %m1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %l2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 4
+ %a2 = add i32 %l2, %x
+ %m2 = mul i32 %a2, 7
+ store i32 %m2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+ store i32 0, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ ret void
+}
+
+; The store to the memory of the lane between the loads does not prevent the
+; wide load, it is placed after the store.
+define void @store_to_absorbed_lane(i32 noundef %x) {
+; CHECK-LABEL: define void @store_to_absorbed_lane(
+; CHECK-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: store i32 0, ptr @out, align 16
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; CHECK-NEXT: store i32 7, ptr @in, align 16
+; CHECK-NEXT: [[A1:%.*]] = add i32 [[L1]], [[X]]
+; CHECK-NEXT: [[M1:%.*]] = mul i32 [[A1]], 5
+; CHECK-NEXT: store i32 [[M1]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 4
+; CHECK-NEXT: [[A2:%.*]] = add i32 [[L2]], [[X]]
+; CHECK-NEXT: [[M2:%.*]] = mul i32 [[A2]], 7
+; CHECK-NEXT: store i32 [[M2]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+; CHECK-NEXT: [[L3:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+; CHECK-NEXT: [[A3:%.*]] = add i32 [[L3]], [[X]]
+; CHECK-NEXT: [[M3:%.*]] = mul i32 [[A3]], 9
+; CHECK-NEXT: store i32 [[M3]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %l1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ store i32 7, ptr @in, align 16
+ %a1 = add i32 %l1, %x
+ %m1 = mul i32 %a1, 5
+ store i32 %m1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %l2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 4
+ %a2 = add i32 %l2, %x
+ %m2 = mul i32 %a2, 7
+ store i32 %m2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+ %l3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %a3 = add i32 %l3, %x
+ %m3 = mul i32 %a3, 9
+ store i32 %m3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ ret void
+}
+
+; The wide load is not allowed below.
+;
+; The memory of the lane is not known to be dereferenceable.
+define void @not_dereferenceable(ptr %p, i32 noundef %x) {
+; CHECK-LABEL: define void @not_dereferenceable(
+; CHECK-SAME: ptr [[P:%.*]], i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: store i32 0, ptr @out, align 16
+; CHECK-NEXT: [[P1:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 1
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[P1]], align 4
+; CHECK-NEXT: [[A1:%.*]] = add i32 [[L1]], [[X]]
+; CHECK-NEXT: [[M1:%.*]] = mul i32 [[A1]], 5
+; CHECK-NEXT: store i32 [[M1]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+; CHECK-NEXT: [[P2:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 2
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[P2]], align 4
+; CHECK-NEXT: [[A2:%.*]] = add i32 [[L2]], [[X]]
+; CHECK-NEXT: [[M2:%.*]] = mul i32 [[A2]], 7
+; CHECK-NEXT: store i32 [[M2]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+; CHECK-NEXT: [[P3:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 3
+; CHECK-NEXT: [[L3:%.*]] = load i32, ptr [[P3]], align 4
+; CHECK-NEXT: [[A3:%.*]] = add i32 [[L3]], [[X]]
+; CHECK-NEXT: [[M3:%.*]] = mul i32 [[A3]], 9
+; CHECK-NEXT: store i32 [[M3]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %p1 = getelementptr inbounds i32, ptr %p, i64 1
+ %l1 = load i32, ptr %p1, align 4
+ %a1 = add i32 %l1, %x
+ %m1 = mul i32 %a1, 5
+ store i32 %m1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %p2 = getelementptr inbounds i32, ptr %p, i64 2
+ %l2 = load i32, ptr %p2, align 4
+ %a2 = add i32 %l2, %x
+ %m2 = mul i32 %a2, 7
+ store i32 %m2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+ %p3 = getelementptr inbounds i32, ptr %p, i64 3
+ %l3 = load i32, ptr %p3, align 4
+ %a3 = add i32 %l3, %x
+ %m3 = mul i32 %a3, 9
+ store i32 %m3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ ret void
+}
+
+; The wide load would start before the dereferenceable memory.
+define void @before_object(ptr dereferenceable(12) align 4 %p, i32 noundef %x) {
+; CHECK-LABEL: define void @before_object(
+; CHECK-SAME: ptr align 4 dereferenceable(12) [[P:%.*]], i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: store i32 0, ptr @out, align 16
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[A1:%.*]] = add i32 [[L1]], [[X]]
+; CHECK-NEXT: [[M1:%.*]] = mul i32 [[A1]], 5
+; CHECK-NEXT: store i32 [[M1]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+; CHECK-NEXT: [[P2:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 1
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[P2]], align 4
+; CHECK-NEXT: [[A2:%.*]] = add i32 [[L2]], [[X]]
+; CHECK-NEXT: [[M2:%.*]] = mul i32 [[A2]], 7
+; CHECK-NEXT: store i32 [[M2]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+; CHECK-NEXT: [[P3:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 2
+; CHECK-NEXT: [[L3:%.*]] = load i32, ptr [[P3]], align 4
+; CHECK-NEXT: [[A3:%.*]] = add i32 [[L3]], [[X]]
+; CHECK-NEXT: [[M3:%.*]] = mul i32 [[A3]], 9
+; CHECK-NEXT: store i32 [[M3]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %l1 = load i32, ptr %p, align 4
+ %a1 = add i32 %l1, %x
+ %m1 = mul i32 %a1, 5
+ store i32 %m1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %p2 = getelementptr inbounds i32, ptr %p, i64 1
+ %l2 = load i32, ptr %p2, align 4
+ %a2 = add i32 %l2, %x
+ %m2 = mul i32 %a2, 7
+ store i32 %m2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+ %p3 = getelementptr inbounds i32, ptr %p, i64 2
+ %l3 = load i32, ptr %p3, align 4
+ %a3 = add i32 %l3, %x
+ %m3 = mul i32 %a3, 9
+ store i32 %m3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ ret void
+}
+
+; The wide load would end after the dereferenceable memory.
+define void @after_object(ptr dereferenceable(12) align 4 %p, i32 noundef %x) {
+; CHECK-LABEL: define void @after_object(
+; CHECK-SAME: ptr align 4 dereferenceable(12) [[P:%.*]], i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[L0:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[A0:%.*]] = add i32 [[L0]], [[X]]
+; CHECK-NEXT: [[M0:%.*]] = mul i32 [[A0]], 3
+; CHECK-NEXT: store i32 [[M0]], ptr @out, align 16
+; CHECK-NEXT: [[P1:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 1
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[P1]], align 4
+; CHECK-NEXT: [[A1:%.*]] = add i32 [[L1]], [[X]]
+; CHECK-NEXT: [[M1:%.*]] = mul i32 [[A1]], 5
+; CHECK-NEXT: store i32 [[M1]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+; CHECK-NEXT: [[P2:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 2
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[P2]], align 4
+; CHECK-NEXT: [[A2:%.*]] = add i32 [[L2]], [[X]]
+; CHECK-NEXT: [[M2:%.*]] = mul i32 [[A2]], 7
+; CHECK-NEXT: store i32 [[M2]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+; CHECK-NEXT: store i32 0, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %l0 = load i32, ptr %p, align 4
+ %a0 = add i32 %l0, %x
+ %m0 = mul i32 %a0, 3
+ store i32 %m0, ptr @out, align 16
+ %p1 = getelementptr inbounds i32, ptr %p, i64 1
+ %l1 = load i32, ptr %p1, align 4
+ %a1 = add i32 %l1, %x
+ %m1 = mul i32 %a1, 5
+ store i32 %m1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %p2 = getelementptr inbounds i32, ptr %p, i64 2
+ %l2 = load i32, ptr %p2, align 4
+ %a2 = add i32 %l2, %x
+ %m2 = mul i32 %a2, 7
+ store i32 %m2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+ store i32 0, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ ret void
+}
+
+; The wide load reads the memory, which is not read by the source.
+define void @sanitized(ptr dereferenceable(16) align 16 %p, i32 noundef %x) sanitize_address {
+; CHECK-LABEL: define void @sanitized(
+; CHECK-SAME: ptr align 16 dereferenceable(16) [[P:%.*]], i32 noundef [[X:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: store i32 0, ptr @out, align 16
+; CHECK-NEXT: [[P1:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 1
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[P1]], align 4
+; CHECK-NEXT: [[A1:%.*]] = add i32 [[L1]], [[X]]
+; CHECK-NEXT: [[M1:%.*]] = mul i32 [[A1]], 5
+; CHECK-NEXT: store i32 [[M1]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+; CHECK-NEXT: [[P2:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 2
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr [[P2]], align 4
+; CHECK-NEXT: [[A2:%.*]] = add i32 [[L2]], [[X]]
+; CHECK-NEXT: [[M2:%.*]] = mul i32 [[A2]], 7
+; CHECK-NEXT: store i32 [[M2]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+; CHECK-NEXT: [[P3:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 3
+; CHECK-NEXT: [[L3:%.*]] = load i32, ptr [[P3]], align 4
+; CHECK-NEXT: [[A3:%.*]] = add i32 [[L3]], [[X]]
+; CHECK-NEXT: [[M3:%.*]] = mul i32 [[A3]], 9
+; CHECK-NEXT: store i32 [[M3]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %p1 = getelementptr inbounds i32, ptr %p, i64 1
+ %l1 = load i32, ptr %p1, align 4
+ %a1 = add i32 %l1, %x
+ %m1 = mul i32 %a1, 5
+ store i32 %m1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %p2 = getelementptr inbounds i32, ptr %p, i64 2
+ %l2 = load i32, ptr %p2, align 4
+ %a2 = add i32 %l2, %x
+ %m2 = mul i32 %a2, 7
+ store i32 %m2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+ %p3 = getelementptr inbounds i32, ptr %p, i64 3
+ %l3 = load i32, ptr %p3, align 4
+ %a3 = add i32 %l3, %x
+ %m3 = mul i32 %a3, 9
+ store i32 %m3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ ret void
+}
+
+; The volatile load cannot be widened.
+define void @volatile_load(i32 noundef %x) {
+; CHECK-LABEL: define void @volatile_load(
+; CHECK-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: store i32 0, ptr @out, align 16
+; CHECK-NEXT: [[L1:%.*]] = load volatile i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; CHECK-NEXT: [[A1:%.*]] = add i32 [[L1]], [[X]]
+; CHECK-NEXT: [[M1:%.*]] = mul i32 [[A1]], 5
+; CHECK-NEXT: store i32 [[M1]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 4
+; CHECK-NEXT: [[A2:%.*]] = add i32 [[L2]], [[X]]
+; CHECK-NEXT: [[M2:%.*]] = mul i32 [[A2]], 7
+; CHECK-NEXT: store i32 [[M2]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+; CHECK-NEXT: [[L3:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+; CHECK-NEXT: [[A3:%.*]] = add i32 [[L3]], [[X]]
+; CHECK-NEXT: [[M3:%.*]] = mul i32 [[A3]], 9
+; CHECK-NEXT: store i32 [[M3]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %l1 = load volatile i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %a1 = add i32 %l1, %x
+ %m1 = mul i32 %a1, 5
+ store i32 %m1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %l2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 4
+ %a2 = add i32 %l2, %x
+ %m2 = mul i32 %a2, 7
+ store i32 %m2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+ %l3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %a3 = add i32 %l3, %x
+ %m3 = mul i32 %a3, 9
+ store i32 %m3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ ret void
+}
+
+; The loads do not match their lanes.
+define void @not_consecutive(i32 noundef %x) {
+; CHECK-LABEL: define void @not_consecutive(
+; CHECK-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: store i32 0, ptr @out, align 16
+; CHECK-NEXT: [[L1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; CHECK-NEXT: [[A1:%.*]] = add i32 [[L1]], [[X]]
+; CHECK-NEXT: [[M1:%.*]] = mul i32 [[A1]], 5
+; CHECK-NEXT: store i32 [[M1]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+; CHECK-NEXT: [[L2:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+; CHECK-NEXT: [[A2:%.*]] = add i32 [[L2]], [[X]]
+; CHECK-NEXT: [[M2:%.*]] = mul i32 [[A2]], 7
+; CHECK-NEXT: store i32 [[M2]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+; CHECK-NEXT: [[L3:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 4
+; CHECK-NEXT: [[A3:%.*]] = add i32 [[L3]], [[X]]
+; CHECK-NEXT: [[M3:%.*]] = mul i32 [[A3]], 9
+; CHECK-NEXT: store i32 [[M3]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %l1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %a1 = add i32 %l1, %x
+ %m1 = mul i32 %a1, 5
+ store i32 %m1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %l2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %a2 = add i32 %l2, %x
+ %m2 = mul i32 %a2, 7
+ store i32 %m2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 4
+ %l3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 4
+ %a3 = add i32 %l3, %x
+ %m3 = mul i32 %a3, 9
+ store i32 %m3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ ret void
+}
More information about the llvm-commits
mailing list