[llvm] [SLP][NFC]Add a test with the non-optimal vectorization, NFC (PR #228385)

via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 03:34:16 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Alexey Bataev (alexey-bataev)

<details>
<summary>Changes</summary>



---

Patch is 41.76 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/228385.diff


2 Files Affected:

- (added) llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll (+303) 
- (added) llvm/test/Transforms/SLPVectorizer/X86/wide-load-absorbed-lane.ll (+401) 


``````````diff
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll b/llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll
new file mode 100644
index 00000000000000..1d7933334c77db
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll
@@ -0,0 +1,303 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu -mattr=+avx2 | FileCheck %s --check-prefix=AVX2
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu -mattr=+sse4.2 | FileCheck %s --check-prefix=SSE4
+
+ at in = global [8 x i32] zeroinitializer, align 16
+ at out = global [8 x i32] zeroinitializer, align 16
+
+; out[i] = (in[i] + x) * i, unrolled: the multiplications by 0 and 1 are folded,
+; so the lane 0 is the absorbing constant and the lane 1 is the copyable add.
+define void @mul_by_index(i32 noundef %x) {
+; AVX2-LABEL: define void @mul_by_index(
+; AVX2-SAME: i32 noundef [[X:%.*]]) #[[ATTR0:[0-9]+]] {
+; AVX2-NEXT:  [[ENTRY:.*:]]
+; AVX2-NEXT:    [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT:    [[TMP7:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT:    [[TMP9:%.*]] = insertelement <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; AVX2-NEXT:    [[TMP10:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT:    [[TMP11:%.*]] = shufflevector <8 x i32> [[TMP9]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT:    [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP7]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP11]], <8 x i32> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT:    [[TMP8:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT:    [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP8]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT:    [[TMP4:%.*]] = add <8 x i32> [[TMP1]], [[TMP3]]
+; AVX2-NEXT:    [[TMP5:%.*]] = mul <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, [[TMP4]]
+; AVX2-NEXT:    store <8 x i32> [[TMP5]], ptr @out, align 16
+; AVX2-NEXT:    ret void
+;
+; SSE4-LABEL: define void @mul_by_index(
+; SSE4-SAME: i32 noundef [[X:%.*]]) #[[ATTR0:[0-9]+]] {
+; SSE4-NEXT:  [[ENTRY:.*:]]
+; SSE4-NEXT:    [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT:    [[TMP9:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT:    [[TMP2:%.*]] = insertelement <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; SSE4-NEXT:    [[TMP12:%.*]] = shufflevector <2 x i32> [[TMP9]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP12]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT:    [[TMP13:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP13]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT:    [[TMP4:%.*]] = add <4 x i32> [[TMP1]], [[TMP3]]
+; SSE4-NEXT:    [[TMP5:%.*]] = mul <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[TMP4]]
+; SSE4-NEXT:    store <4 x i32> [[TMP5]], ptr @out, align 16
+; SSE4-NEXT:    [[TMP6:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT:    [[TMP10:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT:    [[TMP11:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SSE4-NEXT:    [[TMP7:%.*]] = add <4 x i32> [[TMP6]], [[TMP11]]
+; SSE4-NEXT:    [[TMP8:%.*]] = mul <4 x i32> [[TMP7]], <i32 4, i32 5, i32 6, i32 7>
+; SSE4-NEXT:    store <4 x i32> [[TMP8]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+; SSE4-NEXT:    ret void
+;
+entry:
+  store i32 0, ptr @out, align 16
+  %0 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+  %add.1 = add i32 %0, %x
+  store i32 %add.1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+  %1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+  %add.2 = add i32 %1, %x
+  %mul.2 = shl i32 %add.2, 1
+  store i32 %mul.2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 8
+  %2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+  %add.3 = add i32 %2, %x
+  %mul.3 = mul i32 %add.3, 3
+  store i32 %mul.3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+  %3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+  %add.4 = add i32 %3, %x
+  %mul.4 = shl i32 %add.4, 2
+  store i32 %mul.4, ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+  %4 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 20), align 4
+  %add.5 = add i32 %4, %x
+  %mul.5 = mul i32 %add.5, 5
+  store i32 %mul.5, ptr getelementptr inbounds nuw (i8, ptr @out, i64 20), align 4
+  %5 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 24), align 8
+  %add.6 = add i32 %5, %x
+  %mul.6 = mul i32 %add.6, 6
+  store i32 %mul.6, ptr getelementptr inbounds nuw (i8, ptr @out, i64 24), align 8
+  %6 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 28), align 4
+  %add.7 = add i32 %6, %x
+  %mul.7 = mul i32 %add.7, 7
+  store i32 %mul.7, ptr getelementptr inbounds nuw (i8, ptr @out, i64 28), align 4
+  ret void
+}
+
+; The wrapping flags of the add must not survive: the lane 0 of the add is not
+; computed from the original operands.
+define void @mul_by_index_nsw(i32 noundef %x) {
+; AVX2-LABEL: define void @mul_by_index_nsw(
+; AVX2-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; AVX2-NEXT:  [[ENTRY:.*:]]
+; AVX2-NEXT:    [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT:    [[TMP1:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT:    [[TMP3:%.*]] = insertelement <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; AVX2-NEXT:    [[TMP10:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT:    [[TMP11:%.*]] = shufflevector <8 x i32> [[TMP3]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT:    [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i32> [[TMP11]], <8 x i32> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT:    [[TMP8:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT:    [[TMP9:%.*]] = shufflevector <8 x i32> [[TMP8]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT:    [[TMP4:%.*]] = add nsw <8 x i32> [[TMP7]], [[TMP9]]
+; AVX2-NEXT:    [[TMP5:%.*]] = mul <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, [[TMP4]]
+; AVX2-NEXT:    store <8 x i32> [[TMP5]], ptr @out, align 16
+; AVX2-NEXT:    ret void
+;
+; SSE4-LABEL: define void @mul_by_index_nsw(
+; SSE4-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; SSE4-NEXT:  [[ENTRY:.*:]]
+; SSE4-NEXT:    [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT:    [[TMP1:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT:    [[TMP2:%.*]] = insertelement <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; SSE4-NEXT:    [[TMP9:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT:    [[TMP11:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP9]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT:    [[TMP12:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT:    [[TMP13:%.*]] = shufflevector <4 x i32> [[TMP12]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT:    [[TMP4:%.*]] = add nsw <4 x i32> [[TMP11]], [[TMP13]]
+; SSE4-NEXT:    [[TMP5:%.*]] = mul <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[TMP4]]
+; SSE4-NEXT:    store <4 x i32> [[TMP5]], ptr @out, align 16
+; SSE4-NEXT:    [[TMP6:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT:    [[TMP10:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SSE4-NEXT:    [[TMP7:%.*]] = add nsw <4 x i32> [[TMP6]], [[TMP3]]
+; SSE4-NEXT:    [[TMP8:%.*]] = mul <4 x i32> [[TMP7]], <i32 4, i32 5, i32 6, i32 7>
+; SSE4-NEXT:    store <4 x i32> [[TMP8]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+; SSE4-NEXT:    ret void
+;
+entry:
+  store i32 0, ptr @out, align 16
+  %0 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+  %add.1 = add nsw i32 %0, %x
+  store i32 %add.1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+  %1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+  %add.2 = add nsw i32 %1, %x
+  %mul.2 = shl i32 %add.2, 1
+  store i32 %mul.2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 8
+  %2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+  %add.3 = add nsw i32 %2, %x
+  %mul.3 = mul i32 %add.3, 3
+  store i32 %mul.3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+  %3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+  %add.4 = add nsw i32 %3, %x
+  %mul.4 = shl i32 %add.4, 2
+  store i32 %mul.4, ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+  %4 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 20), align 4
+  %add.5 = add nsw i32 %4, %x
+  %mul.5 = mul i32 %add.5, 5
+  store i32 %mul.5, ptr getelementptr inbounds nuw (i8, ptr @out, i64 20), align 4
+  %5 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 24), align 8
+  %add.6 = add nsw i32 %5, %x
+  %mul.6 = mul i32 %add.6, 6
+  store i32 %mul.6, ptr getelementptr inbounds nuw (i8, ptr @out, i64 24), align 8
+  %6 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 28), align 4
+  %add.7 = add nsw i32 %6, %x
+  %mul.7 = mul i32 %add.7, 7
+  store i32 %mul.7, ptr getelementptr inbounds nuw (i8, ptr @out, i64 28), align 4
+  ret void
+}
+
+; The same with the absorbing zero of the and.
+define void @and_with_zero_lane(i32 noundef %x) {
+; AVX2-LABEL: define void @and_with_zero_lane(
+; AVX2-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; AVX2-NEXT:  [[ENTRY:.*:]]
+; AVX2-NEXT:    [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT:    [[TMP7:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT:    [[TMP9:%.*]] = insertelement <8 x i32> <i32 -1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; AVX2-NEXT:    [[TMP10:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT:    [[TMP11:%.*]] = shufflevector <8 x i32> [[TMP9]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT:    [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP7]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP11]], <8 x i32> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT:    [[TMP8:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT:    [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP8]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT:    [[TMP4:%.*]] = add <8 x i32> [[TMP1]], [[TMP3]]
+; AVX2-NEXT:    [[TMP5:%.*]] = and <8 x i32> <i32 0, i32 15, i32 255, i32 4095, i32 65535, i32 1048575, i32 16777215, i32 268435455>, [[TMP4]]
+; AVX2-NEXT:    store <8 x i32> [[TMP5]], ptr @out, align 16
+; AVX2-NEXT:    ret void
+;
+; SSE4-LABEL: define void @and_with_zero_lane(
+; SSE4-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; SSE4-NEXT:  [[ENTRY:.*:]]
+; SSE4-NEXT:    [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT:    [[TMP9:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT:    [[TMP2:%.*]] = insertelement <4 x i32> <i32 -1, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; SSE4-NEXT:    [[TMP12:%.*]] = shufflevector <2 x i32> [[TMP9]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP12]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT:    [[TMP13:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP13]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT:    [[TMP4:%.*]] = add <4 x i32> [[TMP1]], [[TMP3]]
+; SSE4-NEXT:    [[TMP5:%.*]] = and <4 x i32> <i32 0, i32 15, i32 255, i32 4095>, [[TMP4]]
+; SSE4-NEXT:    store <4 x i32> [[TMP5]], ptr @out, align 16
+; SSE4-NEXT:    [[TMP6:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT:    [[TMP10:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT:    [[TMP11:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SSE4-NEXT:    [[TMP7:%.*]] = add <4 x i32> [[TMP6]], [[TMP11]]
+; SSE4-NEXT:    [[TMP8:%.*]] = and <4 x i32> [[TMP7]], <i32 65535, i32 1048575, i32 16777215, i32 268435455>
+; SSE4-NEXT:    store <4 x i32> [[TMP8]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+; SSE4-NEXT:    ret void
+;
+entry:
+  store i32 0, ptr @out, align 16
+  %0 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+  %add.1 = add i32 %0, %x
+  %and.1 = and i32 %add.1, 15
+  store i32 %and.1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+  %1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+  %add.2 = add i32 %1, %x
+  %and.2 = and i32 %add.2, 255
+  store i32 %and.2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 8
+  %2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+  %add.3 = add i32 %2, %x
+  %and.3 = and i32 %add.3, 4095
+  store i32 %and.3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+  %3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+  %add.4 = add i32 %3, %x
+  %and.4 = and i32 %add.4, 65535
+  store i32 %and.4, ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+  %4 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 20), align 4
+  %add.5 = add i32 %4, %x
+  %and.5 = and i32 %add.5, 1048575
+  store i32 %and.5, ptr getelementptr inbounds nuw (i8, ptr @out, i64 20), align 4
+  %5 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 24), align 8
+  %add.6 = add i32 %5, %x
+  %and.6 = and i32 %add.6, 16777215
+  store i32 %and.6, ptr getelementptr inbounds nuw (i8, ptr @out, i64 24), align 8
+  %6 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 28), align 4
+  %add.7 = add i32 %6, %x
+  %and.7 = and i32 %add.7, 268435455
+  store i32 %and.7, ptr getelementptr inbounds nuw (i8, ptr @out, i64 28), align 4
+  ret void
+}
+
+; The scalar x may be poison, so the broadcast is frozen to keep the lane of the
+; folded multiplication non-poison.
+define void @mul_by_index_poison_x(i32 %x) {
+; AVX2-LABEL: define void @mul_by_index_poison_x(
+; AVX2-SAME: i32 [[X:%.*]]) #[[ATTR0]] {
+; AVX2-NEXT:  [[ENTRY:.*:]]
+; AVX2-NEXT:    [[L1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT:    [[TMP0:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT:    [[TMP8:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT:    [[TMP2:%.*]] = insertelement <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[L1]], i64 1
+; AVX2-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP8]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT:    [[TMP9:%.*]] = shufflevector <8 x i32> [[TMP2]], <8 x i32> [[TMP3]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT:    [[TMP10:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP9]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT:    [[TMP7:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT:    [[TMP4:%.*]] = shufflevector <8 x i32> [[TMP7]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT:    [[TMP5:%.*]] = add <8 x i32> [[TMP1]], [[TMP4]]
+; AVX2-NEXT:    [[TMP6:%.*]] = mul <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, [[TMP5]]
+; AVX2-NEXT:    store <8 x i32> [[TMP6]], ptr @out, align 16
+; AVX2-NEXT:    ret void
+;
+; SSE4-LABEL: define void @mul_by_index_poison_x(
+; SSE4-SAME: i32 [[X:%.*]]) #[[ATTR0]] {
+; SSE4-NEXT:  [[ENTRY:.*:]]
+; SSE4-NEXT:    [[L1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT:    [[TMP0:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT:    [[TMP10:%.*]] = insertelement <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison>, i32 [[L1]], i64 1
+; SSE4-NEXT:    [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> [[TMP2]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT:    [[TMP11:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT:    [[TMP4:%.*]] = shufflevector <4 x i32> [[TMP11]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[TMP1]], [[TMP4]]
+; SSE4-NEXT:    [[TMP6:%.*]] = mul <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[TMP5]]
+; SSE4-NEXT:    store <4 x i32> [[TMP6]], ptr @out, align 16
+; SSE4-NEXT:    [[TMP7:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT:    [[TMP12:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT:    [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP12]]...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/228385


More information about the llvm-commits mailing list