[llvm] [SLP][NFC]Add a test with the non-optimal vectorization, NFC (PR #228385)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 03:34:16 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: Alexey Bataev (alexey-bataev)
<details>
<summary>Changes</summary>
---
Patch is 41.76 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/228385.diff
2 Files Affected:
- (added) llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll (+303)
- (added) llvm/test/Transforms/SLPVectorizer/X86/wide-load-absorbed-lane.ll (+401)
``````````diff
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll b/llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll
new file mode 100644
index 00000000000000..1d7933334c77db
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/absorbing-copyable-lane.ll
@@ -0,0 +1,303 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu -mattr=+avx2 | FileCheck %s --check-prefix=AVX2
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu -mattr=+sse4.2 | FileCheck %s --check-prefix=SSE4
+
+ at in = global [8 x i32] zeroinitializer, align 16
+ at out = global [8 x i32] zeroinitializer, align 16
+
+; out[i] = (in[i] + x) * i, unrolled: the multiplications by 0 and 1 are folded,
+; so the lane 0 is the absorbing constant and the lane 1 is the copyable add.
+define void @mul_by_index(i32 noundef %x) {
+; AVX2-LABEL: define void @mul_by_index(
+; AVX2-SAME: i32 noundef [[X:%.*]]) #[[ATTR0:[0-9]+]] {
+; AVX2-NEXT: [[ENTRY:.*:]]
+; AVX2-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT: [[TMP7:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT: [[TMP9:%.*]] = insertelement <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; AVX2-NEXT: [[TMP10:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP11:%.*]] = shufflevector <8 x i32> [[TMP9]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT: [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP7]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP11]], <8 x i32> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT: [[TMP8:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT: [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP8]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT: [[TMP4:%.*]] = add <8 x i32> [[TMP1]], [[TMP3]]
+; AVX2-NEXT: [[TMP5:%.*]] = mul <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, [[TMP4]]
+; AVX2-NEXT: store <8 x i32> [[TMP5]], ptr @out, align 16
+; AVX2-NEXT: ret void
+;
+; SSE4-LABEL: define void @mul_by_index(
+; SSE4-SAME: i32 noundef [[X:%.*]]) #[[ATTR0:[0-9]+]] {
+; SSE4-NEXT: [[ENTRY:.*:]]
+; SSE4-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT: [[TMP9:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT: [[TMP2:%.*]] = insertelement <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; SSE4-NEXT: [[TMP12:%.*]] = shufflevector <2 x i32> [[TMP9]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT: [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP12]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT: [[TMP13:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT: [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP13]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT: [[TMP4:%.*]] = add <4 x i32> [[TMP1]], [[TMP3]]
+; SSE4-NEXT: [[TMP5:%.*]] = mul <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[TMP4]]
+; SSE4-NEXT: store <4 x i32> [[TMP5]], ptr @out, align 16
+; SSE4-NEXT: [[TMP6:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT: [[TMP11:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SSE4-NEXT: [[TMP7:%.*]] = add <4 x i32> [[TMP6]], [[TMP11]]
+; SSE4-NEXT: [[TMP8:%.*]] = mul <4 x i32> [[TMP7]], <i32 4, i32 5, i32 6, i32 7>
+; SSE4-NEXT: store <4 x i32> [[TMP8]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+; SSE4-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %0 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %add.1 = add i32 %0, %x
+ store i32 %add.1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+ %add.2 = add i32 %1, %x
+ %mul.2 = shl i32 %add.2, 1
+ store i32 %mul.2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 8
+ %2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %add.3 = add i32 %2, %x
+ %mul.3 = mul i32 %add.3, 3
+ store i32 %mul.3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ %3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+ %add.4 = add i32 %3, %x
+ %mul.4 = shl i32 %add.4, 2
+ store i32 %mul.4, ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+ %4 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 20), align 4
+ %add.5 = add i32 %4, %x
+ %mul.5 = mul i32 %add.5, 5
+ store i32 %mul.5, ptr getelementptr inbounds nuw (i8, ptr @out, i64 20), align 4
+ %5 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 24), align 8
+ %add.6 = add i32 %5, %x
+ %mul.6 = mul i32 %add.6, 6
+ store i32 %mul.6, ptr getelementptr inbounds nuw (i8, ptr @out, i64 24), align 8
+ %6 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 28), align 4
+ %add.7 = add i32 %6, %x
+ %mul.7 = mul i32 %add.7, 7
+ store i32 %mul.7, ptr getelementptr inbounds nuw (i8, ptr @out, i64 28), align 4
+ ret void
+}
+
+; The wrapping flags of the add must not survive: the lane 0 of the add is not
+; computed from the original operands.
+define void @mul_by_index_nsw(i32 noundef %x) {
+; AVX2-LABEL: define void @mul_by_index_nsw(
+; AVX2-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; AVX2-NEXT: [[ENTRY:.*:]]
+; AVX2-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT: [[TMP1:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT: [[TMP3:%.*]] = insertelement <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; AVX2-NEXT: [[TMP10:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP11:%.*]] = shufflevector <8 x i32> [[TMP3]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT: [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP7:%.*]] = shufflevector <8 x i32> [[TMP11]], <8 x i32> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT: [[TMP8:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT: [[TMP9:%.*]] = shufflevector <8 x i32> [[TMP8]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT: [[TMP4:%.*]] = add nsw <8 x i32> [[TMP7]], [[TMP9]]
+; AVX2-NEXT: [[TMP5:%.*]] = mul <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, [[TMP4]]
+; AVX2-NEXT: store <8 x i32> [[TMP5]], ptr @out, align 16
+; AVX2-NEXT: ret void
+;
+; SSE4-LABEL: define void @mul_by_index_nsw(
+; SSE4-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; SSE4-NEXT: [[ENTRY:.*:]]
+; SSE4-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT: [[TMP1:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT: [[TMP2:%.*]] = insertelement <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; SSE4-NEXT: [[TMP9:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT: [[TMP11:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP9]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT: [[TMP12:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT: [[TMP13:%.*]] = shufflevector <4 x i32> [[TMP12]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT: [[TMP4:%.*]] = add nsw <4 x i32> [[TMP11]], [[TMP13]]
+; SSE4-NEXT: [[TMP5:%.*]] = mul <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[TMP4]]
+; SSE4-NEXT: store <4 x i32> [[TMP5]], ptr @out, align 16
+; SSE4-NEXT: [[TMP6:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT: [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SSE4-NEXT: [[TMP7:%.*]] = add nsw <4 x i32> [[TMP6]], [[TMP3]]
+; SSE4-NEXT: [[TMP8:%.*]] = mul <4 x i32> [[TMP7]], <i32 4, i32 5, i32 6, i32 7>
+; SSE4-NEXT: store <4 x i32> [[TMP8]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+; SSE4-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %0 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %add.1 = add nsw i32 %0, %x
+ store i32 %add.1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+ %add.2 = add nsw i32 %1, %x
+ %mul.2 = shl i32 %add.2, 1
+ store i32 %mul.2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 8
+ %2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %add.3 = add nsw i32 %2, %x
+ %mul.3 = mul i32 %add.3, 3
+ store i32 %mul.3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ %3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+ %add.4 = add nsw i32 %3, %x
+ %mul.4 = shl i32 %add.4, 2
+ store i32 %mul.4, ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+ %4 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 20), align 4
+ %add.5 = add nsw i32 %4, %x
+ %mul.5 = mul i32 %add.5, 5
+ store i32 %mul.5, ptr getelementptr inbounds nuw (i8, ptr @out, i64 20), align 4
+ %5 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 24), align 8
+ %add.6 = add nsw i32 %5, %x
+ %mul.6 = mul i32 %add.6, 6
+ store i32 %mul.6, ptr getelementptr inbounds nuw (i8, ptr @out, i64 24), align 8
+ %6 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 28), align 4
+ %add.7 = add nsw i32 %6, %x
+ %mul.7 = mul i32 %add.7, 7
+ store i32 %mul.7, ptr getelementptr inbounds nuw (i8, ptr @out, i64 28), align 4
+ ret void
+}
+
+; The same with the absorbing zero of the and.
+define void @and_with_zero_lane(i32 noundef %x) {
+; AVX2-LABEL: define void @and_with_zero_lane(
+; AVX2-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; AVX2-NEXT: [[ENTRY:.*:]]
+; AVX2-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT: [[TMP7:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT: [[TMP9:%.*]] = insertelement <8 x i32> <i32 -1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; AVX2-NEXT: [[TMP10:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP11:%.*]] = shufflevector <8 x i32> [[TMP9]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT: [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP7]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP11]], <8 x i32> [[TMP6]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT: [[TMP8:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT: [[TMP3:%.*]] = shufflevector <8 x i32> [[TMP8]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT: [[TMP4:%.*]] = add <8 x i32> [[TMP1]], [[TMP3]]
+; AVX2-NEXT: [[TMP5:%.*]] = and <8 x i32> <i32 0, i32 15, i32 255, i32 4095, i32 65535, i32 1048575, i32 16777215, i32 268435455>, [[TMP4]]
+; AVX2-NEXT: store <8 x i32> [[TMP5]], ptr @out, align 16
+; AVX2-NEXT: ret void
+;
+; SSE4-LABEL: define void @and_with_zero_lane(
+; SSE4-SAME: i32 noundef [[X:%.*]]) #[[ATTR0]] {
+; SSE4-NEXT: [[ENTRY:.*:]]
+; SSE4-NEXT: [[TMP0:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT: [[TMP9:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT: [[TMP2:%.*]] = insertelement <4 x i32> <i32 -1, i32 poison, i32 poison, i32 poison>, i32 [[TMP0]], i64 1
+; SSE4-NEXT: [[TMP12:%.*]] = shufflevector <2 x i32> [[TMP9]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT: [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP12]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT: [[TMP13:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT: [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP13]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT: [[TMP4:%.*]] = add <4 x i32> [[TMP1]], [[TMP3]]
+; SSE4-NEXT: [[TMP5:%.*]] = and <4 x i32> <i32 0, i32 15, i32 255, i32 4095>, [[TMP4]]
+; SSE4-NEXT: store <4 x i32> [[TMP5]], ptr @out, align 16
+; SSE4-NEXT: [[TMP6:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT: [[TMP11:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SSE4-NEXT: [[TMP7:%.*]] = add <4 x i32> [[TMP6]], [[TMP11]]
+; SSE4-NEXT: [[TMP8:%.*]] = and <4 x i32> [[TMP7]], <i32 65535, i32 1048575, i32 16777215, i32 268435455>
+; SSE4-NEXT: store <4 x i32> [[TMP8]], ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+; SSE4-NEXT: ret void
+;
+entry:
+ store i32 0, ptr @out, align 16
+ %0 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+ %add.1 = add i32 %0, %x
+ %and.1 = and i32 %add.1, 15
+ store i32 %and.1, ptr getelementptr inbounds nuw (i8, ptr @out, i64 4), align 4
+ %1 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+ %add.2 = add i32 %1, %x
+ %and.2 = and i32 %add.2, 255
+ store i32 %and.2, ptr getelementptr inbounds nuw (i8, ptr @out, i64 8), align 8
+ %2 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 12), align 4
+ %add.3 = add i32 %2, %x
+ %and.3 = and i32 %add.3, 4095
+ store i32 %and.3, ptr getelementptr inbounds nuw (i8, ptr @out, i64 12), align 4
+ %3 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+ %add.4 = add i32 %3, %x
+ %and.4 = and i32 %add.4, 65535
+ store i32 %and.4, ptr getelementptr inbounds nuw (i8, ptr @out, i64 16), align 16
+ %4 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 20), align 4
+ %add.5 = add i32 %4, %x
+ %and.5 = and i32 %add.5, 1048575
+ store i32 %and.5, ptr getelementptr inbounds nuw (i8, ptr @out, i64 20), align 4
+ %5 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 24), align 8
+ %add.6 = add i32 %5, %x
+ %and.6 = and i32 %add.6, 16777215
+ store i32 %and.6, ptr getelementptr inbounds nuw (i8, ptr @out, i64 24), align 8
+ %6 = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 28), align 4
+ %add.7 = add i32 %6, %x
+ %and.7 = and i32 %add.7, 268435455
+ store i32 %and.7, ptr getelementptr inbounds nuw (i8, ptr @out, i64 28), align 4
+ ret void
+}
+
+; The scalar x may be poison, so the broadcast is frozen to keep the lane of the
+; folded multiplication non-poison.
+define void @mul_by_index_poison_x(i32 %x) {
+; AVX2-LABEL: define void @mul_by_index_poison_x(
+; AVX2-SAME: i32 [[X:%.*]]) #[[ATTR0]] {
+; AVX2-NEXT: [[ENTRY:.*:]]
+; AVX2-NEXT: [[L1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; AVX2-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; AVX2-NEXT: [[TMP8:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; AVX2-NEXT: [[TMP2:%.*]] = insertelement <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[L1]], i64 1
+; AVX2-NEXT: [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP8]], <4 x i32> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP9:%.*]] = shufflevector <8 x i32> [[TMP2]], <8 x i32> [[TMP3]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
+; AVX2-NEXT: [[TMP10:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP9]], <8 x i32> [[TMP10]], <8 x i32> <i32 0, i32 1, i32 8, i32 9, i32 4, i32 5, i32 6, i32 7>
+; AVX2-NEXT: [[TMP7:%.*]] = insertelement <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; AVX2-NEXT: [[TMP4:%.*]] = shufflevector <8 x i32> [[TMP7]], <8 x i32> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 1>
+; AVX2-NEXT: [[TMP5:%.*]] = add <8 x i32> [[TMP1]], [[TMP4]]
+; AVX2-NEXT: [[TMP6:%.*]] = mul <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, [[TMP5]]
+; AVX2-NEXT: store <8 x i32> [[TMP6]], ptr @out, align 16
+; AVX2-NEXT: ret void
+;
+; SSE4-LABEL: define void @mul_by_index_poison_x(
+; SSE4-SAME: i32 [[X:%.*]]) #[[ATTR0]] {
+; SSE4-NEXT: [[ENTRY:.*:]]
+; SSE4-NEXT: [[L1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @in, i64 4), align 4
+; SSE4-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 8), align 8
+; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison>, i32 [[L1]], i64 1
+; SSE4-NEXT: [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; SSE4-NEXT: [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> [[TMP2]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+; SSE4-NEXT: [[TMP11:%.*]] = insertelement <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>, i32 [[X]], i64 1
+; SSE4-NEXT: [[TMP4:%.*]] = shufflevector <4 x i32> [[TMP11]], <4 x i32> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
+; SSE4-NEXT: [[TMP5:%.*]] = add <4 x i32> [[TMP1]], [[TMP4]]
+; SSE4-NEXT: [[TMP6:%.*]] = mul <4 x i32> <i32 0, i32 1, i32 2, i32 3>, [[TMP5]]
+; SSE4-NEXT: store <4 x i32> [[TMP6]], ptr @out, align 16
+; SSE4-NEXT: [[TMP7:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @in, i64 16), align 16
+; SSE4-NEXT: [[TMP12:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; SSE4-NEXT: [[TMP3:%.*]] = shufflevector <4 x i32> [[TMP12]]...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/228385
More information about the llvm-commits
mailing list