[llvm] [SLP][NFC]Add extra tests for bitcasts-based vectorization, NFC (PR #224916)
via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 20 04:36:05 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-powerpc
@llvm/pr-subscribers-llvm-transforms
Author: Alexey Bataev (alexey-bataev)
<details>
<summary>Changes</summary>
---
Full diff: https://github.com/llvm/llvm-project/pull/224916.diff
2 Files Affected:
- (added) llvm/test/Transforms/SLPVectorizer/PowerPC/extracted-subfields-be.ll (+120)
- (modified) llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll (+74-4)
``````````diff
diff --git a/llvm/test/Transforms/SLPVectorizer/PowerPC/extracted-subfields-be.ll b/llvm/test/Transforms/SLPVectorizer/PowerPC/extracted-subfields-be.ll
new file mode 100644
index 00000000000000..0b7db38eeb8a6c
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/PowerPC/extracted-subfields-be.ll
@@ -0,0 +1,120 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S --passes=slp-vectorizer -mtriple=powerpc64-unknown-linux-gnu < %s | FileCheck %s
+
+; The field-to-lane mapping of the bitcast to the field vector is defined for
+; little-endian targets only: the gather of extracted sub-fields must not be
+; emitted as a bitcast plus a permutation on big-endian targets.
+
+define i16 @sum8_i64(ptr %x) {
+; CHECK-LABEL: define i16 @sum8_i64(
+; CHECK-SAME: ptr [[X:%.*]]) {
+; CHECK-NEXT: [[START:.*:]]
+; CHECK-NEXT: [[L:%.*]] = load i64, ptr [[X]], align 8
+; CHECK-NEXT: [[T0:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT: [[B0:%.*]] = and i16 [[T0]], 255
+; CHECK-NEXT: [[T1:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT: [[B1:%.*]] = lshr i16 [[T1]], 8
+; CHECK-NEXT: [[A1:%.*]] = add nuw nsw i16 [[B0]], [[B1]]
+; CHECK-NEXT: [[S2:%.*]] = lshr i64 [[L]], 16
+; CHECK-NEXT: [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT: [[B2:%.*]] = and i16 [[T2]], 255
+; CHECK-NEXT: [[A2:%.*]] = add nuw nsw i16 [[A1]], [[B2]]
+; CHECK-NEXT: [[S3:%.*]] = lshr i64 [[L]], 24
+; CHECK-NEXT: [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT: [[B3:%.*]] = and i16 [[T3]], 255
+; CHECK-NEXT: [[A3:%.*]] = add nuw nsw i16 [[A2]], [[B3]]
+; CHECK-NEXT: [[S4:%.*]] = lshr i64 [[L]], 32
+; CHECK-NEXT: [[T4:%.*]] = trunc i64 [[S4]] to i16
+; CHECK-NEXT: [[B4:%.*]] = and i16 [[T4]], 255
+; CHECK-NEXT: [[A4:%.*]] = add nuw nsw i16 [[A3]], [[B4]]
+; CHECK-NEXT: [[S5:%.*]] = lshr i64 [[L]], 40
+; CHECK-NEXT: [[T5:%.*]] = trunc i64 [[S5]] to i16
+; CHECK-NEXT: [[B5:%.*]] = and i16 [[T5]], 255
+; CHECK-NEXT: [[A5:%.*]] = add nuw nsw i16 [[A4]], [[B5]]
+; CHECK-NEXT: [[S6:%.*]] = lshr i64 [[L]], 48
+; CHECK-NEXT: [[T6:%.*]] = trunc i64 [[S6]] to i16
+; CHECK-NEXT: [[B6:%.*]] = and i16 [[T6]], 255
+; CHECK-NEXT: [[A6:%.*]] = add nuw nsw i16 [[A5]], [[B6]]
+; CHECK-NEXT: [[S7:%.*]] = lshr i64 [[L]], 56
+; CHECK-NEXT: [[B7:%.*]] = trunc i64 [[S7]] to i16
+; CHECK-NEXT: [[A7:%.*]] = add nuw nsw i16 [[A6]], [[B7]]
+; CHECK-NEXT: ret i16 [[A7]]
+;
+start:
+ %l = load i64, ptr %x, align 8
+ %t0 = trunc i64 %l to i16
+ %b0 = and i16 %t0, 255
+ %t1 = trunc i64 %l to i16
+ %b1 = lshr i16 %t1, 8
+ %a1 = add nuw nsw i16 %b0, %b1
+ %s2 = lshr i64 %l, 16
+ %t2 = trunc i64 %s2 to i16
+ %b2 = and i16 %t2, 255
+ %a2 = add nuw nsw i16 %a1, %b2
+ %s3 = lshr i64 %l, 24
+ %t3 = trunc i64 %s3 to i16
+ %b3 = and i16 %t3, 255
+ %a3 = add nuw nsw i16 %a2, %b3
+ %s4 = lshr i64 %l, 32
+ %t4 = trunc i64 %s4 to i16
+ %b4 = and i16 %t4, 255
+ %a4 = add nuw nsw i16 %a3, %b4
+ %s5 = lshr i64 %l, 40
+ %t5 = trunc i64 %s5 to i16
+ %b5 = and i16 %t5, 255
+ %a5 = add nuw nsw i16 %a4, %b5
+ %s6 = lshr i64 %l, 48
+ %t6 = trunc i64 %s6 to i16
+ %b6 = and i16 %t6, 255
+ %a6 = add nuw nsw i16 %a5, %b6
+ %s7 = lshr i64 %l, 56
+ %b7 = trunc i64 %s7 to i16
+ %a7 = add nuw nsw i16 %a6, %b7
+ ret i16 %a7
+}
+
+define void @store_bytes_i64(ptr %p, ptr %out) {
+; CHECK-LABEL: define void @store_bytes_i64(
+; CHECK-SAME: ptr [[P:%.*]], ptr [[OUT:%.*]]) {
+; CHECK-NEXT: [[START:.*:]]
+; CHECK-NEXT: [[L:%.*]] = load i64, ptr [[P]], align 8
+; CHECK-NEXT: [[T0:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT: [[B0:%.*]] = and i16 [[T0]], 255
+; CHECK-NEXT: [[T1:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT: [[B1:%.*]] = lshr i16 [[T1]], 8
+; CHECK-NEXT: [[S2:%.*]] = lshr i64 [[L]], 16
+; CHECK-NEXT: [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT: [[B2:%.*]] = and i16 [[T2]], 255
+; CHECK-NEXT: [[S3:%.*]] = lshr i64 [[L]], 24
+; CHECK-NEXT: [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT: [[B3:%.*]] = and i16 [[T3]], 255
+; CHECK-NEXT: store i16 [[B0]], ptr [[OUT]], align 2
+; CHECK-NEXT: [[O1:%.*]] = getelementptr i16, ptr [[OUT]], i64 1
+; CHECK-NEXT: store i16 [[B1]], ptr [[O1]], align 2
+; CHECK-NEXT: [[O2:%.*]] = getelementptr i16, ptr [[OUT]], i64 2
+; CHECK-NEXT: store i16 [[B2]], ptr [[O2]], align 2
+; CHECK-NEXT: [[O3:%.*]] = getelementptr i16, ptr [[OUT]], i64 3
+; CHECK-NEXT: store i16 [[B3]], ptr [[O3]], align 2
+; CHECK-NEXT: ret void
+;
+start:
+ %l = load i64, ptr %p, align 8
+ %t0 = trunc i64 %l to i16
+ %b0 = and i16 %t0, 255
+ %t1 = trunc i64 %l to i16
+ %b1 = lshr i16 %t1, 8
+ %s2 = lshr i64 %l, 16
+ %t2 = trunc i64 %s2 to i16
+ %b2 = and i16 %t2, 255
+ %s3 = lshr i64 %l, 24
+ %t3 = trunc i64 %s3 to i16
+ %b3 = and i16 %t3, 255
+ store i16 %b0, ptr %out, align 2
+ %o1 = getelementptr i16, ptr %out, i64 1
+ store i16 %b1, ptr %o1, align 2
+ %o2 = getelementptr i16, ptr %out, i64 2
+ store i16 %b2, ptr %o2, align 2
+ %o3 = getelementptr i16, ptr %out, i64 3
+ store i16 %b3, ptr %o3, align 2
+ ret void
+}
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll b/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
index fc2b3fb28cae6b..7c58c1881aa515 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
@@ -143,8 +143,8 @@ start:
ret i16 %a7
}
-; 16-bit sub-fields: matched, but the 4-lane vector reduction is not
-; profitable on the baseline target.
+; 16-bit sub-fields: matched and reduced as the field vector with the
+; extension to the i32 lanes.
define i32 @sum4_halves_i64(ptr %p) {
; CHECK-LABEL: define i32 @sum4_halves_i64(
@@ -184,8 +184,8 @@ start:
ret i32 %a3
}
-; Only 4 bytes of an i32: the vector reduction is not profitable on the
-; baseline target.
+; Only 4 bytes of an i32: reduced as the field vector with the extension to
+; the i16 lanes.
define i16 @sum4_i32(ptr %p) {
; CHECK-LABEL: define i16 @sum4_i32(
@@ -382,3 +382,73 @@ start:
%a = add i16 %w0, %w1
ret i16 %a
}
+
+; Same fields as in sum8_i64, but reduced in a scrambled order: the lane
+; order of the reduction root is unobservable, the fields are emitted in the
+; natural order and no permutation is needed.
+
+define i16 @sum8_i64_scrambled(ptr %x) {
+; CHECK-LABEL: define i16 @sum8_i64_scrambled(
+; CHECK-SAME: ptr [[X:%.*]]) {
+; CHECK-NEXT: [[START:.*:]]
+; CHECK-NEXT: [[L:%.*]] = load i64, ptr [[X]], align 8
+; CHECK-NEXT: [[S7:%.*]] = lshr i64 [[L]], 56
+; CHECK-NEXT: [[S4:%.*]] = lshr i64 [[L]], 32
+; CHECK-NEXT: [[S6:%.*]] = lshr i64 [[L]], 48
+; CHECK-NEXT: [[S2:%.*]] = lshr i64 [[L]], 16
+; CHECK-NEXT: [[S1:%.*]] = lshr i64 [[L]], 8
+; CHECK-NEXT: [[S5:%.*]] = lshr i64 [[L]], 40
+; CHECK-NEXT: [[S3:%.*]] = lshr i64 [[L]], 24
+; CHECK-NEXT: [[B7:%.*]] = trunc i64 [[S7]] to i16
+; CHECK-NEXT: [[T4:%.*]] = trunc i64 [[S4]] to i16
+; CHECK-NEXT: [[T6:%.*]] = trunc i64 [[S6]] to i16
+; CHECK-NEXT: [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT: [[T1:%.*]] = trunc i64 [[S1]] to i16
+; CHECK-NEXT: [[T5:%.*]] = trunc i64 [[S5]] to i16
+; CHECK-NEXT: [[T0:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT: [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <8 x i16> poison, i16 [[T3]], i64 0
+; CHECK-NEXT: [[TMP8:%.*]] = insertelement <8 x i16> [[TMP0]], i16 [[T0]], i64 1
+; CHECK-NEXT: [[TMP9:%.*]] = insertelement <8 x i16> [[TMP8]], i16 [[T5]], i64 2
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <8 x i16> [[TMP9]], i16 [[T1]], i64 3
+; CHECK-NEXT: [[TMP4:%.*]] = insertelement <8 x i16> [[TMP3]], i16 [[T2]], i64 4
+; CHECK-NEXT: [[TMP5:%.*]] = insertelement <8 x i16> [[TMP4]], i16 [[T6]], i64 5
+; CHECK-NEXT: [[TMP6:%.*]] = insertelement <8 x i16> [[TMP5]], i16 [[T4]], i64 6
+; CHECK-NEXT: [[TMP7:%.*]] = insertelement <8 x i16> [[TMP6]], i16 [[B7]], i64 7
+; CHECK-NEXT: [[TMP1:%.*]] = and <8 x i16> [[TMP7]], <i16 255, i16 255, i16 255, i16 255, i16 255, i16 255, i16 255, i16 -1>
+; CHECK-NEXT: [[TMP2:%.*]] = call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[TMP1]])
+; CHECK-NEXT: ret i16 [[TMP2]]
+;
+start:
+ %l = load i64, ptr %x, align 8
+ %t0 = trunc i64 %l to i16
+ %b0 = and i16 %t0, 255
+ %s1 = lshr i64 %l, 8
+ %t1 = trunc i64 %s1 to i16
+ %b1 = and i16 %t1, 255
+ %s2 = lshr i64 %l, 16
+ %t2 = trunc i64 %s2 to i16
+ %b2 = and i16 %t2, 255
+ %s3 = lshr i64 %l, 24
+ %t3 = trunc i64 %s3 to i16
+ %b3 = and i16 %t3, 255
+ %s4 = lshr i64 %l, 32
+ %t4 = trunc i64 %s4 to i16
+ %b4 = and i16 %t4, 255
+ %s5 = lshr i64 %l, 40
+ %t5 = trunc i64 %s5 to i16
+ %b5 = and i16 %t5, 255
+ %s6 = lshr i64 %l, 48
+ %t6 = trunc i64 %s6 to i16
+ %b6 = and i16 %t6, 255
+ %s7 = lshr i64 %l, 56
+ %b7 = trunc i64 %s7 to i16
+ %a1 = add nuw nsw i16 %b3, %b0
+ %a2 = add nuw nsw i16 %a1, %b5
+ %a3 = add nuw nsw i16 %a2, %b1
+ %a4 = add nuw nsw i16 %a3, %b7
+ %a5 = add nuw nsw i16 %a4, %b2
+ %a6 = add nuw nsw i16 %a5, %b6
+ %a7 = add nuw nsw i16 %a6, %b4
+ ret i16 %a7
+}
``````````
</details>
https://github.com/llvm/llvm-project/pull/224916
More information about the llvm-commits
mailing list