[llvm] [SLP][NFC]Add extra tests for bitcasts-based vectorization, NFC (PR #224916)

via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 20 04:36:05 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-powerpc

@llvm/pr-subscribers-llvm-transforms

Author: Alexey Bataev (alexey-bataev)

<details>
<summary>Changes</summary>



---
Full diff: https://github.com/llvm/llvm-project/pull/224916.diff


2 Files Affected:

- (added) llvm/test/Transforms/SLPVectorizer/PowerPC/extracted-subfields-be.ll (+120) 
- (modified) llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll (+74-4) 


``````````diff
diff --git a/llvm/test/Transforms/SLPVectorizer/PowerPC/extracted-subfields-be.ll b/llvm/test/Transforms/SLPVectorizer/PowerPC/extracted-subfields-be.ll
new file mode 100644
index 00000000000000..0b7db38eeb8a6c
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/PowerPC/extracted-subfields-be.ll
@@ -0,0 +1,120 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S --passes=slp-vectorizer -mtriple=powerpc64-unknown-linux-gnu < %s | FileCheck %s
+
+; The field-to-lane mapping of the bitcast to the field vector is defined for
+; little-endian targets only: the gather of extracted sub-fields must not be
+; emitted as a bitcast plus a permutation on big-endian targets.
+
+define i16 @sum8_i64(ptr %x) {
+; CHECK-LABEL: define i16 @sum8_i64(
+; CHECK-SAME: ptr [[X:%.*]]) {
+; CHECK-NEXT:  [[START:.*:]]
+; CHECK-NEXT:    [[L:%.*]] = load i64, ptr [[X]], align 8
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT:    [[B0:%.*]] = and i16 [[T0]], 255
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT:    [[B1:%.*]] = lshr i16 [[T1]], 8
+; CHECK-NEXT:    [[A1:%.*]] = add nuw nsw i16 [[B0]], [[B1]]
+; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[L]], 16
+; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT:    [[B2:%.*]] = and i16 [[T2]], 255
+; CHECK-NEXT:    [[A2:%.*]] = add nuw nsw i16 [[A1]], [[B2]]
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[L]], 24
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[B3:%.*]] = and i16 [[T3]], 255
+; CHECK-NEXT:    [[A3:%.*]] = add nuw nsw i16 [[A2]], [[B3]]
+; CHECK-NEXT:    [[S4:%.*]] = lshr i64 [[L]], 32
+; CHECK-NEXT:    [[T4:%.*]] = trunc i64 [[S4]] to i16
+; CHECK-NEXT:    [[B4:%.*]] = and i16 [[T4]], 255
+; CHECK-NEXT:    [[A4:%.*]] = add nuw nsw i16 [[A3]], [[B4]]
+; CHECK-NEXT:    [[S5:%.*]] = lshr i64 [[L]], 40
+; CHECK-NEXT:    [[T5:%.*]] = trunc i64 [[S5]] to i16
+; CHECK-NEXT:    [[B5:%.*]] = and i16 [[T5]], 255
+; CHECK-NEXT:    [[A5:%.*]] = add nuw nsw i16 [[A4]], [[B5]]
+; CHECK-NEXT:    [[S6:%.*]] = lshr i64 [[L]], 48
+; CHECK-NEXT:    [[T6:%.*]] = trunc i64 [[S6]] to i16
+; CHECK-NEXT:    [[B6:%.*]] = and i16 [[T6]], 255
+; CHECK-NEXT:    [[A6:%.*]] = add nuw nsw i16 [[A5]], [[B6]]
+; CHECK-NEXT:    [[S7:%.*]] = lshr i64 [[L]], 56
+; CHECK-NEXT:    [[B7:%.*]] = trunc i64 [[S7]] to i16
+; CHECK-NEXT:    [[A7:%.*]] = add nuw nsw i16 [[A6]], [[B7]]
+; CHECK-NEXT:    ret i16 [[A7]]
+;
+start:
+  %l = load i64, ptr %x, align 8
+  %t0 = trunc i64 %l to i16
+  %b0 = and i16 %t0, 255
+  %t1 = trunc i64 %l to i16
+  %b1 = lshr i16 %t1, 8
+  %a1 = add nuw nsw i16 %b0, %b1
+  %s2 = lshr i64 %l, 16
+  %t2 = trunc i64 %s2 to i16
+  %b2 = and i16 %t2, 255
+  %a2 = add nuw nsw i16 %a1, %b2
+  %s3 = lshr i64 %l, 24
+  %t3 = trunc i64 %s3 to i16
+  %b3 = and i16 %t3, 255
+  %a3 = add nuw nsw i16 %a2, %b3
+  %s4 = lshr i64 %l, 32
+  %t4 = trunc i64 %s4 to i16
+  %b4 = and i16 %t4, 255
+  %a4 = add nuw nsw i16 %a3, %b4
+  %s5 = lshr i64 %l, 40
+  %t5 = trunc i64 %s5 to i16
+  %b5 = and i16 %t5, 255
+  %a5 = add nuw nsw i16 %a4, %b5
+  %s6 = lshr i64 %l, 48
+  %t6 = trunc i64 %s6 to i16
+  %b6 = and i16 %t6, 255
+  %a6 = add nuw nsw i16 %a5, %b6
+  %s7 = lshr i64 %l, 56
+  %b7 = trunc i64 %s7 to i16
+  %a7 = add nuw nsw i16 %a6, %b7
+  ret i16 %a7
+}
+
+define void @store_bytes_i64(ptr %p, ptr %out) {
+; CHECK-LABEL: define void @store_bytes_i64(
+; CHECK-SAME: ptr [[P:%.*]], ptr [[OUT:%.*]]) {
+; CHECK-NEXT:  [[START:.*:]]
+; CHECK-NEXT:    [[L:%.*]] = load i64, ptr [[P]], align 8
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT:    [[B0:%.*]] = and i16 [[T0]], 255
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT:    [[B1:%.*]] = lshr i16 [[T1]], 8
+; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[L]], 16
+; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT:    [[B2:%.*]] = and i16 [[T2]], 255
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[L]], 24
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[B3:%.*]] = and i16 [[T3]], 255
+; CHECK-NEXT:    store i16 [[B0]], ptr [[OUT]], align 2
+; CHECK-NEXT:    [[O1:%.*]] = getelementptr i16, ptr [[OUT]], i64 1
+; CHECK-NEXT:    store i16 [[B1]], ptr [[O1]], align 2
+; CHECK-NEXT:    [[O2:%.*]] = getelementptr i16, ptr [[OUT]], i64 2
+; CHECK-NEXT:    store i16 [[B2]], ptr [[O2]], align 2
+; CHECK-NEXT:    [[O3:%.*]] = getelementptr i16, ptr [[OUT]], i64 3
+; CHECK-NEXT:    store i16 [[B3]], ptr [[O3]], align 2
+; CHECK-NEXT:    ret void
+;
+start:
+  %l = load i64, ptr %p, align 8
+  %t0 = trunc i64 %l to i16
+  %b0 = and i16 %t0, 255
+  %t1 = trunc i64 %l to i16
+  %b1 = lshr i16 %t1, 8
+  %s2 = lshr i64 %l, 16
+  %t2 = trunc i64 %s2 to i16
+  %b2 = and i16 %t2, 255
+  %s3 = lshr i64 %l, 24
+  %t3 = trunc i64 %s3 to i16
+  %b3 = and i16 %t3, 255
+  store i16 %b0, ptr %out, align 2
+  %o1 = getelementptr i16, ptr %out, i64 1
+  store i16 %b1, ptr %o1, align 2
+  %o2 = getelementptr i16, ptr %out, i64 2
+  store i16 %b2, ptr %o2, align 2
+  %o3 = getelementptr i16, ptr %out, i64 3
+  store i16 %b3, ptr %o3, align 2
+  ret void
+}
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll b/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
index fc2b3fb28cae6b..7c58c1881aa515 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
@@ -143,8 +143,8 @@ start:
   ret i16 %a7
 }
 
-; 16-bit sub-fields: matched, but the 4-lane vector reduction is not
-; profitable on the baseline target.
+; 16-bit sub-fields: matched and reduced as the field vector with the
+; extension to the i32 lanes.
 
 define i32 @sum4_halves_i64(ptr %p) {
 ; CHECK-LABEL: define i32 @sum4_halves_i64(
@@ -184,8 +184,8 @@ start:
   ret i32 %a3
 }
 
-; Only 4 bytes of an i32: the vector reduction is not profitable on the
-; baseline target.
+; Only 4 bytes of an i32: reduced as the field vector with the extension to
+; the i16 lanes.
 
 define i16 @sum4_i32(ptr %p) {
 ; CHECK-LABEL: define i16 @sum4_i32(
@@ -382,3 +382,73 @@ start:
   %a = add i16 %w0, %w1
   ret i16 %a
 }
+
+; Same fields as in sum8_i64, but reduced in a scrambled order: the lane
+; order of the reduction root is unobservable, the fields are emitted in the
+; natural order and no permutation is needed.
+
+define i16 @sum8_i64_scrambled(ptr %x) {
+; CHECK-LABEL: define i16 @sum8_i64_scrambled(
+; CHECK-SAME: ptr [[X:%.*]]) {
+; CHECK-NEXT:  [[START:.*:]]
+; CHECK-NEXT:    [[L:%.*]] = load i64, ptr [[X]], align 8
+; CHECK-NEXT:    [[S7:%.*]] = lshr i64 [[L]], 56
+; CHECK-NEXT:    [[S4:%.*]] = lshr i64 [[L]], 32
+; CHECK-NEXT:    [[S6:%.*]] = lshr i64 [[L]], 48
+; CHECK-NEXT:    [[S2:%.*]] = lshr i64 [[L]], 16
+; CHECK-NEXT:    [[S1:%.*]] = lshr i64 [[L]], 8
+; CHECK-NEXT:    [[S5:%.*]] = lshr i64 [[L]], 40
+; CHECK-NEXT:    [[S3:%.*]] = lshr i64 [[L]], 24
+; CHECK-NEXT:    [[B7:%.*]] = trunc i64 [[S7]] to i16
+; CHECK-NEXT:    [[T4:%.*]] = trunc i64 [[S4]] to i16
+; CHECK-NEXT:    [[T6:%.*]] = trunc i64 [[S6]] to i16
+; CHECK-NEXT:    [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[S1]] to i16
+; CHECK-NEXT:    [[T5:%.*]] = trunc i64 [[S5]] to i16
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT:    [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <8 x i16> poison, i16 [[T3]], i64 0
+; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <8 x i16> [[TMP0]], i16 [[T0]], i64 1
+; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <8 x i16> [[TMP8]], i16 [[T5]], i64 2
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <8 x i16> [[TMP9]], i16 [[T1]], i64 3
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <8 x i16> [[TMP3]], i16 [[T2]], i64 4
+; CHECK-NEXT:    [[TMP5:%.*]] = insertelement <8 x i16> [[TMP4]], i16 [[T6]], i64 5
+; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <8 x i16> [[TMP5]], i16 [[T4]], i64 6
+; CHECK-NEXT:    [[TMP7:%.*]] = insertelement <8 x i16> [[TMP6]], i16 [[B7]], i64 7
+; CHECK-NEXT:    [[TMP1:%.*]] = and <8 x i16> [[TMP7]], <i16 255, i16 255, i16 255, i16 255, i16 255, i16 255, i16 255, i16 -1>
+; CHECK-NEXT:    [[TMP2:%.*]] = call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[TMP1]])
+; CHECK-NEXT:    ret i16 [[TMP2]]
+;
+start:
+  %l = load i64, ptr %x, align 8
+  %t0 = trunc i64 %l to i16
+  %b0 = and i16 %t0, 255
+  %s1 = lshr i64 %l, 8
+  %t1 = trunc i64 %s1 to i16
+  %b1 = and i16 %t1, 255
+  %s2 = lshr i64 %l, 16
+  %t2 = trunc i64 %s2 to i16
+  %b2 = and i16 %t2, 255
+  %s3 = lshr i64 %l, 24
+  %t3 = trunc i64 %s3 to i16
+  %b3 = and i16 %t3, 255
+  %s4 = lshr i64 %l, 32
+  %t4 = trunc i64 %s4 to i16
+  %b4 = and i16 %t4, 255
+  %s5 = lshr i64 %l, 40
+  %t5 = trunc i64 %s5 to i16
+  %b5 = and i16 %t5, 255
+  %s6 = lshr i64 %l, 48
+  %t6 = trunc i64 %s6 to i16
+  %b6 = and i16 %t6, 255
+  %s7 = lshr i64 %l, 56
+  %b7 = trunc i64 %s7 to i16
+  %a1 = add nuw nsw i16 %b3, %b0
+  %a2 = add nuw nsw i16 %a1, %b5
+  %a3 = add nuw nsw i16 %a2, %b1
+  %a4 = add nuw nsw i16 %a3, %b7
+  %a5 = add nuw nsw i16 %a4, %b2
+  %a6 = add nuw nsw i16 %a5, %b6
+  %a7 = add nuw nsw i16 %a6, %b4
+  ret i16 %a7
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/224916


More information about the llvm-commits mailing list