[llvm] [SLP][NFC]Add extra tests for missed vectorization, NFC (PR #226159)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 06:20:26 PDT 2026
https://github.com/alexey-bataev created https://github.com/llvm/llvm-project/pull/226159
None
>From 3fc76e2436b4e317b3c633fb3f6f6a7a9a96607f Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Thu, 24 Sep 2026 06:20:13 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../ARM/extracted-subfields-cost.ll | 99 +++++++++++++++++++
.../SLPVectorizer/X86/extracted-subfields.ll | 74 ++++++++++++++
2 files changed, 173 insertions(+)
create mode 100644 llvm/test/Transforms/SLPVectorizer/ARM/extracted-subfields-cost.ll
diff --git a/llvm/test/Transforms/SLPVectorizer/ARM/extracted-subfields-cost.ll b/llvm/test/Transforms/SLPVectorizer/ARM/extracted-subfields-cost.ll
new file mode 100644
index 00000000000000..05627617404d53
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/ARM/extracted-subfields-cost.ll
@@ -0,0 +1,99 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -S -passes=slp-vectorizer -mtriple=thumbv8.1m.main-none-eabi -mattr=+mve -pass-remarks-output=%t < %s | FileCheck %s
+; RUN: FileCheck --input-file=%t --check-prefix=YAML %s
+
+; The shift of the top field folds into the reduction add, so the scalar cost
+; credited back for it is 0.
+
+; YAML: --- !Missed
+; YAML-NEXT: Pass: slp-vectorizer
+; YAML-NEXT: Name: HorSLPNotBeneficial
+; YAML-NEXT: Function: sum4_bytes_i32
+; YAML-NEXT: Args:
+; YAML-NEXT: - String: 'Vectorizing horizontal reduction is possible '
+; YAML-NEXT: - String: 'but not beneficial with cost '
+; YAML-NEXT: - Cost: '6'
+; YAML-NEXT: - String: ' and threshold '
+; YAML-NEXT: - Threshold: '0'
+define i32 @sum4_bytes_i32(ptr %p) {
+; CHECK-LABEL: define i32 @sum4_bytes_i32(
+; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[L:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[B0:%.*]] = and i32 [[L]], 255
+; CHECK-NEXT: [[S1:%.*]] = lshr i32 [[L]], 8
+; CHECK-NEXT: [[B1:%.*]] = and i32 [[S1]], 255
+; CHECK-NEXT: [[S2:%.*]] = lshr i32 [[L]], 16
+; CHECK-NEXT: [[B2:%.*]] = and i32 [[S2]], 255
+; CHECK-NEXT: [[B3:%.*]] = lshr i32 [[L]], 24
+; CHECK-NEXT: [[A1:%.*]] = add i32 [[B0]], [[B1]]
+; CHECK-NEXT: [[A2:%.*]] = add i32 [[A1]], [[B2]]
+; CHECK-NEXT: [[TMP2:%.*]] = add i32 [[A2]], [[B3]]
+; CHECK-NEXT: ret i32 [[TMP2]]
+;
+entry:
+ %l = load i32, ptr %p, align 4
+ %b0 = and i32 %l, 255
+ %s1 = lshr i32 %l, 8
+ %b1 = and i32 %s1, 255
+ %s2 = lshr i32 %l, 16
+ %b2 = and i32 %s2, 255
+ %b3 = lshr i32 %l, 24
+ %a1 = add i32 %b0, %b1
+ %a2 = add i32 %a1, %b2
+ %a3 = add i32 %a2, %b3
+ ret i32 %a3
+}
+
+; The extensions of the extracted fields are not extending loads, so their
+; scalar cost is credited back in full.
+
+; YAML: sum4_zext_fields
+; YAML: --- !Missed
+; YAML-NEXT: Pass: slp-vectorizer
+; YAML-NEXT: Name: NotBeneficial
+; YAML-NEXT: Function: sum4_zext_fields
+; YAML-NEXT: Args:
+; YAML-NEXT: - String: 'List vectorization was possible but not beneficial with cost '
+; YAML-NEXT: - Cost: '0'
+; YAML-NEXT: - String: ' >= '
+; YAML-NEXT: - Treshold: '0'
+define i16 @sum4_zext_fields(ptr %p) {
+; CHECK-LABEL: define i16 @sum4_zext_fields(
+; CHECK-SAME: ptr [[P:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[L:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[T0:%.*]] = trunc i32 [[L]] to i8
+; CHECK-NEXT: [[S1:%.*]] = lshr i32 [[L]], 8
+; CHECK-NEXT: [[T1:%.*]] = trunc i32 [[S1]] to i8
+; CHECK-NEXT: [[S2:%.*]] = lshr i32 [[L]], 16
+; CHECK-NEXT: [[T2:%.*]] = trunc i32 [[S2]] to i8
+; CHECK-NEXT: [[S3:%.*]] = lshr i32 [[L]], 24
+; CHECK-NEXT: [[T3:%.*]] = trunc i32 [[S3]] to i8
+; CHECK-NEXT: [[Z0:%.*]] = zext i8 [[T0]] to i16
+; CHECK-NEXT: [[Z1:%.*]] = zext i8 [[T1]] to i16
+; CHECK-NEXT: [[Z2:%.*]] = zext i8 [[T2]] to i16
+; CHECK-NEXT: [[Z3:%.*]] = zext i8 [[T3]] to i16
+; CHECK-NEXT: [[A1:%.*]] = add i16 [[Z0]], [[Z1]]
+; CHECK-NEXT: [[A2:%.*]] = add i16 [[A1]], [[Z2]]
+; CHECK-NEXT: [[TMP2:%.*]] = add i16 [[A2]], [[Z3]]
+; CHECK-NEXT: ret i16 [[TMP2]]
+;
+entry:
+ %l = load i32, ptr %p, align 4
+ %t0 = trunc i32 %l to i8
+ %s1 = lshr i32 %l, 8
+ %t1 = trunc i32 %s1 to i8
+ %s2 = lshr i32 %l, 16
+ %t2 = trunc i32 %s2 to i8
+ %s3 = lshr i32 %l, 24
+ %t3 = trunc i32 %s3 to i8
+ %z0 = zext i8 %t0 to i16
+ %z1 = zext i8 %t1 to i16
+ %z2 = zext i8 %t2 to i16
+ %z3 = zext i8 %t3 to i16
+ %a1 = add i16 %z0, %z1
+ %a2 = add i16 %a1, %z2
+ %a3 = add i16 %a2, %z3
+ ret i16 %a3
+}
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll b/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
index 7c58c1881aa515..bfc135afa7d079 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/extracted-subfields.ll
@@ -452,3 +452,77 @@ start:
%a7 = add nuw nsw i16 %a6, %b4
ret i16 %a7
}
+
+; A gathered field with an external use stays as a scalar; the lane order of
+; the reduction root gather is still unobservable, so the permutation is
+; elided.
+
+define i16 @sum8_i64_ext_use(ptr %p, ptr %out) {
+; CHECK-LABEL: define i16 @sum8_i64_ext_use(
+; CHECK-SAME: ptr [[P:%.*]], ptr [[OUT:%.*]]) {
+; CHECK-NEXT: [[START:.*:]]
+; CHECK-NEXT: [[L:%.*]] = load i64, ptr [[P]], align 8
+; CHECK-NEXT: [[T0:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT: [[B0:%.*]] = and i16 [[T0]], 255
+; CHECK-NEXT: [[T1:%.*]] = trunc i64 [[L]] to i16
+; CHECK-NEXT: [[B1:%.*]] = lshr i16 [[T1]], 8
+; CHECK-NEXT: [[S2:%.*]] = lshr i64 [[L]], 16
+; CHECK-NEXT: [[T2:%.*]] = trunc i64 [[S2]] to i16
+; CHECK-NEXT: [[B2:%.*]] = and i16 [[T2]], 255
+; CHECK-NEXT: [[S3:%.*]] = lshr i64 [[L]], 24
+; CHECK-NEXT: [[T3:%.*]] = trunc i64 [[S3]] to i16
+; CHECK-NEXT: [[B3:%.*]] = and i16 [[T3]], 255
+; CHECK-NEXT: [[S4:%.*]] = lshr i64 [[L]], 32
+; CHECK-NEXT: [[T4:%.*]] = trunc i64 [[S4]] to i16
+; CHECK-NEXT: [[B4:%.*]] = and i16 [[T4]], 255
+; CHECK-NEXT: [[S5:%.*]] = lshr i64 [[L]], 40
+; CHECK-NEXT: [[T5:%.*]] = trunc i64 [[S5]] to i16
+; CHECK-NEXT: [[B5:%.*]] = and i16 [[T5]], 255
+; CHECK-NEXT: [[S6:%.*]] = lshr i64 [[L]], 48
+; CHECK-NEXT: [[T6:%.*]] = trunc i64 [[S6]] to i16
+; CHECK-NEXT: [[B6:%.*]] = and i16 [[T6]], 255
+; CHECK-NEXT: [[S7:%.*]] = lshr i64 [[L]], 56
+; CHECK-NEXT: [[B7:%.*]] = trunc i64 [[S7]] to i16
+; CHECK-NEXT: [[A1:%.*]] = add nuw nsw i16 [[B0]], [[B1]]
+; CHECK-NEXT: [[A2:%.*]] = add nuw nsw i16 [[A1]], [[B2]]
+; CHECK-NEXT: [[A3:%.*]] = add nuw nsw i16 [[A2]], [[B3]]
+; CHECK-NEXT: [[A4:%.*]] = add nuw nsw i16 [[A3]], [[B4]]
+; CHECK-NEXT: [[A5:%.*]] = add nuw nsw i16 [[A4]], [[B5]]
+; CHECK-NEXT: [[A6:%.*]] = add nuw nsw i16 [[A5]], [[B6]]
+; CHECK-NEXT: [[TMP2:%.*]] = add nuw nsw i16 [[A6]], [[B7]]
+; CHECK-NEXT: store i16 [[B3]], ptr [[OUT]], align 2
+; CHECK-NEXT: ret i16 [[TMP2]]
+;
+start:
+ %l = load i64, ptr %p, align 8
+ %t0 = trunc i64 %l to i16
+ %b0 = and i16 %t0, 255
+ %t1 = trunc i64 %l to i16
+ %b1 = lshr i16 %t1, 8
+ %s2 = lshr i64 %l, 16
+ %t2 = trunc i64 %s2 to i16
+ %b2 = and i16 %t2, 255
+ %s3 = lshr i64 %l, 24
+ %t3 = trunc i64 %s3 to i16
+ %b3 = and i16 %t3, 255
+ %s4 = lshr i64 %l, 32
+ %t4 = trunc i64 %s4 to i16
+ %b4 = and i16 %t4, 255
+ %s5 = lshr i64 %l, 40
+ %t5 = trunc i64 %s5 to i16
+ %b5 = and i16 %t5, 255
+ %s6 = lshr i64 %l, 48
+ %t6 = trunc i64 %s6 to i16
+ %b6 = and i16 %t6, 255
+ %s7 = lshr i64 %l, 56
+ %b7 = trunc i64 %s7 to i16
+ %a1 = add nuw nsw i16 %b0, %b1
+ %a2 = add nuw nsw i16 %a1, %b2
+ %a3 = add nuw nsw i16 %a2, %b3
+ %a4 = add nuw nsw i16 %a3, %b4
+ %a5 = add nuw nsw i16 %a4, %b5
+ %a6 = add nuw nsw i16 %a5, %b6
+ %a7 = add nuw nsw i16 %a6, %b7
+ store i16 %b3, ptr %out, align 2
+ ret i16 %a7
+}
More information about the llvm-commits
mailing list