[llvm] 42b0f56 - [LLVM][CodeGen][SVE] Prefer uadalp over sabalb/sabalt. (#216301)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Aug 14 05:26:53 PDT 2026
Author: Paul Walker
Date: 2026-08-14T13:26:49+01:00
New Revision: 42b0f565056c08d190d718d9ca000b5b0d4bce1b
URL: https://github.com/llvm/llvm-project/commit/42b0f565056c08d190d718d9ca000b5b0d4bce1b
DIFF: https://github.com/llvm/llvm-project/commit/42b0f565056c08d190d718d9ca000b5b0d4bce1b.diff
LOG: [LLVM][CodeGen][SVE] Prefer uadalp over sabalb/sabalt. (#216301)
Partially reverts https://github.com/llvm/llvm-project/pull/212800
becuase for SVE2 using uadalp has better accumulator throughput than a
sabalb/sabalt sequence.
Added:
Modified:
llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td
llvm/test/CodeGen/AArch64/vector-absolute-difference.ll
Removed:
################################################################################
diff --git a/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td b/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td
index 5b6ed41e55dc1..2987b1dc85a7c 100644
--- a/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td
+++ b/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td
@@ -4231,20 +4231,6 @@ let Predicates = [HasSVE2_or_SME] in {
defm UABALB_ZZZ : sve2_int_abs
diff _accum_long<0b10, "uabalb", int_aarch64_sve_uabalb>;
defm UABALT_ZZZ : sve2_int_abs
diff _accum_long<0b11, "uabalt", int_aarch64_sve_uabalt>;
- def : Pat<(nxv8i16 (partial_reduce_umla nxv8i16:$Acc, (AArch64sabd_p (SVEAnyPredicate), nxv16i8:$Op1, nxv16i8:$Op2), (nxv16i8 (splat_vector (i32 1))))),
- (SABALT_ZZZ_H (SABALB_ZZZ_H $Acc, $Op1, $Op2), $Op1, $Op2)>;
- def : Pat<(nxv4i32 (partial_reduce_umla nxv4i32:$Acc, (AArch64sabd_p (SVEAnyPredicate), nxv8i16:$Op1, nxv8i16:$Op2), (nxv8i16 (splat_vector (i32 1))))),
- (SABALT_ZZZ_S (SABALB_ZZZ_S $Acc, $Op1, $Op2), $Op1, $Op2)>;
- def : Pat<(nxv2i64 (partial_reduce_umla nxv2i64:$Acc, (AArch64sabd_p (SVEAnyPredicate), nxv4i32:$Op1, nxv4i32:$Op2), (nxv4i32 (splat_vector (i32 1))))),
- (SABALT_ZZZ_D (SABALB_ZZZ_D $Acc, $Op1, $Op2), $Op1, $Op2)>;
-
- def : Pat<(nxv8i16 (partial_reduce_umla nxv8i16:$Acc, (AArch64uabd_p (SVEAnyPredicate), nxv16i8:$Op1, nxv16i8:$Op2), (nxv16i8 (splat_vector (i32 1))))),
- (UABALT_ZZZ_H (UABALB_ZZZ_H $Acc, $Op1, $Op2), $Op1, $Op2)>;
- def : Pat<(nxv4i32 (partial_reduce_umla nxv4i32:$Acc, (AArch64uabd_p (SVEAnyPredicate), nxv8i16:$Op1, nxv8i16:$Op2), (nxv8i16 (splat_vector (i32 1))))),
- (UABALT_ZZZ_S (UABALB_ZZZ_S $Acc, $Op1, $Op2), $Op1, $Op2)>;
- def : Pat<(nxv2i64 (partial_reduce_umla nxv2i64:$Acc, (AArch64uabd_p (SVEAnyPredicate), nxv4i32:$Op1, nxv4i32:$Op2), (nxv4i32 (splat_vector (i32 1))))),
- (UABALT_ZZZ_D (UABALB_ZZZ_D $Acc, $Op1, $Op2), $Op1, $Op2)>;
-
// SVE2 integer add/subtract long with carry
defm ADCLB_ZZZ : sve2_int_addsub_long_carry<0b00, "adclb", int_aarch64_sve_adclb>;
defm ADCLT_ZZZ : sve2_int_addsub_long_carry<0b01, "adclt", int_aarch64_sve_adclt>;
diff --git a/llvm/test/CodeGen/AArch64/vector-absolute-
diff erence.ll b/llvm/test/CodeGen/AArch64/vector-absolute-
diff erence.ll
index 99ec1710c3687..da32588cea1a0 100644
--- a/llvm/test/CodeGen/AArch64/vector-absolute-
diff erence.ll
+++ b/llvm/test/CodeGen/AArch64/vector-absolute-
diff erence.ll
@@ -31,8 +31,10 @@ define <vscale x 16 x i8> @uabs_nxv16i8(<vscale x 16 x i8> %a, <vscale x 16 x i8
define <vscale x 8 x i16> @sabs_nxv16i8_wide_add(<vscale x 8 x i16> %acc, <vscale x 16 x i8> %a, <vscale x 16 x i8> %b) {
; SVE2-LABEL: sabs_nxv16i8_wide_add:
; SVE2: // %bb.0:
-; SVE2-NEXT: sabalb z0.h, z1.b, z2.b
-; SVE2-NEXT: sabalt z0.h, z1.b, z2.b
+; SVE2-NEXT: ptrue p0.b
+; SVE2-NEXT: sabd z1.b, p0/m, z1.b, z2.b
+; SVE2-NEXT: ptrue p0.h
+; SVE2-NEXT: uadalp z0.h, p0/m, z1.b
; SVE2-NEXT: ret
;
; SVE2p3-LABEL: sabs_nxv16i8_wide_add:
@@ -50,8 +52,10 @@ define <vscale x 8 x i16> @sabs_nxv16i8_wide_add(<vscale x 8 x i16> %acc, <vscal
define <vscale x 4 x i32> @sabs_nxv8i16_wide_add(<vscale x 4 x i32> %acc, <vscale x 8 x i16> %a, <vscale x 8 x i16> %b) {
; SVE2-LABEL: sabs_nxv8i16_wide_add:
; SVE2: // %bb.0:
-; SVE2-NEXT: sabalb z0.s, z1.h, z2.h
-; SVE2-NEXT: sabalt z0.s, z1.h, z2.h
+; SVE2-NEXT: ptrue p0.h
+; SVE2-NEXT: sabd z1.h, p0/m, z1.h, z2.h
+; SVE2-NEXT: ptrue p0.s
+; SVE2-NEXT: uadalp z0.s, p0/m, z1.h
; SVE2-NEXT: ret
;
; SVE2p3-LABEL: sabs_nxv8i16_wide_add:
@@ -69,8 +73,10 @@ define <vscale x 4 x i32> @sabs_nxv8i16_wide_add(<vscale x 4 x i32> %acc, <vscal
define <vscale x 2 x i64> @sabs_nxv4i32_wide_add(<vscale x 2 x i64> %acc, <vscale x 4 x i32> %a, <vscale x 4 x i32> %b) {
; SVE2-LABEL: sabs_nxv4i32_wide_add:
; SVE2: // %bb.0:
-; SVE2-NEXT: sabalb z0.d, z1.s, z2.s
-; SVE2-NEXT: sabalt z0.d, z1.s, z2.s
+; SVE2-NEXT: ptrue p0.s
+; SVE2-NEXT: sabd z1.s, p0/m, z1.s, z2.s
+; SVE2-NEXT: ptrue p0.d
+; SVE2-NEXT: uadalp z0.d, p0/m, z1.s
; SVE2-NEXT: ret
;
; SVE2p3-LABEL: sabs_nxv4i32_wide_add:
@@ -126,8 +132,10 @@ define <16 x i8> @uabs_v16i8(<16 x i8> %a, <16 x i8> %b) {
define <vscale x 8 x i16> @uabs_nxv16i8_wide_add(<vscale x 8 x i16> %acc, <vscale x 16 x i8> %a, <vscale x 16 x i8> %b) {
; SVE2-LABEL: uabs_nxv16i8_wide_add:
; SVE2: // %bb.0:
-; SVE2-NEXT: uabalb z0.h, z1.b, z2.b
-; SVE2-NEXT: uabalt z0.h, z1.b, z2.b
+; SVE2-NEXT: ptrue p0.b
+; SVE2-NEXT: uabd z1.b, p0/m, z1.b, z2.b
+; SVE2-NEXT: ptrue p0.h
+; SVE2-NEXT: uadalp z0.h, p0/m, z1.b
; SVE2-NEXT: ret
;
; SVE2p3-LABEL: uabs_nxv16i8_wide_add:
@@ -145,8 +153,10 @@ define <vscale x 8 x i16> @uabs_nxv16i8_wide_add(<vscale x 8 x i16> %acc, <vscal
define <vscale x 4 x i32> @uabs_nxv8i16_wide_add(<vscale x 4 x i32> %acc, <vscale x 8 x i16> %a, <vscale x 8 x i16> %b) {
; SVE2-LABEL: uabs_nxv8i16_wide_add:
; SVE2: // %bb.0:
-; SVE2-NEXT: uabalb z0.s, z1.h, z2.h
-; SVE2-NEXT: uabalt z0.s, z1.h, z2.h
+; SVE2-NEXT: ptrue p0.h
+; SVE2-NEXT: uabd z1.h, p0/m, z1.h, z2.h
+; SVE2-NEXT: ptrue p0.s
+; SVE2-NEXT: uadalp z0.s, p0/m, z1.h
; SVE2-NEXT: ret
;
; SVE2p3-LABEL: uabs_nxv8i16_wide_add:
@@ -164,8 +174,10 @@ define <vscale x 4 x i32> @uabs_nxv8i16_wide_add(<vscale x 4 x i32> %acc, <vscal
define <vscale x 2 x i64> @uabs_nxv4i32_wide_add(<vscale x 2 x i64> %acc, <vscale x 4 x i32> %a, <vscale x 4 x i32> %b) {
; SVE2-LABEL: uabs_nxv4i32_wide_add:
; SVE2: // %bb.0:
-; SVE2-NEXT: uabalb z0.d, z1.s, z2.s
-; SVE2-NEXT: uabalt z0.d, z1.s, z2.s
+; SVE2-NEXT: ptrue p0.s
+; SVE2-NEXT: uabd z1.s, p0/m, z1.s, z2.s
+; SVE2-NEXT: ptrue p0.d
+; SVE2-NEXT: uadalp z0.d, p0/m, z1.s
; SVE2-NEXT: ret
;
; SVE2p3-LABEL: uabs_nxv4i32_wide_add:
More information about the llvm-commits
mailing list