[llvm] 42b0f56 - [LLVM][CodeGen][SVE] Prefer uadalp over sabalb/sabalt. (#216301)

via llvm-commits llvm-commits at lists.llvm.org
Fri Aug 14 05:26:53 PDT 2026


Author: Paul Walker
Date: 2026-08-14T13:26:49+01:00
New Revision: 42b0f565056c08d190d718d9ca000b5b0d4bce1b

URL: https://github.com/llvm/llvm-project/commit/42b0f565056c08d190d718d9ca000b5b0d4bce1b
DIFF: https://github.com/llvm/llvm-project/commit/42b0f565056c08d190d718d9ca000b5b0d4bce1b.diff

LOG: [LLVM][CodeGen][SVE] Prefer uadalp over sabalb/sabalt. (#216301)

Partially reverts https://github.com/llvm/llvm-project/pull/212800
becuase for SVE2 using uadalp has better accumulator throughput than a
sabalb/sabalt sequence.

Added: 
    

Modified: 
    llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td
    llvm/test/CodeGen/AArch64/vector-absolute-difference.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td b/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td
index 5b6ed41e55dc1..2987b1dc85a7c 100644
--- a/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td
+++ b/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td
@@ -4231,20 +4231,6 @@ let Predicates = [HasSVE2_or_SME] in {
   defm UABALB_ZZZ : sve2_int_abs
diff _accum_long<0b10, "uabalb", int_aarch64_sve_uabalb>;
   defm UABALT_ZZZ : sve2_int_abs
diff _accum_long<0b11, "uabalt", int_aarch64_sve_uabalt>;
 
-  def : Pat<(nxv8i16 (partial_reduce_umla nxv8i16:$Acc, (AArch64sabd_p (SVEAnyPredicate), nxv16i8:$Op1, nxv16i8:$Op2), (nxv16i8 (splat_vector (i32 1))))),
-            (SABALT_ZZZ_H (SABALB_ZZZ_H $Acc, $Op1, $Op2), $Op1, $Op2)>;
-  def : Pat<(nxv4i32 (partial_reduce_umla nxv4i32:$Acc, (AArch64sabd_p (SVEAnyPredicate), nxv8i16:$Op1, nxv8i16:$Op2), (nxv8i16 (splat_vector (i32 1))))),
-            (SABALT_ZZZ_S (SABALB_ZZZ_S $Acc, $Op1, $Op2), $Op1, $Op2)>;
-  def : Pat<(nxv2i64 (partial_reduce_umla nxv2i64:$Acc, (AArch64sabd_p (SVEAnyPredicate), nxv4i32:$Op1, nxv4i32:$Op2), (nxv4i32 (splat_vector (i32 1))))),
-            (SABALT_ZZZ_D (SABALB_ZZZ_D $Acc, $Op1, $Op2), $Op1, $Op2)>;
-
-  def : Pat<(nxv8i16 (partial_reduce_umla nxv8i16:$Acc, (AArch64uabd_p (SVEAnyPredicate), nxv16i8:$Op1, nxv16i8:$Op2), (nxv16i8 (splat_vector (i32 1))))),
-            (UABALT_ZZZ_H (UABALB_ZZZ_H $Acc, $Op1, $Op2), $Op1, $Op2)>;
-  def : Pat<(nxv4i32 (partial_reduce_umla nxv4i32:$Acc, (AArch64uabd_p (SVEAnyPredicate), nxv8i16:$Op1, nxv8i16:$Op2), (nxv8i16 (splat_vector (i32 1))))),
-            (UABALT_ZZZ_S (UABALB_ZZZ_S $Acc, $Op1, $Op2), $Op1, $Op2)>;
-  def : Pat<(nxv2i64 (partial_reduce_umla nxv2i64:$Acc, (AArch64uabd_p (SVEAnyPredicate), nxv4i32:$Op1, nxv4i32:$Op2), (nxv4i32 (splat_vector (i32 1))))),
-            (UABALT_ZZZ_D (UABALB_ZZZ_D $Acc, $Op1, $Op2), $Op1, $Op2)>;
-
   // SVE2 integer add/subtract long with carry
   defm ADCLB_ZZZ : sve2_int_addsub_long_carry<0b00, "adclb", int_aarch64_sve_adclb>;
   defm ADCLT_ZZZ : sve2_int_addsub_long_carry<0b01, "adclt", int_aarch64_sve_adclt>;

diff  --git a/llvm/test/CodeGen/AArch64/vector-absolute-
diff erence.ll b/llvm/test/CodeGen/AArch64/vector-absolute-
diff erence.ll
index 99ec1710c3687..da32588cea1a0 100644
--- a/llvm/test/CodeGen/AArch64/vector-absolute-
diff erence.ll
+++ b/llvm/test/CodeGen/AArch64/vector-absolute-
diff erence.ll
@@ -31,8 +31,10 @@ define <vscale x 16 x i8> @uabs_nxv16i8(<vscale x 16 x i8> %a, <vscale x 16 x i8
 define <vscale x 8 x i16> @sabs_nxv16i8_wide_add(<vscale x 8 x i16> %acc, <vscale x 16 x i8> %a, <vscale x 16 x i8> %b) {
 ; SVE2-LABEL: sabs_nxv16i8_wide_add:
 ; SVE2:       // %bb.0:
-; SVE2-NEXT:    sabalb z0.h, z1.b, z2.b
-; SVE2-NEXT:    sabalt z0.h, z1.b, z2.b
+; SVE2-NEXT:    ptrue p0.b
+; SVE2-NEXT:    sabd z1.b, p0/m, z1.b, z2.b
+; SVE2-NEXT:    ptrue p0.h
+; SVE2-NEXT:    uadalp z0.h, p0/m, z1.b
 ; SVE2-NEXT:    ret
 ;
 ; SVE2p3-LABEL: sabs_nxv16i8_wide_add:
@@ -50,8 +52,10 @@ define <vscale x 8 x i16> @sabs_nxv16i8_wide_add(<vscale x 8 x i16> %acc, <vscal
 define <vscale x 4 x i32> @sabs_nxv8i16_wide_add(<vscale x 4 x i32> %acc, <vscale x 8 x i16> %a, <vscale x 8 x i16> %b) {
 ; SVE2-LABEL: sabs_nxv8i16_wide_add:
 ; SVE2:       // %bb.0:
-; SVE2-NEXT:    sabalb z0.s, z1.h, z2.h
-; SVE2-NEXT:    sabalt z0.s, z1.h, z2.h
+; SVE2-NEXT:    ptrue p0.h
+; SVE2-NEXT:    sabd z1.h, p0/m, z1.h, z2.h
+; SVE2-NEXT:    ptrue p0.s
+; SVE2-NEXT:    uadalp z0.s, p0/m, z1.h
 ; SVE2-NEXT:    ret
 ;
 ; SVE2p3-LABEL: sabs_nxv8i16_wide_add:
@@ -69,8 +73,10 @@ define <vscale x 4 x i32> @sabs_nxv8i16_wide_add(<vscale x 4 x i32> %acc, <vscal
 define <vscale x 2 x i64> @sabs_nxv4i32_wide_add(<vscale x 2 x i64> %acc, <vscale x 4 x i32> %a, <vscale x 4 x i32> %b) {
 ; SVE2-LABEL: sabs_nxv4i32_wide_add:
 ; SVE2:       // %bb.0:
-; SVE2-NEXT:    sabalb z0.d, z1.s, z2.s
-; SVE2-NEXT:    sabalt z0.d, z1.s, z2.s
+; SVE2-NEXT:    ptrue p0.s
+; SVE2-NEXT:    sabd z1.s, p0/m, z1.s, z2.s
+; SVE2-NEXT:    ptrue p0.d
+; SVE2-NEXT:    uadalp z0.d, p0/m, z1.s
 ; SVE2-NEXT:    ret
 ;
 ; SVE2p3-LABEL: sabs_nxv4i32_wide_add:
@@ -126,8 +132,10 @@ define <16 x i8> @uabs_v16i8(<16 x i8> %a, <16 x i8> %b) {
 define <vscale x 8 x i16> @uabs_nxv16i8_wide_add(<vscale x 8 x i16> %acc, <vscale x 16 x i8> %a, <vscale x 16 x i8> %b) {
 ; SVE2-LABEL: uabs_nxv16i8_wide_add:
 ; SVE2:       // %bb.0:
-; SVE2-NEXT:    uabalb z0.h, z1.b, z2.b
-; SVE2-NEXT:    uabalt z0.h, z1.b, z2.b
+; SVE2-NEXT:    ptrue p0.b
+; SVE2-NEXT:    uabd z1.b, p0/m, z1.b, z2.b
+; SVE2-NEXT:    ptrue p0.h
+; SVE2-NEXT:    uadalp z0.h, p0/m, z1.b
 ; SVE2-NEXT:    ret
 ;
 ; SVE2p3-LABEL: uabs_nxv16i8_wide_add:
@@ -145,8 +153,10 @@ define <vscale x 8 x i16> @uabs_nxv16i8_wide_add(<vscale x 8 x i16> %acc, <vscal
 define <vscale x 4 x i32> @uabs_nxv8i16_wide_add(<vscale x 4 x i32> %acc, <vscale x 8 x i16> %a, <vscale x 8 x i16> %b) {
 ; SVE2-LABEL: uabs_nxv8i16_wide_add:
 ; SVE2:       // %bb.0:
-; SVE2-NEXT:    uabalb z0.s, z1.h, z2.h
-; SVE2-NEXT:    uabalt z0.s, z1.h, z2.h
+; SVE2-NEXT:    ptrue p0.h
+; SVE2-NEXT:    uabd z1.h, p0/m, z1.h, z2.h
+; SVE2-NEXT:    ptrue p0.s
+; SVE2-NEXT:    uadalp z0.s, p0/m, z1.h
 ; SVE2-NEXT:    ret
 ;
 ; SVE2p3-LABEL: uabs_nxv8i16_wide_add:
@@ -164,8 +174,10 @@ define <vscale x 4 x i32> @uabs_nxv8i16_wide_add(<vscale x 4 x i32> %acc, <vscal
 define <vscale x 2 x i64> @uabs_nxv4i32_wide_add(<vscale x 2 x i64> %acc, <vscale x 4 x i32> %a, <vscale x 4 x i32> %b) {
 ; SVE2-LABEL: uabs_nxv4i32_wide_add:
 ; SVE2:       // %bb.0:
-; SVE2-NEXT:    uabalb z0.d, z1.s, z2.s
-; SVE2-NEXT:    uabalt z0.d, z1.s, z2.s
+; SVE2-NEXT:    ptrue p0.s
+; SVE2-NEXT:    uabd z1.s, p0/m, z1.s, z2.s
+; SVE2-NEXT:    ptrue p0.d
+; SVE2-NEXT:    uadalp z0.d, p0/m, z1.s
 ; SVE2-NEXT:    ret
 ;
 ; SVE2p3-LABEL: uabs_nxv4i32_wide_add:


        


More information about the llvm-commits mailing list