[llvm] [AArch64] Select ADDHN instead of SHRN when possible (PR #225094)

via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 22 02:26:22 PDT 2026


https://github.com/KRM7 updated https://github.com/llvm/llvm-project/pull/225094

>From 869a7e3dfbc26d6354714c2e10b3918efc48bc1a Mon Sep 17 00:00:00 2001
From: Krisztian Rugasi <krisztian.rugasi at arm.com>
Date: Mon, 21 Sep 2026 11:00:24 +0000
Subject: [PATCH] [AArch64] Select ADDHN instead of SHRN when possible

The throughput of the ADDHN instruction is either double or the same
as the throughput of the SHRN instruction almost everywhere, so it's
better to use ADDHN when possible.
---
 llvm/lib/Target/AArch64/AArch64InstrInfo.td | 23 +++++++
 llvm/test/CodeGen/AArch64/arm64-vadd.ll     | 69 +++++++++++++++++++++
 llvm/test/CodeGen/AArch64/clmul-fixed.ll    |  4 +-
 llvm/test/CodeGen/AArch64/neon-rshrn.ll     | 12 ++--
 4 files changed, 100 insertions(+), 8 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.td b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
index 0d8099b911894d..9b3a5d80c9f178 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.td
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
@@ -7931,6 +7931,29 @@ defm : AddSubHNPatterns<ADDHNv2i64_v2i32, ADDHNv2i64_v4i32,
                         SUBHNv2i64_v2i32, SUBHNv2i64_v4i32,
                         v2i32, v2i64, 32>;
 
+def : Pat<(v8i8 (trunc (AArch64vlshr (v8i16 V128:$Rn), (i32 7)))),
+          (ADDHNv8i16_v8i8 V128:$Rn, V128:$Rn)>;
+def : Pat<(v4i16 (trunc (AArch64vlshr (v4i32 V128:$Rn), (i32 15)))),
+          (ADDHNv4i32_v4i16 V128:$Rn, V128:$Rn)>;
+def : Pat<(v2i32 (trunc (AArch64vlshr (v2i64 V128:$Rn), (i32 31)))),
+          (ADDHNv2i64_v2i32 V128:$Rn, V128:$Rn)>;
+
+def : Pat<(v16i8 (concat_vectors (v8i8 V64:$Rd),
+                                 (trunc (AArch64vlshr (v8i16 V128:$Rn),
+                                                      (i32 7))))),
+          (ADDHNv8i16_v16i8 (INSERT_SUBREG (IMPLICIT_DEF), V64:$Rd, dsub),
+                           V128:$Rn, V128:$Rn)>;
+def : Pat<(v8i16 (concat_vectors (v4i16 V64:$Rd),
+                                 (trunc (AArch64vlshr (v4i32 V128:$Rn),
+                                                      (i32 15))))),
+          (ADDHNv4i32_v8i16 (INSERT_SUBREG (IMPLICIT_DEF), V64:$Rd, dsub),
+                           V128:$Rn, V128:$Rn)>;
+def : Pat<(v4i32 (concat_vectors (v2i32 V64:$Rd),
+                                 (trunc (AArch64vlshr (v2i64 V128:$Rn),
+                                                      (i32 31))))),
+          (ADDHNv2i64_v4i32 (INSERT_SUBREG (IMPLICIT_DEF), V64:$Rd, dsub),
+                           V128:$Rn, V128:$Rn)>;
+
 //----------------------------------------------------------------------------
 // AdvSIMD bitwise extract from vector instruction.
 //----------------------------------------------------------------------------
diff --git a/llvm/test/CodeGen/AArch64/arm64-vadd.ll b/llvm/test/CodeGen/AArch64/arm64-vadd.ll
index 3cf01150712c94..b65397b318b886 100644
--- a/llvm/test/CodeGen/AArch64/arm64-vadd.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-vadd.ll
@@ -1351,3 +1351,72 @@ define <4 x i32> @neg_narrow_i32(<4 x i64> %a) {
   %vshrn_n = trunc nuw <4 x i64> %s to <4 x i32>
   ret <4 x i32> %vshrn_n
 }
+
+define <8 x i8> @addhn_shift_i8(<8 x i16> %a) nounwind {
+; CHECK-LABEL: addhn_shift_i8:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    addhn v0.8b, v0.8h, v0.8h
+; CHECK-NEXT:    ret
+  %1 = lshr <8 x i16> %a, splat (i16 7)
+  %2 = trunc <8 x i16> %1 to <8 x i8>
+  ret <8 x i8> %2
+}
+
+define <4 x i16> @addhn_shift_i16(<4 x i32> %a) nounwind {
+; CHECK-LABEL: addhn_shift_i16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    addhn v0.4h, v0.4s, v0.4s
+; CHECK-NEXT:    ret
+  %1 = lshr <4 x i32> %a, splat (i32 15)
+  %2 = trunc <4 x i32> %1 to <4 x i16>
+  ret <4 x i16> %2
+}
+
+define <2 x i32> @addhn_shift_i32(<2 x i64> %a) nounwind {
+; CHECK-LABEL: addhn_shift_i32:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    addhn v0.2s, v0.2d, v0.2d
+; CHECK-NEXT:    ret
+  %1 = lshr <2 x i64> %a, splat (i64 31)
+  %2 = trunc <2 x i64> %1 to <2 x i32>
+  ret <2 x i32> %2
+}
+
+define <16 x i8> @addhn2_shift_i8(<8 x i16> %a, <8 x i8> %b) nounwind {
+; CHECK-LABEL: addhn2_shift_i8:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    // kill: def $d1 killed $d1 def $q1
+; CHECK-NEXT:    addhn2 v1.16b, v0.8h, v0.8h
+; CHECK-NEXT:    mov v0.16b, v1.16b
+; CHECK-NEXT:    ret
+  %1 = lshr <8 x i16> %a, splat (i16 7)
+  %2 = trunc <8 x i16> %1 to <8 x i8>
+  %3 = shufflevector <8 x i8> %b, <8 x i8> %2, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+  ret <16 x i8> %3
+}
+
+define <8 x i16> @addhn2_shift_i16(<4 x i32> %a, <4 x i16> %b) nounwind {
+; CHECK-LABEL: addhn2_shift_i16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    // kill: def $d1 killed $d1 def $q1
+; CHECK-NEXT:    addhn2 v1.8h, v0.4s, v0.4s
+; CHECK-NEXT:    mov v0.16b, v1.16b
+; CHECK-NEXT:    ret
+  %1 = lshr <4 x i32> %a, splat (i32 15)
+  %2 = trunc <4 x i32> %1 to <4 x i16>
+  %3 = shufflevector <4 x i16> %b, <4 x i16> %2, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+  ret <8 x i16> %3
+}
+
+define <4 x i32> @addhn2_shift_i32(<2 x i64> %a, <2 x i32> %b) nounwind {
+; CHECK-LABEL: addhn2_shift_i32:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    // kill: def $d1 killed $d1 def $q1
+; CHECK-NEXT:    addhn2 v1.4s, v0.2d, v0.2d
+; CHECK-NEXT:    mov v0.16b, v1.16b
+; CHECK-NEXT:    ret
+  %1 = lshr <2 x i64> %a, splat (i64 31)
+  %2 = trunc <2 x i64> %1 to <2 x i32>
+  %3 = shufflevector <2 x i32> %b, <2 x i32> %2, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+  ret <4 x i32> %3
+}
diff --git a/llvm/test/CodeGen/AArch64/clmul-fixed.ll b/llvm/test/CodeGen/AArch64/clmul-fixed.ll
index 4e5e4c07e281f7..be088994757d9a 100644
--- a/llvm/test/CodeGen/AArch64/clmul-fixed.ll
+++ b/llvm/test/CodeGen/AArch64/clmul-fixed.ll
@@ -1925,7 +1925,7 @@ define <4 x i16> @clmulr_v4i16_neon(<4 x i16> %a, <4 x i16> %b) nounwind {
 ; CHECK-AES-NEXT:    mov v2.d[1], v4.d[0]
 ; CHECK-AES-NEXT:    mov v0.d[1], v3.d[0]
 ; CHECK-AES-NEXT:    uzp1 v0.4s, v0.4s, v2.4s
-; CHECK-AES-NEXT:    shrn v0.4h, v0.4s, #15
+; CHECK-AES-NEXT:    addhn v0.4h, v0.4s, v0.4s
 ; CHECK-AES-NEXT:    ret
   %a.ext = zext <4 x i16> %a to <4 x i32>
   %b.ext = zext <4 x i16> %b to <4 x i32>
@@ -2148,7 +2148,7 @@ define <2 x i32> @clmulr_v2i32_neon(<2 x i32> %a, <2 x i32> %b) nounwind {
 ; CHECK-AES-NEXT:    pmull2 v2.1q, v0.2d, v1.2d
 ; CHECK-AES-NEXT:    pmull v0.1q, v0.1d, v1.1d
 ; CHECK-AES-NEXT:    mov v0.d[1], v2.d[0]
-; CHECK-AES-NEXT:    shrn v0.2s, v0.2d, #31
+; CHECK-AES-NEXT:    addhn v0.2s, v0.2d, v0.2d
 ; CHECK-AES-NEXT:    ret
   %a.ext = zext <2 x i32> %a to <2 x i64>
   %b.ext = zext <2 x i32> %b to <2 x i64>
diff --git a/llvm/test/CodeGen/AArch64/neon-rshrn.ll b/llvm/test/CodeGen/AArch64/neon-rshrn.ll
index 4cc8df59c95db7..5782fc38f245a8 100644
--- a/llvm/test/CodeGen/AArch64/neon-rshrn.ll
+++ b/llvm/test/CodeGen/AArch64/neon-rshrn.ll
@@ -899,8 +899,8 @@ entry:
 define <16 x i8> @or_rshrn_v16i16_7(<16 x i16> %a) {
 ; CHECK-LABEL: or_rshrn_v16i16_7:
 ; CHECK:       // %bb.0: // %entry
-; CHECK-NEXT:    shrn v0.8b, v0.8h, #7
-; CHECK-NEXT:    shrn2 v0.16b, v1.8h, #7
+; CHECK-NEXT:    addhn v0.8b, v0.8h, v0.8h
+; CHECK-NEXT:    addhn2 v0.16b, v1.8h, v1.8h
 ; CHECK-NEXT:    ret
 entry:
   %b = or disjoint <16 x i16> %a, <i16 64, i16 64, i16 64, i16 64, i16 64, i16 64, i16 64, i16 64, i16 64, i16 64, i16 64, i16 64, i16 64, i16 64, i16 64, i16 64>
@@ -924,8 +924,8 @@ entry:
 define <8 x i16> @or_rshrn_v8i32_15(<8 x i32> %a) {
 ; CHECK-LABEL: or_rshrn_v8i32_15:
 ; CHECK:       // %bb.0: // %entry
-; CHECK-NEXT:    shrn v0.4h, v0.4s, #15
-; CHECK-NEXT:    shrn2 v0.8h, v1.4s, #15
+; CHECK-NEXT:    addhn v0.4h, v0.4s, v0.4s
+; CHECK-NEXT:    addhn2 v0.8h, v1.4s, v1.4s
 ; CHECK-NEXT:    ret
 entry:
   %b = or disjoint <8 x i32> %a, <i32 16384, i32 16384, i32 16384, i32 16384, i32 16384, i32 16384, i32 16384, i32 16384>
@@ -949,8 +949,8 @@ entry:
 define <4 x i32> @or_rshrn_v4i64_31(<4 x i64> %a) {
 ; CHECK-LABEL: or_rshrn_v4i64_31:
 ; CHECK:       // %bb.0: // %entry
-; CHECK-NEXT:    shrn v0.2s, v0.2d, #31
-; CHECK-NEXT:    shrn2 v0.4s, v1.2d, #31
+; CHECK-NEXT:    addhn v0.2s, v0.2d, v0.2d
+; CHECK-NEXT:    addhn2 v0.4s, v1.2d, v1.2d
 ; CHECK-NEXT:    ret
 entry:
   %b = or disjoint <4 x i64> %a, <i64 1073741824, i64 1073741824, i64 1073741824, i64 1073741824>



More information about the llvm-commits mailing list