[llvm] [AArch64][NeoverseV1] Load/Store Register Half with scaling (PR #222282)

Julien Villette via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 28 06:26:17 PDT 2026


https://github.com/jvillette38 updated https://github.com/llvm/llvm-project/pull/222282

>From 2c1e6f5c54a2f2c97c75f3d2e15a5c22c68e1b60 Mon Sep 17 00:00:00 2001
From: Julien Villette <julien.villette at sipearl.com>
Date: Tue, 8 Sep 2026 17:23:00 +0200
Subject: [PATCH 1/2] [AArch64][NeoverseV1] Load/Store Register Half with
 scaling

On Neoverse V1, LDRH/STRH with scaling has a specific execution compared to other LDR/STR.

Opcode            | Latency   | Pipelines
-------------------------------------------
Common LDR        | 4         | L
LDRH with scaling | 5         | I, L
Common STR        | 1         | L01, D
STRH with scaling | 2         | I, L01, D

Verified on Neoverse V1 with micro benchmarks and compared all addressing mode variantes.
Neoverse V2 is unaffected (confirmed by SOG documentation and
microbenchmarks).
---
 .../Target/AArch64/AArch64SchedNeoverseV1.td   | 16 ++++++++++++++++
 .../Target/AArch64/AArch64SchedPredNeoverse.td | 10 ++++++++++
 .../AArch64/Neoverse/V1-basic-instructions.s   | 18 +++++++++---------
 3 files changed, 35 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td b/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
index b8bab7ece3a2b7..650da4c1fd8d32 100644
--- a/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
+++ b/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
@@ -252,6 +252,8 @@ let Latency = 8, NumMicroOps = 3 in
 def V1Write_8c_1L_2V        : SchedWriteRes<[V1UnitL, V1UnitV, V1UnitV]>;
 let Latency = 6, NumMicroOps = 3 in
 def V1Write_6c_3L           : SchedWriteRes<[V1UnitL, V1UnitL, V1UnitL]>;
+let Latency = 1, NumMicroOps = 3 in
+def V1Write_1c_1I_1L01_1D: SchedWriteRes<[V1UnitI, V1UnitL01, V1UnitD]>;
 let Latency = 2, NumMicroOps = 3 in
 def V1Write_2c_1L01_1S_1V   : SchedWriteRes<[V1UnitL01, V1UnitS, V1UnitV]>;
 let Latency = 4, NumMicroOps = 3 in
@@ -722,6 +724,13 @@ def : SchedAlias<WriteLD, V1Write_4c_1L>;
 def : SchedAlias<WriteLDIdx, V1Write_4c_1L>;
 def : SchedAlias<WriteAdr,   V1Write_1c_1I>;
 
+// Load register, register offset, extend, scale by 2
+// Load register, register offset, extend
+def V1WriteLDRH : SchedWriteVariant<[
+  SchedVar<NeoverseScaledIdxPred, [V1Write_5c_1I_1L]>,
+  SchedVar<NoSchedPred,	  [V1Write_4c_1L]>]>;
+def : InstRW<[V1WriteLDRH, ReadAdrBase], (instregex "^LDRS?H[HWX]ro[WX]$")>;
+
 // Load pair, immed offset
 def : SchedAlias<WriteLDHi, V1Write_4c_1L>;
 def : InstRW<[V1Write_4c_1L, V1Write_0c_0Z], (instrs LDPWi, LDNPWi)>;
@@ -745,6 +754,13 @@ def : SchedAlias<WriteST, V1Write_1c_1L01_1D>;
 // Store register, immed offset, index
 def : SchedAlias<WriteSTIdx, V1Write_1c_1L01_1D>;
 
+// Store register, register offset, extend, scale by 2
+// Store register, register offset, extend
+def V1WriteSTRH : SchedWriteVariant<[
+  SchedVar<NeoverseScaledIdxPred, [V1Write_1c_1I_1L01_1D]>,
+  SchedVar<NoSchedPred,	  [V1Write_1c_1L01_1D]>]>;
+def : InstRW<[V1WriteSTRH], (instregex "^STRS?H[HWX]ro[WX]$")>;
+
 // Store pair, immed offset
 def : SchedAlias<WriteSTP, V1Write_1c_1L01_1D>;
 
diff --git a/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td b/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td
index 737d7fe9ecc8c6..461e84473961e9 100644
--- a/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td
+++ b/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td
@@ -106,3 +106,13 @@ def NeoverseAllActivePredicate : MCSchedPredicate<
                                    ]>>;
 
 
+// Identify a load or store using the register offset addressing mode
+// with a scaled register.
+def NeoverseScaledIdxFn       : TIIPredicate<"isNeoverseScaledAddr",
+                                     MCOpcodeSwitchStatement<
+                                       [MCOpcodeSwitchCase<
+                                          IsLoadStoreRegOffsetOp.ValidOpcodes,
+                                          MCReturnStatement<
+                                            CheckAny<[CheckMemScaled]>>>],
+                                       MCReturnStatement<FalsePred>>>;
+def NeoverseScaledIdxPred     : MCSchedPredicate<NeoverseScaledIdxFn>;
diff --git a/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s b/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s
index da6a36c654ee80..e283a021f2c704 100644
--- a/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s
+++ b/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s
@@ -1036,16 +1036,16 @@
 # CHECK-NEXT:  1      4     0.33    *                   ldrsb	x18, [x22, w10, sxtw]
 # CHECK-NEXT:  1      4     0.33    *                   ldrsh	w3, [sp, x5]
 # CHECK-NEXT:  1      4     0.33    *                   ldrsh	w9, [x27, x6]
-# CHECK-NEXT:  1      4     0.33    *                   ldrh	w10, [x30, x7, lsl #1]
+# CHECK-NEXT:  2      5     0.33    *                   ldrh	w10, [x30, x7, lsl #1]
 # CHECK-NEXT:  2      1     0.50           *            strh	w11, [x29, x3, sxtx]
 # CHECK-NEXT:  1      4     0.33    *                   ldrh	w12, [x28, xzr, sxtx]
-# CHECK-NEXT:  1      4     0.33    *                   ldrsh	x13, [x27, x5, sxtx #1]
+# CHECK-NEXT:  2      5     0.33    *                   ldrsh	x13, [x27, x5, sxtx #1]
 # CHECK-NEXT:  1      4     0.33    *                   ldrh	w14, [x26, w6, uxtw]
 # CHECK-NEXT:  1      4     0.33    *                   ldrh	w15, [x25, w7, uxtw]
-# CHECK-NEXT:  1      4     0.33    *                   ldrsh	w16, [x24, w8, uxtw #1]
+# CHECK-NEXT:  2      5     0.33    *                   ldrsh	w16, [x24, w8, uxtw #1]
 # CHECK-NEXT:  1      4     0.33    *                   ldrh	w17, [x23, w9, sxtw]
 # CHECK-NEXT:  1      4     0.33    *                   ldrh	w18, [x22, w10, sxtw]
-# CHECK-NEXT:  2      1     0.50           *            strh	w19, [x21, wzr, sxtw #1]
+# CHECK-NEXT:  3      1     0.50           *            strh	w19, [x21, wzr, sxtw #1]
 # CHECK-NEXT:  1      6     0.33    *                   ldr	b25, [x21, w8, uxtw]
 # CHECK-NEXT:  1      6     0.33    *                   ldr	b8, [x30, x10]
 # CHECK-NEXT:  2      2     0.50           *            str	b14, [x13, x25]
@@ -1273,7 +1273,7 @@
 
 # CHECK:      Resource pressure per iteration:
 # CHECK-NEXT: [0.0]  [0.1]  [1.0]  [1.1]  [2.0]  [2.1]  [2.2]  [3]    [4.0]  [4.1]  [5]    [6]    [7.0]  [7.1]  [8]    [9]    [10]   [11]
-# CHECK-NEXT: 13.00  13.00  34.00  34.00  48.00  48.00  48.00  98.67  171.67 171.67 323.50 210.50 143.00 143.00 190.00 56.50  65.50  13.00
+# CHECK-NEXT: 13.00  13.00  34.00  34.00  48.00  48.00  48.00  98.67  171.67 171.67 324.50 211.50 144.00 144.00 190.00 56.50  65.50  13.00
 
 # CHECK:      Resource pressure by instruction:
 # CHECK-NEXT: [0.0]  [0.1]  [1.0]  [1.1]  [2.0]  [2.1]  [2.2]  [3]    [4.0]  [4.1]  [5]    [6]    [7.0]  [7.1]  [8]    [9]    [10]   [11]   Instructions:
@@ -2303,16 +2303,16 @@
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrsb	x18, [x22, w10, sxtw]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrsh	w3, [sp, x5]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrsh	w9, [x27, x6]
-# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w10, [x30, x7, lsl #1]
+# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33   0.25   0.25   0.25   0.25    -      -      -      -     ldrh	w10, [x30, x7, lsl #1]
 # CHECK-NEXT:  -      -     0.50   0.50    -      -      -      -     0.50   0.50    -      -      -      -      -      -      -      -     strh	w11, [x29, x3, sxtx]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w12, [x28, xzr, sxtx]
-# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrsh	x13, [x27, x5, sxtx #1]
+# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33   0.25   0.25   0.25   0.25    -      -      -      -     ldrsh	x13, [x27, x5, sxtx #1]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w14, [x26, w6, uxtw]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w15, [x25, w7, uxtw]
-# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrsh	w16, [x24, w8, uxtw #1]
+# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33   0.25   0.25   0.25   0.25    -      -      -      -     ldrsh	w16, [x24, w8, uxtw #1]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w17, [x23, w9, sxtw]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w18, [x22, w10, sxtw]
-# CHECK-NEXT:  -      -     0.50   0.50    -      -      -      -     0.50   0.50    -      -      -      -      -      -      -      -     strh	w19, [x21, wzr, sxtw #1]
+# CHECK-NEXT:  -      -     0.50   0.50    -      -      -      -     0.50   0.50   0.25   0.25   0.25   0.25    -      -      -      -     strh	w19, [x21, wzr, sxtw #1]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldr	b25, [x21, w8, uxtw]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldr	b8, [x30, x10]
 # CHECK-NEXT:  -      -      -      -      -      -      -      -     0.50   0.50    -      -      -      -     0.50   0.50    -      -     str	b14, [x13, x25]

>From 160af50a873cd4a04d996cd1fd4f38f9b431c3d7 Mon Sep 17 00:00:00 2001
From: Julien Villette <julien.villette at sipearl.com>
Date: Mon, 14 Sep 2026 08:36:44 +0200
Subject: [PATCH 2/2] [AArch64][NeoverseV1] Load/Store Register Half with
 scaling.

Fix comment on Stores.
---
 llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td b/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
index 650da4c1fd8d32..c0d32b88d5a23b 100644
--- a/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
+++ b/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
@@ -754,8 +754,8 @@ def : SchedAlias<WriteST, V1Write_1c_1L01_1D>;
 // Store register, immed offset, index
 def : SchedAlias<WriteSTIdx, V1Write_1c_1L01_1D>;
 
-// Store register, register offset, extend, scale by 2
-// Store register, register offset, extend
+// Store register, register offset, scaled by 2
+// Store register, register offset, extend, scale by 1
 def V1WriteSTRH : SchedWriteVariant<[
   SchedVar<NeoverseScaledIdxPred, [V1Write_1c_1I_1L01_1D]>,
   SchedVar<NoSchedPred,	  [V1Write_1c_1L01_1D]>]>;



More information about the llvm-commits mailing list