[llvm] 778dd2c - [AArch64][NeoverseV1] Load/Store Register Half with scaling (#222282)

via llvm-commits llvm-commits at lists.llvm.org
Thu Oct 1 01:41:08 PDT 2026


Author: Julien Villette
Date: 2026-10-01T08:40:55Z
New Revision: 778dd2c1b8050f69bee779893c27e7647c07eff3

URL: https://github.com/llvm/llvm-project/commit/778dd2c1b8050f69bee779893c27e7647c07eff3
DIFF: https://github.com/llvm/llvm-project/commit/778dd2c1b8050f69bee779893c27e7647c07eff3.diff

LOG: [AArch64][NeoverseV1] Load/Store Register Half with scaling (#222282)

On Neoverse V1, LDRH/STRH with scaling has a specific execution compared
to other LDR/STR.

Opcode            | Latency   | Pipelines
-------------------------------------------
Common LDR        | 4         | L
LDRH with scaling | 5         | I, L
Common STR        | 1         | L01, D
STRH with scaling | 2         | I, L01, D

Verified on Neoverse V1 with micro benchmarks and compared all
addressing mode variantes. Neoverse V2 is unaffected (confirmed by SOG
documentation and microbenchmarks).

Added: 
    

Modified: 
    llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
    llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td
    llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td b/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
index 05a48a23be875..9ad17b9c91e0a 100644
--- a/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
+++ b/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
@@ -253,6 +253,8 @@ let Latency = 8, NumMicroOps = 3 in
 def V1Write_8c_1L_2V        : SchedWriteRes<[V1UnitL, V1UnitV, V1UnitV]>;
 let Latency = 6, NumMicroOps = 3 in
 def V1Write_6c_3L           : SchedWriteRes<[V1UnitL, V1UnitL, V1UnitL]>;
+let Latency = 1, NumMicroOps = 3 in
+def V1Write_1c_1I_1L01_1D: SchedWriteRes<[V1UnitI, V1UnitL01, V1UnitD]>;
 let Latency = 2, NumMicroOps = 3 in
 def V1Write_2c_1L01_1S_1V   : SchedWriteRes<[V1UnitL01, V1UnitS, V1UnitV]>;
 let Latency = 4, NumMicroOps = 3 in
@@ -741,6 +743,13 @@ def : SchedAlias<WriteLD, V1Write_4c_1L>;
 def : SchedAlias<WriteLDIdx, V1Write_4c_1L>;
 def : SchedAlias<WriteAdr,   V1Write_1c_1I>;
 
+// Load register, register offset, extend, scale by 2
+// Load register, register offset, extend
+def V1WriteLDRH : SchedWriteVariant<[
+  SchedVar<NeoverseScaledIdxPred, [V1Write_5c_1I_1L]>,
+  SchedVar<NoSchedPred,	  [V1Write_4c_1L]>]>;
+def : InstRW<[V1WriteLDRH, ReadAdrBase], (instregex "^LDRS?H[HWX]ro[WX]$")>;
+
 // Load pair, immed offset
 def : SchedAlias<WriteLDHi, V1Write_4c_1L>;
 def : InstRW<[V1Write_4c_1L, V1Write_0c_0Z], (instrs LDPWi, LDNPWi)>;
@@ -764,6 +773,13 @@ def : SchedAlias<WriteST, V1Write_1c_1L01_1D>;
 // Store register, immed offset, index
 def : SchedAlias<WriteSTIdx, V1Write_1c_1L01_1D>;
 
+// Store register, register offset, scaled by 2
+// Store register, register offset, extend, scale by 1
+def V1WriteSTRH : SchedWriteVariant<[
+  SchedVar<NeoverseScaledIdxPred, [V1Write_1c_1I_1L01_1D]>,
+  SchedVar<NoSchedPred,	  [V1Write_1c_1L01_1D]>]>;
+def : InstRW<[V1WriteSTRH], (instregex "^STRS?H[HWX]ro[WX]$")>;
+
 // Store pair, immed offset
 def : SchedAlias<WriteSTP, V1Write_1c_1L01_1D>;
 

diff  --git a/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td b/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td
index 5b9c3e0b460f6..2594864a5a311 100644
--- a/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td
+++ b/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td
@@ -120,3 +120,13 @@ def NeoverseAllActivePredicate : MCSchedPredicate<
                                    ]>>;
 
 
+// Identify a load or store using the register offset addressing mode
+// with a scaled register.
+def NeoverseScaledIdxFn       : TIIPredicate<"isNeoverseScaledAddr",
+                                     MCOpcodeSwitchStatement<
+                                       [MCOpcodeSwitchCase<
+                                          IsLoadStoreRegOffsetOp.ValidOpcodes,
+                                          MCReturnStatement<
+                                            CheckAny<[CheckMemScaled]>>>],
+                                       MCReturnStatement<FalsePred>>>;
+def NeoverseScaledIdxPred     : MCSchedPredicate<NeoverseScaledIdxFn>;

diff  --git a/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s b/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s
index 16e9de608f46d..b1eefa477180a 100644
--- a/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s
+++ b/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s
@@ -1046,16 +1046,16 @@
 # CHECK-NEXT:  1      4     0.33    *                   ldrsb	x18, [x22, w10, sxtw]
 # CHECK-NEXT:  1      4     0.33    *                   ldrsh	w3, [sp, x5]
 # CHECK-NEXT:  1      4     0.33    *                   ldrsh	w9, [x27, x6]
-# CHECK-NEXT:  1      4     0.33    *                   ldrh	w10, [x30, x7, lsl #1]
+# CHECK-NEXT:  2      5     0.33    *                   ldrh	w10, [x30, x7, lsl #1]
 # CHECK-NEXT:  2      1     0.50           *            strh	w11, [x29, x3, sxtx]
 # CHECK-NEXT:  1      4     0.33    *                   ldrh	w12, [x28, xzr, sxtx]
-# CHECK-NEXT:  1      4     0.33    *                   ldrsh	x13, [x27, x5, sxtx #1]
+# CHECK-NEXT:  2      5     0.33    *                   ldrsh	x13, [x27, x5, sxtx #1]
 # CHECK-NEXT:  1      4     0.33    *                   ldrh	w14, [x26, w6, uxtw]
 # CHECK-NEXT:  1      4     0.33    *                   ldrh	w15, [x25, w7, uxtw]
-# CHECK-NEXT:  1      4     0.33    *                   ldrsh	w16, [x24, w8, uxtw #1]
+# CHECK-NEXT:  2      5     0.33    *                   ldrsh	w16, [x24, w8, uxtw #1]
 # CHECK-NEXT:  1      4     0.33    *                   ldrh	w17, [x23, w9, sxtw]
 # CHECK-NEXT:  1      4     0.33    *                   ldrh	w18, [x22, w10, sxtw]
-# CHECK-NEXT:  2      1     0.50           *            strh	w19, [x21, wzr, sxtw #1]
+# CHECK-NEXT:  3      1     0.50           *            strh	w19, [x21, wzr, sxtw #1]
 # CHECK-NEXT:  1      6     0.33    *                   ldr	b25, [x21, w8, uxtw]
 # CHECK-NEXT:  1      6     0.33    *                   ldr	b8, [x30, x10]
 # CHECK-NEXT:  2      2     0.50           *            str	b14, [x13, x25]
@@ -1286,7 +1286,7 @@
 
 # CHECK:      Resource pressure per iteration:
 # CHECK-NEXT: [0.0]  [0.1]  [1.0]  [1.1]  [2.0]  [2.1]  [2.2]  [3]    [4.0]  [4.1]  [5]    [6]    [7.0]  [7.1]  [8]    [9]    [10]   [11]
-# CHECK-NEXT: 13.00  13.00  34.00  34.00  51.00  51.00  51.00  98.67  171.67 171.67 321.00 208.00 152.00 152.00 208.00 56.50  47.50  13.00
+# CHECK-NEXT: 13.00  13.00  34.00  34.00  51.00  51.00  51.00  98.67  171.67 171.67 322.00 209.00 153.00 153.00 208.00 56.50  47.50  13.00
 
 # CHECK:      Resource pressure by instruction:
 # CHECK-NEXT: [0.0]  [0.1]  [1.0]  [1.1]  [2.0]  [2.1]  [2.2]  [3]    [4.0]  [4.1]  [5]    [6]    [7.0]  [7.1]  [8]    [9]    [10]   [11]   Instructions:
@@ -2326,16 +2326,16 @@
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrsb	x18, [x22, w10, sxtw]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrsh	w3, [sp, x5]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrsh	w9, [x27, x6]
-# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w10, [x30, x7, lsl #1]
+# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33   0.25   0.25   0.25   0.25    -      -      -      -     ldrh	w10, [x30, x7, lsl #1]
 # CHECK-NEXT:  -      -     0.50   0.50    -      -      -      -     0.50   0.50    -      -      -      -      -      -      -      -     strh	w11, [x29, x3, sxtx]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w12, [x28, xzr, sxtx]
-# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrsh	x13, [x27, x5, sxtx #1]
+# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33   0.25   0.25   0.25   0.25    -      -      -      -     ldrsh	x13, [x27, x5, sxtx #1]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w14, [x26, w6, uxtw]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w15, [x25, w7, uxtw]
-# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrsh	w16, [x24, w8, uxtw #1]
+# CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33   0.25   0.25   0.25   0.25    -      -      -      -     ldrsh	w16, [x24, w8, uxtw #1]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w17, [x23, w9, sxtw]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldrh	w18, [x22, w10, sxtw]
-# CHECK-NEXT:  -      -     0.50   0.50    -      -      -      -     0.50   0.50    -      -      -      -      -      -      -      -     strh	w19, [x21, wzr, sxtw #1]
+# CHECK-NEXT:  -      -     0.50   0.50    -      -      -      -     0.50   0.50   0.25   0.25   0.25   0.25    -      -      -      -     strh	w19, [x21, wzr, sxtw #1]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldr	b25, [x21, w8, uxtw]
 # CHECK-NEXT:  -      -      -      -      -      -      -     0.33   0.33   0.33    -      -      -      -      -      -      -      -     ldr	b8, [x30, x10]
 # CHECK-NEXT:  -      -      -      -      -      -      -      -     0.50   0.50    -      -      -      -     0.50   0.50    -      -     str	b14, [x13, x25]


        


More information about the llvm-commits mailing list