[llvm] 778dd2c - [AArch64][NeoverseV1] Load/Store Register Half with scaling (#222282)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Oct 1 01:41:08 PDT 2026
Author: Julien Villette
Date: 2026-10-01T08:40:55Z
New Revision: 778dd2c1b8050f69bee779893c27e7647c07eff3
URL: https://github.com/llvm/llvm-project/commit/778dd2c1b8050f69bee779893c27e7647c07eff3
DIFF: https://github.com/llvm/llvm-project/commit/778dd2c1b8050f69bee779893c27e7647c07eff3.diff
LOG: [AArch64][NeoverseV1] Load/Store Register Half with scaling (#222282)
On Neoverse V1, LDRH/STRH with scaling has a specific execution compared
to other LDR/STR.
Opcode | Latency | Pipelines
-------------------------------------------
Common LDR | 4 | L
LDRH with scaling | 5 | I, L
Common STR | 1 | L01, D
STRH with scaling | 2 | I, L01, D
Verified on Neoverse V1 with micro benchmarks and compared all
addressing mode variantes. Neoverse V2 is unaffected (confirmed by SOG
documentation and microbenchmarks).
Added:
Modified:
llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td
llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s
Removed:
################################################################################
diff --git a/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td b/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
index 05a48a23be875..9ad17b9c91e0a 100644
--- a/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
+++ b/llvm/lib/Target/AArch64/AArch64SchedNeoverseV1.td
@@ -253,6 +253,8 @@ let Latency = 8, NumMicroOps = 3 in
def V1Write_8c_1L_2V : SchedWriteRes<[V1UnitL, V1UnitV, V1UnitV]>;
let Latency = 6, NumMicroOps = 3 in
def V1Write_6c_3L : SchedWriteRes<[V1UnitL, V1UnitL, V1UnitL]>;
+let Latency = 1, NumMicroOps = 3 in
+def V1Write_1c_1I_1L01_1D: SchedWriteRes<[V1UnitI, V1UnitL01, V1UnitD]>;
let Latency = 2, NumMicroOps = 3 in
def V1Write_2c_1L01_1S_1V : SchedWriteRes<[V1UnitL01, V1UnitS, V1UnitV]>;
let Latency = 4, NumMicroOps = 3 in
@@ -741,6 +743,13 @@ def : SchedAlias<WriteLD, V1Write_4c_1L>;
def : SchedAlias<WriteLDIdx, V1Write_4c_1L>;
def : SchedAlias<WriteAdr, V1Write_1c_1I>;
+// Load register, register offset, extend, scale by 2
+// Load register, register offset, extend
+def V1WriteLDRH : SchedWriteVariant<[
+ SchedVar<NeoverseScaledIdxPred, [V1Write_5c_1I_1L]>,
+ SchedVar<NoSchedPred, [V1Write_4c_1L]>]>;
+def : InstRW<[V1WriteLDRH, ReadAdrBase], (instregex "^LDRS?H[HWX]ro[WX]$")>;
+
// Load pair, immed offset
def : SchedAlias<WriteLDHi, V1Write_4c_1L>;
def : InstRW<[V1Write_4c_1L, V1Write_0c_0Z], (instrs LDPWi, LDNPWi)>;
@@ -764,6 +773,13 @@ def : SchedAlias<WriteST, V1Write_1c_1L01_1D>;
// Store register, immed offset, index
def : SchedAlias<WriteSTIdx, V1Write_1c_1L01_1D>;
+// Store register, register offset, scaled by 2
+// Store register, register offset, extend, scale by 1
+def V1WriteSTRH : SchedWriteVariant<[
+ SchedVar<NeoverseScaledIdxPred, [V1Write_1c_1I_1L01_1D]>,
+ SchedVar<NoSchedPred, [V1Write_1c_1L01_1D]>]>;
+def : InstRW<[V1WriteSTRH], (instregex "^STRS?H[HWX]ro[WX]$")>;
+
// Store pair, immed offset
def : SchedAlias<WriteSTP, V1Write_1c_1L01_1D>;
diff --git a/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td b/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td
index 5b9c3e0b460f6..2594864a5a311 100644
--- a/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td
+++ b/llvm/lib/Target/AArch64/AArch64SchedPredNeoverse.td
@@ -120,3 +120,13 @@ def NeoverseAllActivePredicate : MCSchedPredicate<
]>>;
+// Identify a load or store using the register offset addressing mode
+// with a scaled register.
+def NeoverseScaledIdxFn : TIIPredicate<"isNeoverseScaledAddr",
+ MCOpcodeSwitchStatement<
+ [MCOpcodeSwitchCase<
+ IsLoadStoreRegOffsetOp.ValidOpcodes,
+ MCReturnStatement<
+ CheckAny<[CheckMemScaled]>>>],
+ MCReturnStatement<FalsePred>>>;
+def NeoverseScaledIdxPred : MCSchedPredicate<NeoverseScaledIdxFn>;
diff --git a/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s b/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s
index 16e9de608f46d..b1eefa477180a 100644
--- a/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s
+++ b/llvm/test/tools/llvm-mca/AArch64/Neoverse/V1-basic-instructions.s
@@ -1046,16 +1046,16 @@
# CHECK-NEXT: 1 4 0.33 * ldrsb x18, [x22, w10, sxtw]
# CHECK-NEXT: 1 4 0.33 * ldrsh w3, [sp, x5]
# CHECK-NEXT: 1 4 0.33 * ldrsh w9, [x27, x6]
-# CHECK-NEXT: 1 4 0.33 * ldrh w10, [x30, x7, lsl #1]
+# CHECK-NEXT: 2 5 0.33 * ldrh w10, [x30, x7, lsl #1]
# CHECK-NEXT: 2 1 0.50 * strh w11, [x29, x3, sxtx]
# CHECK-NEXT: 1 4 0.33 * ldrh w12, [x28, xzr, sxtx]
-# CHECK-NEXT: 1 4 0.33 * ldrsh x13, [x27, x5, sxtx #1]
+# CHECK-NEXT: 2 5 0.33 * ldrsh x13, [x27, x5, sxtx #1]
# CHECK-NEXT: 1 4 0.33 * ldrh w14, [x26, w6, uxtw]
# CHECK-NEXT: 1 4 0.33 * ldrh w15, [x25, w7, uxtw]
-# CHECK-NEXT: 1 4 0.33 * ldrsh w16, [x24, w8, uxtw #1]
+# CHECK-NEXT: 2 5 0.33 * ldrsh w16, [x24, w8, uxtw #1]
# CHECK-NEXT: 1 4 0.33 * ldrh w17, [x23, w9, sxtw]
# CHECK-NEXT: 1 4 0.33 * ldrh w18, [x22, w10, sxtw]
-# CHECK-NEXT: 2 1 0.50 * strh w19, [x21, wzr, sxtw #1]
+# CHECK-NEXT: 3 1 0.50 * strh w19, [x21, wzr, sxtw #1]
# CHECK-NEXT: 1 6 0.33 * ldr b25, [x21, w8, uxtw]
# CHECK-NEXT: 1 6 0.33 * ldr b8, [x30, x10]
# CHECK-NEXT: 2 2 0.50 * str b14, [x13, x25]
@@ -1286,7 +1286,7 @@
# CHECK: Resource pressure per iteration:
# CHECK-NEXT: [0.0] [0.1] [1.0] [1.1] [2.0] [2.1] [2.2] [3] [4.0] [4.1] [5] [6] [7.0] [7.1] [8] [9] [10] [11]
-# CHECK-NEXT: 13.00 13.00 34.00 34.00 51.00 51.00 51.00 98.67 171.67 171.67 321.00 208.00 152.00 152.00 208.00 56.50 47.50 13.00
+# CHECK-NEXT: 13.00 13.00 34.00 34.00 51.00 51.00 51.00 98.67 171.67 171.67 322.00 209.00 153.00 153.00 208.00 56.50 47.50 13.00
# CHECK: Resource pressure by instruction:
# CHECK-NEXT: [0.0] [0.1] [1.0] [1.1] [2.0] [2.1] [2.2] [3] [4.0] [4.1] [5] [6] [7.0] [7.1] [8] [9] [10] [11] Instructions:
@@ -2326,16 +2326,16 @@
# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldrsb x18, [x22, w10, sxtw]
# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldrsh w3, [sp, x5]
# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldrsh w9, [x27, x6]
-# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldrh w10, [x30, x7, lsl #1]
+# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 0.25 0.25 0.25 0.25 - - - - ldrh w10, [x30, x7, lsl #1]
# CHECK-NEXT: - - 0.50 0.50 - - - - 0.50 0.50 - - - - - - - - strh w11, [x29, x3, sxtx]
# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldrh w12, [x28, xzr, sxtx]
-# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldrsh x13, [x27, x5, sxtx #1]
+# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 0.25 0.25 0.25 0.25 - - - - ldrsh x13, [x27, x5, sxtx #1]
# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldrh w14, [x26, w6, uxtw]
# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldrh w15, [x25, w7, uxtw]
-# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldrsh w16, [x24, w8, uxtw #1]
+# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 0.25 0.25 0.25 0.25 - - - - ldrsh w16, [x24, w8, uxtw #1]
# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldrh w17, [x23, w9, sxtw]
# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldrh w18, [x22, w10, sxtw]
-# CHECK-NEXT: - - 0.50 0.50 - - - - 0.50 0.50 - - - - - - - - strh w19, [x21, wzr, sxtw #1]
+# CHECK-NEXT: - - 0.50 0.50 - - - - 0.50 0.50 0.25 0.25 0.25 0.25 - - - - strh w19, [x21, wzr, sxtw #1]
# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldr b25, [x21, w8, uxtw]
# CHECK-NEXT: - - - - - - - 0.33 0.33 0.33 - - - - - - - - ldr b8, [x30, x10]
# CHECK-NEXT: - - - - - - - - 0.50 0.50 - - - - 0.50 0.50 - - str b14, [x13, x25]
More information about the llvm-commits
mailing list