[llvm] [AArch64] Fold the local-exec TLS low relocation into loads and stores (PR #215531)

Jiang Ning via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 11 05:01:11 PDT 2026


https://github.com/jiang-ning-ultra created https://github.com/llvm/llvm-project/pull/215531

Emit the low part of the 24-bit local-exec sequence as an ADDlow so that
it folds into the addressing mode of the access, saving one instruction.

This re-lands the local-exec part of r327316 (7bc64bd889ad), whose ELF
changes r327503 (bde677289acc) reverted because neither LLD nor GNU
binutils implemented R_AARCH64_TLSLE_LDST*_TPREL_LO12_NC at the time.
Both handle the 8 to 64-bit variants now; bfd still has no LDST128, so
128-bit accesses stay unfolded.

>From 0006ff265848107461d27e9d20d0126236393499 Mon Sep 17 00:00:00 2001
From: JiangNing <jiangninghx at foxmail.com>
Date: Tue, 11 Aug 2026 17:40:12 +0800
Subject: [PATCH] [AArch64] Fold the local-exec TLS low relocation into loads
 and stores

Emit the low part of the 24-bit local-exec sequence as an ADDlow so that
it folds into the addressing mode of the access, saving one instruction.

This re-lands the local-exec part of r327316, whose ELF changes r327503
reverted because neither LLD nor GNU binutils implemented
R_AARCH64_TLSLE_LDST*_TPREL_LO12_NC at the time. Both handle the 8 to
64-bit variants now; bfd still has no LDST128, so 128-bit accesses stay
unfolded.
---
 .../Target/AArch64/AArch64ISelDAGToDAG.cpp    | 17 +++++++++++++-
 .../Target/AArch64/AArch64ISelLowering.cpp    |  8 +++----
 .../CodeGen/AArch64/arm64-tls-local-exec.ll   | 22 ++++++++++++++++---
 3 files changed, 39 insertions(+), 8 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelDAGToDAG.cpp b/llvm/lib/Target/AArch64/AArch64ISelDAGToDAG.cpp
index f22806678211f..84df7517ab811 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelDAGToDAG.cpp
@@ -1254,6 +1254,17 @@ static bool isWorthFoldingADDlow(SDValue N) {
   return true;
 }
 
+/// Check whether \p GAN is the low part of a TLS address computation, i.e. the
+/// second operand of an ADDlow. The target flags on their own do not tell the
+/// ELF local-exec (:tprel_lo12_nc:) case apart from the ELF local-dynamic
+/// (:dtprel_lo12_nc:) or the COFF (:secrel_lo12:) one, so callers that depend
+/// on local-exec semantics have to check the object format as well. Local
+/// dynamic never gets here because it does not build an ADDlow.
+static bool isTLSLo12(const GlobalAddressSDNode *GAN) {
+  return GAN->getTargetFlags() ==
+         (AArch64II::MO_TLS | AArch64II::MO_PAGEOFF | AArch64II::MO_NC);
+}
+
 /// Check if the immediate offset is valid as a scaled immediate.
 static bool isValidAsScaledImmediate(int64_t Offset, unsigned Range,
                                      unsigned Size) {
@@ -1349,8 +1360,12 @@ bool AArch64DAGToDAGISel::SelectAddrModeIndexed(SDValue N, unsigned Size,
     if (!GAN)
       return true;
 
+    // Folding the low part of an ELF local-exec TLS address into a 128-bit
+    // access needs R_AARCH64_TLSLE_LDST128_TPREL_LO12_NC, which the GNU bfd
+    // linker does not support, so keep materialising the address with an add.
     if (GAN->getOffset() % Size == 0 &&
-        GAN->getGlobal()->getPointerAlignment(DL) >= Size)
+        GAN->getGlobal()->getPointerAlignment(DL) >= Size &&
+        !(Size > 8 && Subtarget->isTargetELF() && isTLSLo12(GAN)))
       return true;
   }
 
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 6003db1c72449..fbb7947de8674 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -11483,10 +11483,10 @@ SDValue AArch64TargetLowering::LowerELFTLSLocalExec(const GlobalValue *GV,
                                       HiVar,
                                       DAG.getTargetConstant(0, DL, MVT::i32)),
                    0);
-    return SDValue(DAG.getMachineNode(AArch64::ADDXri, DL, PtrVT, Addr,
-                                      LoVar,
-                                      DAG.getTargetConstant(0, DL, MVT::i32)),
-                   0);
+    // Emit the low part as an ADDlow so that it can be folded into the
+    // addressing mode of a following load or store, turning the add into a
+    // :tprel_lo12_nc: relocation on the memory access itself.
+    return DAG.getNode(AArch64ISD::ADDlow, DL, PtrVT, Addr, LoVar);
   }
 
   case 32: {
diff --git a/llvm/test/CodeGen/AArch64/arm64-tls-local-exec.ll b/llvm/test/CodeGen/AArch64/arm64-tls-local-exec.ll
index 59d5500ce534e..a2cdb5dda0e19 100644
--- a/llvm/test/CodeGen/AArch64/arm64-tls-local-exec.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-tls-local-exec.ll
@@ -27,6 +27,7 @@
 ; RUN: llc -mtriple=arm64-none-linux-gnu -filetype=obj < %s -code-model=large | llvm-objdump -r - | FileCheck --check-prefix=CHECK-24-RELOC %s
 
 @local_exec_var = thread_local(localexec) global i32 0
+ at vec_local_exec_var = thread_local(localexec) global <2 x i64> zeroinitializer, align 16
 
 define i32 @test_local_exec() {
 ; CHECK-LABEL: test_local_exec:
@@ -40,11 +41,10 @@ define i32 @test_local_exec() {
 
 ; CHECK-24: mrs x[[R1:[0-9]+]], TPIDR_EL0
 ; CHECK-24: add x[[R2:[0-9]+]], x[[R1]], :tprel_hi12:local_exec_var
-; CHECK-24: add x[[R3:[0-9]+]], x[[R2]], :tprel_lo12_nc:local_exec_var
-; CHECK-24: ldr w0, [x[[R3]]]
+; CHECK-24: ldr w0, [x[[R2]], :tprel_lo12_nc:local_exec_var]
 
 ; CHECK-24-RELOC: R_AARCH64_TLSLE_ADD_TPREL_HI12
-; CHECK-24-RELOC: R_AARCH64_TLSLE_ADD_TPREL_LO12_NC
+; CHECK-24-RELOC: R_AARCH64_TLSLE_LDST32_TPREL_LO12_NC
 
 ; CHECK-32: movz x[[R2:[0-9]+]], #:tprel_g1:local_exec_var
 ; CHECK-32: mrs x[[R1:[0-9]+]], TPIDR_EL0
@@ -104,3 +104,19 @@ define ptr @test_local_exec_addr() {
 ; CHECK-48-RELOC: R_AARCH64_TLSLE_MOVW_TPREL_G1_NC
 ; CHECK-48-RELOC: R_AARCH64_TLSLE_MOVW_TPREL_G0_NC
 }
+
+; A 128-bit access would need R_AARCH64_TLSLE_LDST128_TPREL_LO12_NC, which not
+; every linker implements, so the low part stays in a separate add.
+define <2 x i64> @test_local_exec_128bit() {
+; CHECK-LABEL: test_local_exec_128bit:
+  %val = load <2 x i64>, ptr @vec_local_exec_var
+
+; CHECK-24: mrs x[[R1:[0-9]+]], TPIDR_EL0
+; CHECK-24: add x[[R2:[0-9]+]], x[[R1]], :tprel_hi12:vec_local_exec_var
+; CHECK-24: add x[[R3:[0-9]+]], x[[R2]], :tprel_lo12_nc:vec_local_exec_var
+; CHECK-24: ldr q0, [x[[R3]]]
+
+; CHECK-24-RELOC: R_AARCH64_TLSLE_ADD_TPREL_HI12 vec_local_exec_var
+; CHECK-24-RELOC-NEXT: R_AARCH64_TLSLE_ADD_TPREL_LO12_NC vec_local_exec_var
+  ret <2 x i64> %val
+}



More information about the llvm-commits mailing list