[lld] [lld][LoongArch] Prevent relaxation oscillation for PCHi20 and CALL (PR #226713)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 28 01:18:22 PDT 2026


https://github.com/heiher updated https://github.com/llvm/llvm-project/pull/226713

>From a2dd33abf127ce2693b2eb22d72af836e64bc007 Mon Sep 17 00:00:00 2001
From: WANG Rui <wangrui at loongson.cn>
Date: Sat, 26 Sep 2026 23:30:31 +0800
Subject: [PATCH] [lld][LoongArch] Prevent relaxation oscillation for LA.PCRel
 and CALL

Relaxation of pcalau12i+addi (relaxPCHi20Lo12, isInt<22>) and
call36/call30 (relaxMediumCall, isInt<28>) can oscillate: shrinking
one section moves a symbol, which flips isInt<N> for other sites and
changes bytesDropped again.

Follow the same approach as RISCV::relaxCall: after a few passes, do
not allow remove to increase beyond the previous pass's value
(cur - delta).  Pass that cap as prevRemove into the two helpers;
range checks may still clear remove (0) when the target goes out of
range.
---
 lld/ELF/Arch/LoongArch.cpp                 | 29 +++++++++----
 lld/test/ELF/loongarch-relax-call-stress.s | 47 ++++++++++++++++++++++
 2 files changed, 69 insertions(+), 7 deletions(-)
 create mode 100644 lld/test/ELF/loongarch-relax-call-stress.s

diff --git a/lld/ELF/Arch/LoongArch.cpp b/lld/ELF/Arch/LoongArch.cpp
index 224c31d5a2f1d..6c65b74b716f2 100644
--- a/lld/ELF/Arch/LoongArch.cpp
+++ b/lld/ELF/Arch/LoongArch.cpp
@@ -1113,7 +1113,7 @@ static bool isPairRelaxable(ArrayRef<Relocation> relocs, size_t i) {
 //   pcaddi $a0, %got_pc_hi20(sym_got)
 static void relaxPCHi20Lo12(Ctx &ctx, const InputSection &sec, size_t i,
                             uint64_t loc, Relocation &rHi20, Relocation &rLo12,
-                            uint32_t &remove) {
+                            uint32_t &remove, uint32_t prevRemove) {
   // check if the relocations are relaxable sequences.
   if (!((rHi20.type == R_LARCH_PCALA_HI20 &&
          rLo12.type == R_LARCH_PCALA_LO12) ||
@@ -1179,6 +1179,11 @@ static void relaxPCHi20Lo12(Ctx &ctx, const InputSection &sec, size_t i,
   if (getD5(currInsn) != getJ5(nextInsn) || getJ5(nextInsn) != getD5(nextInsn))
     return;
 
+  // When the caller specifies the old value of `remove`, disallow its
+  // increment.
+  if (prevRemove < 4)
+    return;
+
   sec.relaxAux->relocTypes[i] = R_LARCH_RELAX;
   if (rHi20.type == R_LARCH_TLS_GD_PC_HI20)
     sec.relaxAux->relocTypes[i + 2] = R_LARCH_TLS_GD_PCREL20_S2;
@@ -1203,7 +1208,8 @@ static void relaxPCHi20Lo12(Ctx &ctx, const InputSection &sec, size_t i,
 // To:
 //   b/bl foo
 static void relaxMediumCall(Ctx &ctx, const InputSection &sec, size_t i,
-                            uint64_t loc, Relocation &r, uint32_t &remove) {
+                            uint64_t loc, Relocation &r, uint32_t &remove,
+                            uint32_t prevRemove) {
   const uint64_t dest =
       (r.expr == R_PLT_PC ? r.sym->getPltVA(ctx) : r.sym->getVA(ctx)) +
       r.addend;
@@ -1213,6 +1219,11 @@ static void relaxMediumCall(Ctx &ctx, const InputSection &sec, size_t i,
   if ((displace & 0x3) != 0 || !isInt<28>(displace))
     return;
 
+  // When the caller specifies the old value of `remove`, disallow its
+  // increment.
+  if (prevRemove < 4)
+    return;
+
   const uint32_t nextInsn = read32le(sec.content().data() + r.offset + 4);
   if (getD5(nextInsn) == R_RA) {
     // convert jirl to bl
@@ -1254,7 +1265,7 @@ static void relaxTlsLe(Ctx &ctx, const InputSection &sec, size_t i,
   }
 }
 
-static bool relax(Ctx &ctx, InputSection &sec) {
+static bool relax(Ctx &ctx, int pass, InputSection &sec) {
   const uint64_t secAddr = sec.getVA();
   const MutableArrayRef<Relocation> relocs = sec.relocs();
   auto &aux = *sec.relaxAux;
@@ -1267,6 +1278,10 @@ static bool relax(Ctx &ctx, InputSection &sec) {
   for (auto [i, r] : llvm::enumerate(relocs)) {
     const uint64_t loc = secAddr + r.offset - delta;
     uint32_t &cur = aux.relocDeltas[i], remove = 0;
+    // Prevent oscillation between states by disallowing the increment of
+    // `remove` after a few passes. The previous `remove` value is
+    // `cur-delta`.
+    uint32_t prevRemove = pass < 4 ? UINT32_MAX : cur - delta;
     switch (r.type) {
     case R_LARCH_ALIGN: {
       const uint64_t addend =
@@ -1298,19 +1313,19 @@ static bool relax(Ctx &ctx, InputSection &sec) {
     case R_LARCH_TLS_LD_PC_HI20:
       // The overflow check for i+2 will be carried out in isPairRelaxable.
       if (isPairRelaxable(relocs, i))
-        relaxPCHi20Lo12(ctx, sec, i, loc, r, relocs[i + 2], remove);
+        relaxPCHi20Lo12(ctx, sec, i, loc, r, relocs[i + 2], remove, prevRemove);
       break;
     case R_LARCH_TLS_DESC_PC_HI20:
       if (r.expr == RE_LOONGARCH_GOT_PAGE_PC || r.expr == R_TPREL) {
         if (relaxable(relocs, i))
           remove = 4;
       } else if (isPairRelaxable(relocs, i))
-        relaxPCHi20Lo12(ctx, sec, i, loc, r, relocs[i + 2], remove);
+        relaxPCHi20Lo12(ctx, sec, i, loc, r, relocs[i + 2], remove, prevRemove);
       break;
     case R_LARCH_CALL30:
     case R_LARCH_CALL36:
       if (relaxable(relocs, i))
-        relaxMediumCall(ctx, sec, i, loc, r, remove);
+        relaxMediumCall(ctx, sec, i, loc, r, remove, prevRemove);
       break;
     case R_LARCH_TLS_LE_HI20_R:
     case R_LARCH_TLS_LE_ADD_R:
@@ -1698,7 +1713,7 @@ bool LoongArch::relaxOnce(int pass) const {
       continue;
     for (InputSection *sec : getInputSections(*osec, storage))
       if (sec->relaxAux)
-        changed |= relax(ctx, *sec);
+        changed |= relax(ctx, pass, *sec);
   }
   return changed;
 }
diff --git a/lld/test/ELF/loongarch-relax-call-stress.s b/lld/test/ELF/loongarch-relax-call-stress.s
new file mode 100644
index 0000000000000..fe4ec567ccbc2
--- /dev/null
+++ b/lld/test/ELF/loongarch-relax-call-stress.s
@@ -0,0 +1,47 @@
+# REQUIRES: loongarch
+##
+## pcalau12i+addi.d -> pcaddi relaxation (relaxPCHi20Lo12, isInt<22>) must not
+## oscillate between remove=0 and remove=4. Without the fix, ld.lld reports
+## "address assignment did not converge".
+##
+## Unlike RISC-V calls (8 -> 4 -> 2 bytes), every LoongArch pair/call36 site has
+## only two states (8 or 4 bytes), so oscillation is always a 0 <-> 4 flip.
+## The flip needs a site whose distance sits exactly at the +-2MiB limit while
+## a ".p2align 4" (R_LARCH_ALIGN) absorbs some shrinks but not others, so the
+## target address jitters between passes.
+##
+## The layout is deliberately fragile: the .space size below was found by
+## searching a model of lld's relaxation loop. Do not "round" it.
+
+# RUN: llvm-mc -filetype=obj -triple=loongarch64 -mattr=+relax %s -o %t.o
+# RUN: ld.lld -e _start %t.o -o %t
+# RUN: llvm-objdump -d --no-show-raw-insn %t | FileCheck %s
+
+## The three short-range sites in .text.a are always relaxed.
+# CHECK-LABEL: <_start>:
+# CHECK-NEXT:    pcaddi $t0,
+# CHECK-NEXT:    pcaddi $t1,
+
+.section .text.a,"ax"
+.globl _start
+_start:
+  la.pcrel $t0, t_d
+  la.pcrel $t1, t_e
+  la.pcrel $t2, t_c            # forward, crosses the 2MiB filler
+.globl t_d
+t_d:
+.globl t_e
+t_e:
+
+.section .text.b,"ax"
+  .space 8
+  la.pcrel $t3, t_a
+  la.pcrel $t4, t_d            # backward, crosses the 2MiB filler
+  .space 2097120    # 2MiB - 32: puts the sites right at the isInt<22> limit
+  la.pcrel $t5, _start         # backward, near the limit
+.globl t_c
+t_c:
+  .p2align 4        # R_LARCH_ALIGN; also makes .text.b 16-byte aligned
+  la.pcrel $t6, t_d
+.globl t_a
+t_a:



More information about the llvm-commits mailing list