[llvm] [BOLT][AArch64] Add support for conditional tail calls in cold code (PR #227693)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 30 06:00:07 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-bolt

Author: AJMullen (andreasmullen)

<details>
<summary>Changes</summary>

**Before:** BOLT handles AArch64 conditional tail calls in cold code using the workaround implemented in \#<!-- -->140669. This leaves the conditional branches unchanged and patches the target function’s original entry to redirect execution to the relocated function.

**After:** This patch adds support for `R_AARCH64_CONDBR19` and `R_AARCH64_TSTBR14` relocations defined in the [ABI](https://github.com/ARM-software/abi-aa/blob/main/aaelf64/aaelf64.rst#<!-- -->576static-aarch64-relocations). BOLT can now update `B.cond`, `CBZ/CBNZ`, and `TBZ/TBNZ` to directly reach relocated functions when they are within range. Branches out-of-range are left unchanged and target the original patched function. 

Assisted-by: Codex

---
Full diff: https://github.com/llvm/llvm-project/pull/227693.diff


10 Files Affected:

- (modified) bolt/include/bolt/Core/Relocation.h (+3-1) 
- (modified) bolt/lib/Core/BinaryFunction.cpp (+10-18) 
- (modified) bolt/lib/Core/BinarySection.cpp (+14-2) 
- (modified) bolt/lib/Core/Relocation.cpp (+26-3) 
- (modified) bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp (+18-7) 
- (added) bolt/test/AArch64/conditional-tailcall-bcond.s (+38) 
- (added) bolt/test/AArch64/conditional-tailcall-cbz.s (+46) 
- (added) bolt/test/AArch64/conditional-tailcall-condbr-range.s (+61) 
- (added) bolt/test/AArch64/conditional-tailcall-tbz.s (+46) 
- (added) bolt/test/AArch64/conditional-tailcall-tstbr-range.s (+61) 


``````````diff
diff --git a/bolt/include/bolt/Core/Relocation.h b/bolt/include/bolt/Core/Relocation.h
index 02d18ff97cf98..f5d9aca3794af 100644
--- a/bolt/include/bolt/Core/Relocation.h
+++ b/bolt/include/bolt/Core/Relocation.h
@@ -112,7 +112,9 @@ class Relocation {
   static bool skipRelocationType(uint32_t Type);
 
   /// Adjust value depending on relocation type (make it PC relative or not).
-  static uint64_t encodeValue(uint32_t Type, uint64_t Value, uint64_t PC);
+  /// OriginalInst is to be able to encode instructions for AArch64.
+  static uint64_t encodeValue(uint32_t Type, uint64_t Value, uint64_t PC,
+                              uint32_t OriginalInst);
 
   /// Return true if there are enough bits to encode the relocation value.
   static bool canEncodeValue(uint32_t Type, uint64_t Value, uint64_t PC);
diff --git a/bolt/lib/Core/BinaryFunction.cpp b/bolt/lib/Core/BinaryFunction.cpp
index 39c4dd5595e3c..721ee02c19b86 100644
--- a/bolt/lib/Core/BinaryFunction.cpp
+++ b/bolt/lib/Core/BinaryFunction.cpp
@@ -1793,22 +1793,10 @@ bool BinaryFunction::scanExternalRefs() {
     // On AArch64, we use instruction patches for fixing references. We make an
     // exception for branch instructions since they require optional
     // relocations.
-    if (BC.isAArch64()) {
-      if (!BranchTargetSymbol) {
-        LLVM_DEBUG(BC.printInstruction(dbgs(), Instruction, AbsoluteInstrAddr));
-        InstructionPatches.push_back({AbsoluteInstrAddr, Instruction});
-        continue;
-      }
-
-      // Conditional tail calls require new relocation types that are currently
-      // not supported. https://github.com/llvm/llvm-project/issues/138264
-      if (BC.MIB->isConditionalBranch(Instruction)) {
-        if (BinaryFunction *TargetBF =
-                BC.getFunctionForSymbol(BranchTargetSymbol)) {
-          TargetBF->setNeedsPatch(true);
-          continue;
-        }
-      }
+    if (BC.isAArch64() && !BranchTargetSymbol) {
+      LLVM_DEBUG(BC.printInstruction(dbgs(), Instruction, AbsoluteInstrAddr));
+      InstructionPatches.push_back({AbsoluteInstrAddr, Instruction});
+      continue;
     }
 
     // Emit the instruction using temp emitter and generate relocations.
@@ -1840,7 +1828,12 @@ bool BinaryFunction::scanExternalRefs() {
         // relocation value encoding.
         Rel->setOptional();
 
-        if (!opts::CompactCodeModel)
+        // The compact code model may allow conditional branches target
+        // addresses to be out of range, therefore be conservative and patch the
+        // target function.
+        const bool IsConditionalBranch = Rel->Type == ELF::R_AARCH64_CONDBR19 ||
+                                         Rel->Type == ELF::R_AARCH64_TSTBR14;
+        if (!opts::CompactCodeModel || IsConditionalBranch)
           if (BinaryFunction *TargetBF = BC.getFunctionForSymbol(Rel->Symbol))
             TargetBF->setNeedsPatch(true);
       }
@@ -1848,7 +1841,6 @@ bool BinaryFunction::scanExternalRefs() {
       Rel->Offset += getAddress() - getOriginSection()->getAddress() + Offset;
       FunctionRelocations.push_back(*Rel);
     }
-
     if (!Success)
       break;
   }
diff --git a/bolt/lib/Core/BinarySection.cpp b/bolt/lib/Core/BinarySection.cpp
index a8620ba83ebfb..17dbd98dc64cc 100644
--- a/bolt/lib/Core/BinarySection.cpp
+++ b/bolt/lib/Core/BinarySection.cpp
@@ -16,6 +16,7 @@
 #include "bolt/Utils/Utils.h"
 #include "llvm/MC/MCStreamer.h"
 #include "llvm/Support/CommandLine.h"
+#include "llvm/Support/Endian.h"
 
 #define DEBUG_TYPE "bolt"
 
@@ -191,8 +192,19 @@ void BinarySection::flushPendingRelocations(raw_fd_ostream &OS,
       ++SkippedPendingRelocations;
       continue;
     }
-    Value = Relocation::encodeValue(Reloc.Type, Value,
-                                    SectionAddress + Reloc.Offset);
+
+    uint32_t OriginalInst = 0;
+    // Are we dealing with B.cond, TBZ/TBNZ, CBZ/CBNZ and need extra info
+    // to be able to encode the instruction.
+    if (BC.isAArch64() && (Reloc.Type == ELF::R_AARCH64_CONDBR19 ||
+                           Reloc.Type == ELF::R_AARCH64_TSTBR14)) {
+      StringRef Contents = getContents();
+      assert(Contents.size() >= Reloc.Offset + 4 &&
+             "Complete instruction must lie in contents.");
+      OriginalInst = support::endian::read32le(Contents.data() + Reloc.Offset);
+    }
+    Value = Relocation::encodeValue(
+        Reloc.Type, Value, SectionAddress + Reloc.Offset, OriginalInst);
 
     safePWrite(OS, reinterpret_cast<const char *>(&Value),
                Relocation::getSizeForType(Reloc.Type),
diff --git a/bolt/lib/Core/Relocation.cpp b/bolt/lib/Core/Relocation.cpp
index b36c2cd8a4c82..5f9e9c9b9aba3 100644
--- a/bolt/lib/Core/Relocation.cpp
+++ b/bolt/lib/Core/Relocation.cpp
@@ -293,10 +293,16 @@ static bool canEncodeValueAArch64(uint32_t Type, uint64_t Value, uint64_t PC) {
   case ELF::R_AARCH64_CALL26:
   case ELF::R_AARCH64_JUMP26:
     return isInt<28>(Value - PC);
+  case ELF::R_AARCH64_CONDBR19:
+    return isInt<21>(Value - PC);
+  case ELF::R_AARCH64_TSTBR14:
+    return isInt<16>(Value - PC);
   }
 }
 
-static uint64_t encodeValueAArch64(uint32_t Type, uint64_t Value, uint64_t PC) {
+static uint64_t encodeValueAArch64(uint32_t Type, uint64_t Value, uint64_t PC,
+                                   uint32_t OriginalInst) {
+  // Assume Value and PC are 4-byte aligned to ensure valid bit manipulation.
   switch (Type) {
   default:
     llvm_unreachable("unsupported relocation");
@@ -324,6 +330,22 @@ static uint64_t encodeValueAArch64(uint32_t Type, uint64_t Value, uint64_t PC) {
     // OP 0001_01 goes in bits 31:26 of B.
     Value = ((Value >> 2) & 0x3ffffff) | 0x14000000ULL;
     break;
+  case ELF::R_AARCH64_CONDBR19:
+    Value -= PC;
+    assert(isInt<21>(Value) &&
+           "only PC +/- 1MB is allowed for conditional branch");
+    // Immediate goes in bits 23:5, which is taken through masking.
+    // Preserve all other bits from the original instruction.
+    Value =
+        (OriginalInst & ~0x00FFFFE0ULL) | (((Value >> 2) & 0x7FFFFULL) << 5);
+    break;
+  case ELF::R_AARCH64_TSTBR14:
+    Value -= PC;
+    assert(isInt<16>(Value) && "only PC +/- 32KB is allowed for test branch");
+    // Immediate goes in bits 18:5, which is taken through masking.
+    // Preserve all other bits from the original instruction.
+    Value = (OriginalInst & ~0x0007FFE0ULL) | (((Value >> 2) & 0x3FFFULL) << 5);
+    break;
   }
   return Value;
 }
@@ -782,12 +804,13 @@ bool Relocation::skipRelocationType(uint32_t Type) {
   }
 }
 
-uint64_t Relocation::encodeValue(uint32_t Type, uint64_t Value, uint64_t PC) {
+uint64_t Relocation::encodeValue(uint32_t Type, uint64_t Value, uint64_t PC,
+                                 uint32_t OriginalInst) {
   switch (Arch) {
   default:
     llvm_unreachable("Unsupported architecture");
   case Triple::aarch64:
-    return encodeValueAArch64(Type, Value, PC);
+    return encodeValueAArch64(Type, Value, PC, OriginalInst);
   case Triple::riscv64:
   case Triple::riscv32:
     return encodeValueRISCV(Type, Value, PC);
diff --git a/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp b/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
index 2a39aa63554d9..8c514748afe06 100644
--- a/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
+++ b/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
@@ -3695,17 +3695,28 @@ class AArch64MCPlusBuilder : public MCPlusBuilder {
   std::optional<Relocation>
   createRelocation(const MCFixup &Fixup,
                    const MCAsmBackend &MAB) const override {
-    MCFixupKindInfo FKI = MAB.getFixupKindInfo(Fixup.getKind());
+    MCFixupKind FKind = Fixup.getKind();
+    MCFixupKindInfo FKI = MAB.getFixupKindInfo(FKind);
 
-    assert(FKI.TargetOffset == 0 && "0-bit relocation offset expected");
-    const uint64_t RelOffset = Fixup.getOffset();
+    switch (FKind) {
+    case MCFixupKind(AArch64::fixup_aarch64_pcrel_branch19):
+    case MCFixupKind(AArch64::fixup_aarch64_pcrel_branch14):
+      assert(FKI.TargetOffset == 5 && "5-bit relocation offset expected");
+      break;
+    default:
+      assert(FKI.TargetOffset == 0 && "0-bit relocation offset expected");
+      break;
+    }
 
     uint32_t RelType;
-    if (Fixup.getKind() == MCFixupKind(AArch64::fixup_aarch64_pcrel_call26))
+    if (FKind == MCFixupKind(AArch64::fixup_aarch64_pcrel_call26))
       RelType = ELF::R_AARCH64_CALL26;
-    else if (Fixup.getKind() ==
-             MCFixupKind(AArch64::fixup_aarch64_pcrel_branch26))
+    else if (FKind == MCFixupKind(AArch64::fixup_aarch64_pcrel_branch26))
       RelType = ELF::R_AARCH64_JUMP26;
+    else if (FKind == MCFixupKind(AArch64::fixup_aarch64_pcrel_branch19))
+      RelType = ELF::R_AARCH64_CONDBR19;
+    else if (FKind == MCFixupKind(AArch64::fixup_aarch64_pcrel_branch14))
+      RelType = ELF::R_AARCH64_TSTBR14;
     else if (Fixup.isPCRel()) {
       switch (FKI.TargetSize) {
       default:
@@ -3735,7 +3746,7 @@ class AArch64MCPlusBuilder : public MCPlusBuilder {
         break;
       }
     }
-
+    const uint64_t RelOffset = Fixup.getOffset();
     auto [RelSymbol, RelAddend] = extractFixupExpr(Fixup);
 
     return Relocation({RelOffset, RelSymbol, RelType, RelAddend, 0});
diff --git a/bolt/test/AArch64/conditional-tailcall-bcond.s b/bolt/test/AArch64/conditional-tailcall-bcond.s
new file mode 100644
index 0000000000000..7fd323fff939e
--- /dev/null
+++ b/bolt/test/AArch64/conditional-tailcall-bcond.s
@@ -0,0 +1,38 @@
+## Check support for conditional tail calls, ensure that conditional branches with
+## B.cond are correctly relocated. Assume that everything is in range with 
+## --no-huge-pages. 
+
+# RUN: llvm-mc -filetype=obj -triple=aarch64-unknown %s -o %t.o
+# RUN: ld.lld --emit-relocs %t.o -o %t.exe
+# RUN: llvm-bolt %t.exe -o %t.bolt --skip-funcs=_start --no-huge-pages 
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED
+
+# NONBOLTED: [[#%x,FOONOBOLT:]] <foo>: 
+# NONBOLTED: {{.*}} <_start>:
+# NONBOLTED: {{.*}} b.ge 0x[[#FOONOBOLT]] <foo> 
+# NONBOLTED: {{.*}} b.pl 0x[[#FOONOBOLT]] <foo>
+
+# BOLTED: {{.*}} <foo.org.0>: 
+# BOLTED: {{.*}} adrp x16, 0x[[#%x,FOO:]] <foo> 
+# BOLTED: {{.*}} <_start>: 
+# BOLTED: {{.*}} b.ge 0x[[#FOO]] <foo> 
+# BOLTED: {{.*}} b.pl 0x[[#FOO]] <foo> 
+# BOLTED: [[#FOO]] <foo>:  
+
+    .type foo, at function
+    .globl foo
+foo: 
+  .rept 3
+    nop
+  .endr  
+  ret
+  .size foo, .-foo
+
+    .type _start, at function
+    .globl _start
+_start: 
+  b.ge foo
+  b.pl foo
+  ret
+  .size _start, .-_start
diff --git a/bolt/test/AArch64/conditional-tailcall-cbz.s b/bolt/test/AArch64/conditional-tailcall-cbz.s
new file mode 100644
index 0000000000000..9176742562b7a
--- /dev/null
+++ b/bolt/test/AArch64/conditional-tailcall-cbz.s
@@ -0,0 +1,46 @@
+## Check support for conditional tail calls, ensure that conditional branches with
+## CBZ/CBNZ are correctly relocated. Assume that everything is in range with 
+## --no-huge-pages. 
+
+# RUN: llvm-mc -filetype=obj -triple=aarch64-unknown-unknown %s -o %t.o
+# RUN: ld.lld --emit-relocs %t.o -o %t.exe
+# RUN: llvm-bolt --no-huge-pages --skip-funcs=_start %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED
+
+# NONBOLTED: [[#%x,BAR:]] <bar>:
+# NONBOLTED: {{.*}} <_start>: 
+# NONBOLTED: {{.*}} cbz w0, 0x[[#BAR]] <bar> 
+# NONBOLTED: {{.*}} cbz x21, 0x[[#BAR]] <bar> 
+# NONBOLTED: {{.*}} cbnz x10, 0x[[#BAR]] <bar> 
+# NONBOLTED: {{.*}} cbnz wzr, 0x[[#BAR]] <bar> 
+
+# BOLTED: [[#%x,BARORG:]] <bar.org.0>: 
+# BOLTED: [[#BARORG]]: {{.*}} adrp x16, 0x[[#%x,BARNEW:]] <bar> 
+# BOLTED: {{.*}} <_start>: 
+# BOLTED: {{.*}} cbz w0, 0x[[#BARNEW]] <bar>
+# BOLTED: {{.*}} cbz x21, 0x[[#BARNEW]] <bar> 
+# BOLTED: {{.*}} cbnz x10, 0x[[#BARNEW]] <bar> 
+# BOLTED: {{.*}} cbnz wzr, 0x[[#BARNEW]] <bar> 
+# BOLTED: [[#BARNEW]] <bar>: 
+
+    .type bar, at function
+    .globl bar
+bar: 
+  .rept 3
+    nop
+  .endr  
+  ret
+  .size bar, .-bar
+
+    .type _start, at function
+    .globl _start
+_start: 
+  cbz w0, bar 
+  cbz x21, bar
+
+  cbnz x10, bar
+  cbnz wzr, bar 
+  ret
+  .size _start, .-_start
+  
\ No newline at end of file
diff --git a/bolt/test/AArch64/conditional-tailcall-condbr-range.s b/bolt/test/AArch64/conditional-tailcall-condbr-range.s
new file mode 100644
index 0000000000000..34077aa80cc52
--- /dev/null
+++ b/bolt/test/AArch64/conditional-tailcall-condbr-range.s
@@ -0,0 +1,61 @@
+## Ensure that conditional calls using the CONDBR19 relocation type are only relocated 
+## if the target is range. Do not relocate the instruction if the target is out of range. 
+## Ensure that this applies in both directions (+/- displacements). 
+
+# RUN: llvm-mc -filetype=obj -triple=aarch64-unknown-unknown %s -o %t.o
+# RUN: ld.lld --emit-relocs --section-start=.text=0x300ff0 %t.o -o %t.exe
+# RUN: llvm-bolt --skip-funcs=_start --align-text=0x1000 %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED_FORWARD
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED_FORWARD
+
+# NONBOLTED_FORWARD: [[#%x,BAR:]] <bar>: 
+# NONBOLTED_FORWARD: {{.*}} <_start>: 
+# NONBOLTED_FORWARD: {{.*}} b.ge 0x[[#BAR]] <bar> 
+# NONBOLTED_FORWARD: {{.*}} b.ge 0x[[#BAR]] <bar> 
+
+# BOLTED_FORWARD: [[#%x,BAROLD:]] <bar.org.0>: 
+# BOLTED_FORWARD: {{.*}} adrp x16, 0x[[#%x,RELOC:]] <bar>
+# BOLTED_FORWARD: {{.*}} <_start>: 
+## The first branch is out of range of the relocated function and hence points to the 
+## original function, whereas the second is within range of the relocated function 
+## and hence points to the newly relocated function. 
+# BOLTED_FORWARD: {{.*}} b.ge 0x[[#BAROLD]] <bar.org.0>
+# BOLTED_FORWARD: {{.*}} b.ge 0x[[#RELOC]] <bar> 
+# BOLTED_FORWARD: [[#RELOC]] <bar>: 
+
+# RUN: ld.lld --emit-relocs --section-start=.text=500FF0 %t.o -o %t.exe
+# RUN: llvm-bolt --skip-funcs=_start --align-text=0x1000 \
+# RUN:   --custom-allocation-vma=0x400000 %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED_BACKWARD
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED_BACKWARD
+
+# NONBOLTED_BACKWARD: [[#%x,BAR:]] <bar>: 
+# NONBOLTED_BACKWARD: {{.*}} <_start>: 
+# NONBOLTED_BACKWARD: {{.*}} b.ge 0x[[#BAR]] <bar> 
+# NONBOLTED_BACKWARD: {{.*}} b.ge 0x[[#BAR]] <bar> 
+
+# BOLTED_BACKWARD: [[#%x,BAROLD:]] <bar.org.0>: 
+# BOLTED_BACKWARD: {{.*}} adrp x16, 0x[[#%x,RELOC:]] <bar>
+# BOLTED_BACKWARD: {{.*}} <_start>: 
+## The first branch is in range of the relocated function and hence is relocated to the 
+## relocated function, whereas the second is not and so points to the original function. 
+# BOLTED_BACKWARD: {{.*}} b.ge 0x[[#RELOC]] <bar>
+# BOLTED_BACKWARD: {{.*}} b.ge 0x[[#BAROLD]] <bar.org.0> 
+# BOLTED_BACKWARD: [[#RELOC]] <bar>: 
+
+    .type bar, at function
+    .globl bar
+bar: 
+  .rept 3
+    nop
+  .endr 
+  ret
+  .size bar, .-bar
+
+    .type _start, at function
+    .globl _start
+_start: 
+  b.ge bar
+  b.ge bar 
+  ret
+  .size _start, .-_start
diff --git a/bolt/test/AArch64/conditional-tailcall-tbz.s b/bolt/test/AArch64/conditional-tailcall-tbz.s
new file mode 100644
index 0000000000000..03a806bf0b365
--- /dev/null
+++ b/bolt/test/AArch64/conditional-tailcall-tbz.s
@@ -0,0 +1,46 @@
+## Check support for conditional tail calls, ensure that conditional branches with
+## TBZ/TBNZ are correctly relocated. Assume that everything is in range by 
+## aligning the text sections of the binaries. 
+
+# RUN: llvm-mc -filetype=obj -triple=aarch64-unknown-unknown %s -o %t.o
+# RUN: ld.lld --emit-relocs --section-start=.text=0x3ff000 %t.o -o %t.exe
+# RUN: llvm-bolt --skip-funcs=_start --align-text=0x1000 %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED
+
+# NONBOLTED: {{0*}}3ff000 <bar>: 
+# NONBOLTED: {{.*}} <_start>: 
+# NONBOLTED: {{.*}} tbz w0, #0x0, 0x3ff000 <bar> 
+# NONBOLTED: {{.*}} tbz x10, #0x20, 0x3ff000 <bar>
+# NONBOLTED: {{.*}} tbnz w21, #0x1f, 0x3ff000 <bar>
+# NONBOLTED: {{.*}} tbnz xzr, #0x3f, 0x3ff000 <bar>
+
+# BOLTED: {{0*}}3ff000 <bar.org.0>:
+# BOLTED: {{0*}}3ff000: {{.*}} adrp x16, 0x[[#%x,BAR:]] <bar>
+# BOLTED: {{.*}} <_start>:
+# BOLTED: {{.*}} tbz w0, #0x0, 0x[[#BAR]] <bar> 
+# BOLTED: {{.*}} tbz x10, #0x20, 0x[[#BAR]] <bar>
+# BOLTED: {{.*}} tbnz w21, #0x1f, 0x[[#BAR]] <bar>
+# BOLTED: {{.*}} tbnz xzr, #0x3f, 0x[[#BAR]] <bar>
+# BOLTED: [[#BAR]] <bar>:
+
+    .type bar, at function
+    .globl bar
+bar: 
+  .rept 3
+    nop
+  .endr  
+  ret
+  .size bar, .-bar
+
+    .type _start, at function
+    .globl _start
+_start: 
+  tbz w0, #0, bar
+  tbz x10, #32, bar
+
+  tbnz w21, #31, bar
+  tbnz xzr, #63, bar
+  ret
+  .size _start, .-_start
+  
\ No newline at end of file
diff --git a/bolt/test/AArch64/conditional-tailcall-tstbr-range.s b/bolt/test/AArch64/conditional-tailcall-tstbr-range.s
new file mode 100644
index 0000000000000..ed37d8201c2c3
--- /dev/null
+++ b/bolt/test/AArch64/conditional-tailcall-tstbr-range.s
@@ -0,0 +1,61 @@
+## Ensure that conditional calls using the TSTBR14 relocation type are only relocated 
+## if the target is range. Do not relocate the instruction if the target is out of range. 
+## Ensure that this applies in both directions (+/- displacements). 
+
+# RUN: llvm-mc -filetype=obj -triple=aarch64-unknown-unknown %s -o %t.o
+# RUN: ld.lld --emit-relocs --section-start=.text=0x3f8ff0 %t.o -o %t.exe
+# RUN: llvm-bolt --skip-funcs=_start --align-text=0x1000 %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED_FORWARD
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED_FORWARD
+
+# NONBOLTED_FORWARD: [[#%x,BAR:]] <bar>: 
+# NONBOLTED_FORWARD: {{.*}} <_start>: 
+# NONBOLTED_FORWARD: {{.*}} tbz x10, #0x20, 0x[[#BAR]] <bar> 
+# NONBOLTED_FORWARD: {{.*}} tbz x10, #0x20, 0x[[#BAR]] <bar> 
+
+# BOLTED_FORWARD: [[#%x,BAROLD:]] <bar.org.0>: 
+# BOLTED_FORWARD: {{.*}} adrp x16, 0x[[#%x,RELOC:]] <bar>
+# BOLTED_FORWARD: {{.*}} <_start>: 
+## The first branch is out of range of the relocated function and hence points to the 
+## original function, whereas the second is within range of the relocated function 
+## and hence points to the newly relocated function. 
+# BOLTED_FORWARD: {{.*}} tbz x10, #0x20, 0x[[#BAROLD]] <bar.org.0>
+# BOLTED_FORWARD: {{.*}} tbz x10, #0x20, 0x[[#RELOC]] <bar> 
+# BOLTED_FORWARD: [[#RELOC]] <bar>: 
+
+# RUN: ld.lld --emit-relocs --section-start=.text=0x408ff0 %t.o -o %t.exe
+# RUN: llvm-bolt --skip-funcs=_start --align-text=0x1000 \
+# RUN:   --custom-allocation-vma=0x400000 %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED_BACKWARD
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED_BACKWARD
+
+# NONBOLTED_BACKWARD: [[#%x,BAR:]] <bar>: 
+# NONBOLTED_BACKWARD: {{.*}} <_start>: 
+# NONBOLTED_BACKWARD: {{.*}} tbz x10, #0x20, 0x[[#BAR]] <bar> 
+# NONBOLTED_BACKWARD: {{.*}} tbz x10, #0x20, 0x[[#BAR]] <bar> 
+
+# BOLTED_BACKWARD: [[#%x,BAROLD:]] <bar.org.0>: 
+# BOLTED_BACKWARD: {{.*}} adrp x16, 0x[[#%x,RELOC:]] <bar>
+# BOLTED_BACKWARD: {{.*}} <_start>: 
+## The first branch is in range of the relocated function and hence is relocated to the 
+## relocated function, whereas the second is not and so points to the original function. 
+# BOLTED_BACKWARD: {{.*}} tbz x10, #0x20, 0x[[#RELOC]] <bar>
+# BOLTED_BACKWARD: {{.*}} tbz x10, #0x20, 0x[[#BAROLD]] <bar.org.0> 
+# BOLTED_BACKWARD: [[#RELOC]] <bar>: 
+
+    .type bar, at function
+    .globl bar
+bar: 
+  .rept 3
+    nop
+  .endr 
+  ret
+  .size bar, .-bar
+
+    .type _start, at function
+    .globl _start
+_start: 
+  tbz x10, #32, bar
+  tbz x10, #32, bar
+  ret
+  .size _start, .-_start

``````````

</details>


https://github.com/llvm/llvm-project/pull/227693


More information about the llvm-commits mailing list