[llvm] [BOLT][AArch64] Add support for conditional tail calls in cold code (PR #227693)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 06:00:07 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-bolt
Author: AJMullen (andreasmullen)
<details>
<summary>Changes</summary>
**Before:** BOLT handles AArch64 conditional tail calls in cold code using the workaround implemented in \#<!-- -->140669. This leaves the conditional branches unchanged and patches the target function’s original entry to redirect execution to the relocated function.
**After:** This patch adds support for `R_AARCH64_CONDBR19` and `R_AARCH64_TSTBR14` relocations defined in the [ABI](https://github.com/ARM-software/abi-aa/blob/main/aaelf64/aaelf64.rst#<!-- -->576static-aarch64-relocations). BOLT can now update `B.cond`, `CBZ/CBNZ`, and `TBZ/TBNZ` to directly reach relocated functions when they are within range. Branches out-of-range are left unchanged and target the original patched function.
Assisted-by: Codex
---
Full diff: https://github.com/llvm/llvm-project/pull/227693.diff
10 Files Affected:
- (modified) bolt/include/bolt/Core/Relocation.h (+3-1)
- (modified) bolt/lib/Core/BinaryFunction.cpp (+10-18)
- (modified) bolt/lib/Core/BinarySection.cpp (+14-2)
- (modified) bolt/lib/Core/Relocation.cpp (+26-3)
- (modified) bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp (+18-7)
- (added) bolt/test/AArch64/conditional-tailcall-bcond.s (+38)
- (added) bolt/test/AArch64/conditional-tailcall-cbz.s (+46)
- (added) bolt/test/AArch64/conditional-tailcall-condbr-range.s (+61)
- (added) bolt/test/AArch64/conditional-tailcall-tbz.s (+46)
- (added) bolt/test/AArch64/conditional-tailcall-tstbr-range.s (+61)
``````````diff
diff --git a/bolt/include/bolt/Core/Relocation.h b/bolt/include/bolt/Core/Relocation.h
index 02d18ff97cf98..f5d9aca3794af 100644
--- a/bolt/include/bolt/Core/Relocation.h
+++ b/bolt/include/bolt/Core/Relocation.h
@@ -112,7 +112,9 @@ class Relocation {
static bool skipRelocationType(uint32_t Type);
/// Adjust value depending on relocation type (make it PC relative or not).
- static uint64_t encodeValue(uint32_t Type, uint64_t Value, uint64_t PC);
+ /// OriginalInst is to be able to encode instructions for AArch64.
+ static uint64_t encodeValue(uint32_t Type, uint64_t Value, uint64_t PC,
+ uint32_t OriginalInst);
/// Return true if there are enough bits to encode the relocation value.
static bool canEncodeValue(uint32_t Type, uint64_t Value, uint64_t PC);
diff --git a/bolt/lib/Core/BinaryFunction.cpp b/bolt/lib/Core/BinaryFunction.cpp
index 39c4dd5595e3c..721ee02c19b86 100644
--- a/bolt/lib/Core/BinaryFunction.cpp
+++ b/bolt/lib/Core/BinaryFunction.cpp
@@ -1793,22 +1793,10 @@ bool BinaryFunction::scanExternalRefs() {
// On AArch64, we use instruction patches for fixing references. We make an
// exception for branch instructions since they require optional
// relocations.
- if (BC.isAArch64()) {
- if (!BranchTargetSymbol) {
- LLVM_DEBUG(BC.printInstruction(dbgs(), Instruction, AbsoluteInstrAddr));
- InstructionPatches.push_back({AbsoluteInstrAddr, Instruction});
- continue;
- }
-
- // Conditional tail calls require new relocation types that are currently
- // not supported. https://github.com/llvm/llvm-project/issues/138264
- if (BC.MIB->isConditionalBranch(Instruction)) {
- if (BinaryFunction *TargetBF =
- BC.getFunctionForSymbol(BranchTargetSymbol)) {
- TargetBF->setNeedsPatch(true);
- continue;
- }
- }
+ if (BC.isAArch64() && !BranchTargetSymbol) {
+ LLVM_DEBUG(BC.printInstruction(dbgs(), Instruction, AbsoluteInstrAddr));
+ InstructionPatches.push_back({AbsoluteInstrAddr, Instruction});
+ continue;
}
// Emit the instruction using temp emitter and generate relocations.
@@ -1840,7 +1828,12 @@ bool BinaryFunction::scanExternalRefs() {
// relocation value encoding.
Rel->setOptional();
- if (!opts::CompactCodeModel)
+ // The compact code model may allow conditional branches target
+ // addresses to be out of range, therefore be conservative and patch the
+ // target function.
+ const bool IsConditionalBranch = Rel->Type == ELF::R_AARCH64_CONDBR19 ||
+ Rel->Type == ELF::R_AARCH64_TSTBR14;
+ if (!opts::CompactCodeModel || IsConditionalBranch)
if (BinaryFunction *TargetBF = BC.getFunctionForSymbol(Rel->Symbol))
TargetBF->setNeedsPatch(true);
}
@@ -1848,7 +1841,6 @@ bool BinaryFunction::scanExternalRefs() {
Rel->Offset += getAddress() - getOriginSection()->getAddress() + Offset;
FunctionRelocations.push_back(*Rel);
}
-
if (!Success)
break;
}
diff --git a/bolt/lib/Core/BinarySection.cpp b/bolt/lib/Core/BinarySection.cpp
index a8620ba83ebfb..17dbd98dc64cc 100644
--- a/bolt/lib/Core/BinarySection.cpp
+++ b/bolt/lib/Core/BinarySection.cpp
@@ -16,6 +16,7 @@
#include "bolt/Utils/Utils.h"
#include "llvm/MC/MCStreamer.h"
#include "llvm/Support/CommandLine.h"
+#include "llvm/Support/Endian.h"
#define DEBUG_TYPE "bolt"
@@ -191,8 +192,19 @@ void BinarySection::flushPendingRelocations(raw_fd_ostream &OS,
++SkippedPendingRelocations;
continue;
}
- Value = Relocation::encodeValue(Reloc.Type, Value,
- SectionAddress + Reloc.Offset);
+
+ uint32_t OriginalInst = 0;
+ // Are we dealing with B.cond, TBZ/TBNZ, CBZ/CBNZ and need extra info
+ // to be able to encode the instruction.
+ if (BC.isAArch64() && (Reloc.Type == ELF::R_AARCH64_CONDBR19 ||
+ Reloc.Type == ELF::R_AARCH64_TSTBR14)) {
+ StringRef Contents = getContents();
+ assert(Contents.size() >= Reloc.Offset + 4 &&
+ "Complete instruction must lie in contents.");
+ OriginalInst = support::endian::read32le(Contents.data() + Reloc.Offset);
+ }
+ Value = Relocation::encodeValue(
+ Reloc.Type, Value, SectionAddress + Reloc.Offset, OriginalInst);
safePWrite(OS, reinterpret_cast<const char *>(&Value),
Relocation::getSizeForType(Reloc.Type),
diff --git a/bolt/lib/Core/Relocation.cpp b/bolt/lib/Core/Relocation.cpp
index b36c2cd8a4c82..5f9e9c9b9aba3 100644
--- a/bolt/lib/Core/Relocation.cpp
+++ b/bolt/lib/Core/Relocation.cpp
@@ -293,10 +293,16 @@ static bool canEncodeValueAArch64(uint32_t Type, uint64_t Value, uint64_t PC) {
case ELF::R_AARCH64_CALL26:
case ELF::R_AARCH64_JUMP26:
return isInt<28>(Value - PC);
+ case ELF::R_AARCH64_CONDBR19:
+ return isInt<21>(Value - PC);
+ case ELF::R_AARCH64_TSTBR14:
+ return isInt<16>(Value - PC);
}
}
-static uint64_t encodeValueAArch64(uint32_t Type, uint64_t Value, uint64_t PC) {
+static uint64_t encodeValueAArch64(uint32_t Type, uint64_t Value, uint64_t PC,
+ uint32_t OriginalInst) {
+ // Assume Value and PC are 4-byte aligned to ensure valid bit manipulation.
switch (Type) {
default:
llvm_unreachable("unsupported relocation");
@@ -324,6 +330,22 @@ static uint64_t encodeValueAArch64(uint32_t Type, uint64_t Value, uint64_t PC) {
// OP 0001_01 goes in bits 31:26 of B.
Value = ((Value >> 2) & 0x3ffffff) | 0x14000000ULL;
break;
+ case ELF::R_AARCH64_CONDBR19:
+ Value -= PC;
+ assert(isInt<21>(Value) &&
+ "only PC +/- 1MB is allowed for conditional branch");
+ // Immediate goes in bits 23:5, which is taken through masking.
+ // Preserve all other bits from the original instruction.
+ Value =
+ (OriginalInst & ~0x00FFFFE0ULL) | (((Value >> 2) & 0x7FFFFULL) << 5);
+ break;
+ case ELF::R_AARCH64_TSTBR14:
+ Value -= PC;
+ assert(isInt<16>(Value) && "only PC +/- 32KB is allowed for test branch");
+ // Immediate goes in bits 18:5, which is taken through masking.
+ // Preserve all other bits from the original instruction.
+ Value = (OriginalInst & ~0x0007FFE0ULL) | (((Value >> 2) & 0x3FFFULL) << 5);
+ break;
}
return Value;
}
@@ -782,12 +804,13 @@ bool Relocation::skipRelocationType(uint32_t Type) {
}
}
-uint64_t Relocation::encodeValue(uint32_t Type, uint64_t Value, uint64_t PC) {
+uint64_t Relocation::encodeValue(uint32_t Type, uint64_t Value, uint64_t PC,
+ uint32_t OriginalInst) {
switch (Arch) {
default:
llvm_unreachable("Unsupported architecture");
case Triple::aarch64:
- return encodeValueAArch64(Type, Value, PC);
+ return encodeValueAArch64(Type, Value, PC, OriginalInst);
case Triple::riscv64:
case Triple::riscv32:
return encodeValueRISCV(Type, Value, PC);
diff --git a/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp b/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
index 2a39aa63554d9..8c514748afe06 100644
--- a/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
+++ b/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
@@ -3695,17 +3695,28 @@ class AArch64MCPlusBuilder : public MCPlusBuilder {
std::optional<Relocation>
createRelocation(const MCFixup &Fixup,
const MCAsmBackend &MAB) const override {
- MCFixupKindInfo FKI = MAB.getFixupKindInfo(Fixup.getKind());
+ MCFixupKind FKind = Fixup.getKind();
+ MCFixupKindInfo FKI = MAB.getFixupKindInfo(FKind);
- assert(FKI.TargetOffset == 0 && "0-bit relocation offset expected");
- const uint64_t RelOffset = Fixup.getOffset();
+ switch (FKind) {
+ case MCFixupKind(AArch64::fixup_aarch64_pcrel_branch19):
+ case MCFixupKind(AArch64::fixup_aarch64_pcrel_branch14):
+ assert(FKI.TargetOffset == 5 && "5-bit relocation offset expected");
+ break;
+ default:
+ assert(FKI.TargetOffset == 0 && "0-bit relocation offset expected");
+ break;
+ }
uint32_t RelType;
- if (Fixup.getKind() == MCFixupKind(AArch64::fixup_aarch64_pcrel_call26))
+ if (FKind == MCFixupKind(AArch64::fixup_aarch64_pcrel_call26))
RelType = ELF::R_AARCH64_CALL26;
- else if (Fixup.getKind() ==
- MCFixupKind(AArch64::fixup_aarch64_pcrel_branch26))
+ else if (FKind == MCFixupKind(AArch64::fixup_aarch64_pcrel_branch26))
RelType = ELF::R_AARCH64_JUMP26;
+ else if (FKind == MCFixupKind(AArch64::fixup_aarch64_pcrel_branch19))
+ RelType = ELF::R_AARCH64_CONDBR19;
+ else if (FKind == MCFixupKind(AArch64::fixup_aarch64_pcrel_branch14))
+ RelType = ELF::R_AARCH64_TSTBR14;
else if (Fixup.isPCRel()) {
switch (FKI.TargetSize) {
default:
@@ -3735,7 +3746,7 @@ class AArch64MCPlusBuilder : public MCPlusBuilder {
break;
}
}
-
+ const uint64_t RelOffset = Fixup.getOffset();
auto [RelSymbol, RelAddend] = extractFixupExpr(Fixup);
return Relocation({RelOffset, RelSymbol, RelType, RelAddend, 0});
diff --git a/bolt/test/AArch64/conditional-tailcall-bcond.s b/bolt/test/AArch64/conditional-tailcall-bcond.s
new file mode 100644
index 0000000000000..7fd323fff939e
--- /dev/null
+++ b/bolt/test/AArch64/conditional-tailcall-bcond.s
@@ -0,0 +1,38 @@
+## Check support for conditional tail calls, ensure that conditional branches with
+## B.cond are correctly relocated. Assume that everything is in range with
+## --no-huge-pages.
+
+# RUN: llvm-mc -filetype=obj -triple=aarch64-unknown %s -o %t.o
+# RUN: ld.lld --emit-relocs %t.o -o %t.exe
+# RUN: llvm-bolt %t.exe -o %t.bolt --skip-funcs=_start --no-huge-pages
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED
+
+# NONBOLTED: [[#%x,FOONOBOLT:]] <foo>:
+# NONBOLTED: {{.*}} <_start>:
+# NONBOLTED: {{.*}} b.ge 0x[[#FOONOBOLT]] <foo>
+# NONBOLTED: {{.*}} b.pl 0x[[#FOONOBOLT]] <foo>
+
+# BOLTED: {{.*}} <foo.org.0>:
+# BOLTED: {{.*}} adrp x16, 0x[[#%x,FOO:]] <foo>
+# BOLTED: {{.*}} <_start>:
+# BOLTED: {{.*}} b.ge 0x[[#FOO]] <foo>
+# BOLTED: {{.*}} b.pl 0x[[#FOO]] <foo>
+# BOLTED: [[#FOO]] <foo>:
+
+ .type foo, at function
+ .globl foo
+foo:
+ .rept 3
+ nop
+ .endr
+ ret
+ .size foo, .-foo
+
+ .type _start, at function
+ .globl _start
+_start:
+ b.ge foo
+ b.pl foo
+ ret
+ .size _start, .-_start
diff --git a/bolt/test/AArch64/conditional-tailcall-cbz.s b/bolt/test/AArch64/conditional-tailcall-cbz.s
new file mode 100644
index 0000000000000..9176742562b7a
--- /dev/null
+++ b/bolt/test/AArch64/conditional-tailcall-cbz.s
@@ -0,0 +1,46 @@
+## Check support for conditional tail calls, ensure that conditional branches with
+## CBZ/CBNZ are correctly relocated. Assume that everything is in range with
+## --no-huge-pages.
+
+# RUN: llvm-mc -filetype=obj -triple=aarch64-unknown-unknown %s -o %t.o
+# RUN: ld.lld --emit-relocs %t.o -o %t.exe
+# RUN: llvm-bolt --no-huge-pages --skip-funcs=_start %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED
+
+# NONBOLTED: [[#%x,BAR:]] <bar>:
+# NONBOLTED: {{.*}} <_start>:
+# NONBOLTED: {{.*}} cbz w0, 0x[[#BAR]] <bar>
+# NONBOLTED: {{.*}} cbz x21, 0x[[#BAR]] <bar>
+# NONBOLTED: {{.*}} cbnz x10, 0x[[#BAR]] <bar>
+# NONBOLTED: {{.*}} cbnz wzr, 0x[[#BAR]] <bar>
+
+# BOLTED: [[#%x,BARORG:]] <bar.org.0>:
+# BOLTED: [[#BARORG]]: {{.*}} adrp x16, 0x[[#%x,BARNEW:]] <bar>
+# BOLTED: {{.*}} <_start>:
+# BOLTED: {{.*}} cbz w0, 0x[[#BARNEW]] <bar>
+# BOLTED: {{.*}} cbz x21, 0x[[#BARNEW]] <bar>
+# BOLTED: {{.*}} cbnz x10, 0x[[#BARNEW]] <bar>
+# BOLTED: {{.*}} cbnz wzr, 0x[[#BARNEW]] <bar>
+# BOLTED: [[#BARNEW]] <bar>:
+
+ .type bar, at function
+ .globl bar
+bar:
+ .rept 3
+ nop
+ .endr
+ ret
+ .size bar, .-bar
+
+ .type _start, at function
+ .globl _start
+_start:
+ cbz w0, bar
+ cbz x21, bar
+
+ cbnz x10, bar
+ cbnz wzr, bar
+ ret
+ .size _start, .-_start
+
\ No newline at end of file
diff --git a/bolt/test/AArch64/conditional-tailcall-condbr-range.s b/bolt/test/AArch64/conditional-tailcall-condbr-range.s
new file mode 100644
index 0000000000000..34077aa80cc52
--- /dev/null
+++ b/bolt/test/AArch64/conditional-tailcall-condbr-range.s
@@ -0,0 +1,61 @@
+## Ensure that conditional calls using the CONDBR19 relocation type are only relocated
+## if the target is range. Do not relocate the instruction if the target is out of range.
+## Ensure that this applies in both directions (+/- displacements).
+
+# RUN: llvm-mc -filetype=obj -triple=aarch64-unknown-unknown %s -o %t.o
+# RUN: ld.lld --emit-relocs --section-start=.text=0x300ff0 %t.o -o %t.exe
+# RUN: llvm-bolt --skip-funcs=_start --align-text=0x1000 %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED_FORWARD
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED_FORWARD
+
+# NONBOLTED_FORWARD: [[#%x,BAR:]] <bar>:
+# NONBOLTED_FORWARD: {{.*}} <_start>:
+# NONBOLTED_FORWARD: {{.*}} b.ge 0x[[#BAR]] <bar>
+# NONBOLTED_FORWARD: {{.*}} b.ge 0x[[#BAR]] <bar>
+
+# BOLTED_FORWARD: [[#%x,BAROLD:]] <bar.org.0>:
+# BOLTED_FORWARD: {{.*}} adrp x16, 0x[[#%x,RELOC:]] <bar>
+# BOLTED_FORWARD: {{.*}} <_start>:
+## The first branch is out of range of the relocated function and hence points to the
+## original function, whereas the second is within range of the relocated function
+## and hence points to the newly relocated function.
+# BOLTED_FORWARD: {{.*}} b.ge 0x[[#BAROLD]] <bar.org.0>
+# BOLTED_FORWARD: {{.*}} b.ge 0x[[#RELOC]] <bar>
+# BOLTED_FORWARD: [[#RELOC]] <bar>:
+
+# RUN: ld.lld --emit-relocs --section-start=.text=500FF0 %t.o -o %t.exe
+# RUN: llvm-bolt --skip-funcs=_start --align-text=0x1000 \
+# RUN: --custom-allocation-vma=0x400000 %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED_BACKWARD
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED_BACKWARD
+
+# NONBOLTED_BACKWARD: [[#%x,BAR:]] <bar>:
+# NONBOLTED_BACKWARD: {{.*}} <_start>:
+# NONBOLTED_BACKWARD: {{.*}} b.ge 0x[[#BAR]] <bar>
+# NONBOLTED_BACKWARD: {{.*}} b.ge 0x[[#BAR]] <bar>
+
+# BOLTED_BACKWARD: [[#%x,BAROLD:]] <bar.org.0>:
+# BOLTED_BACKWARD: {{.*}} adrp x16, 0x[[#%x,RELOC:]] <bar>
+# BOLTED_BACKWARD: {{.*}} <_start>:
+## The first branch is in range of the relocated function and hence is relocated to the
+## relocated function, whereas the second is not and so points to the original function.
+# BOLTED_BACKWARD: {{.*}} b.ge 0x[[#RELOC]] <bar>
+# BOLTED_BACKWARD: {{.*}} b.ge 0x[[#BAROLD]] <bar.org.0>
+# BOLTED_BACKWARD: [[#RELOC]] <bar>:
+
+ .type bar, at function
+ .globl bar
+bar:
+ .rept 3
+ nop
+ .endr
+ ret
+ .size bar, .-bar
+
+ .type _start, at function
+ .globl _start
+_start:
+ b.ge bar
+ b.ge bar
+ ret
+ .size _start, .-_start
diff --git a/bolt/test/AArch64/conditional-tailcall-tbz.s b/bolt/test/AArch64/conditional-tailcall-tbz.s
new file mode 100644
index 0000000000000..03a806bf0b365
--- /dev/null
+++ b/bolt/test/AArch64/conditional-tailcall-tbz.s
@@ -0,0 +1,46 @@
+## Check support for conditional tail calls, ensure that conditional branches with
+## TBZ/TBNZ are correctly relocated. Assume that everything is in range by
+## aligning the text sections of the binaries.
+
+# RUN: llvm-mc -filetype=obj -triple=aarch64-unknown-unknown %s -o %t.o
+# RUN: ld.lld --emit-relocs --section-start=.text=0x3ff000 %t.o -o %t.exe
+# RUN: llvm-bolt --skip-funcs=_start --align-text=0x1000 %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED
+
+# NONBOLTED: {{0*}}3ff000 <bar>:
+# NONBOLTED: {{.*}} <_start>:
+# NONBOLTED: {{.*}} tbz w0, #0x0, 0x3ff000 <bar>
+# NONBOLTED: {{.*}} tbz x10, #0x20, 0x3ff000 <bar>
+# NONBOLTED: {{.*}} tbnz w21, #0x1f, 0x3ff000 <bar>
+# NONBOLTED: {{.*}} tbnz xzr, #0x3f, 0x3ff000 <bar>
+
+# BOLTED: {{0*}}3ff000 <bar.org.0>:
+# BOLTED: {{0*}}3ff000: {{.*}} adrp x16, 0x[[#%x,BAR:]] <bar>
+# BOLTED: {{.*}} <_start>:
+# BOLTED: {{.*}} tbz w0, #0x0, 0x[[#BAR]] <bar>
+# BOLTED: {{.*}} tbz x10, #0x20, 0x[[#BAR]] <bar>
+# BOLTED: {{.*}} tbnz w21, #0x1f, 0x[[#BAR]] <bar>
+# BOLTED: {{.*}} tbnz xzr, #0x3f, 0x[[#BAR]] <bar>
+# BOLTED: [[#BAR]] <bar>:
+
+ .type bar, at function
+ .globl bar
+bar:
+ .rept 3
+ nop
+ .endr
+ ret
+ .size bar, .-bar
+
+ .type _start, at function
+ .globl _start
+_start:
+ tbz w0, #0, bar
+ tbz x10, #32, bar
+
+ tbnz w21, #31, bar
+ tbnz xzr, #63, bar
+ ret
+ .size _start, .-_start
+
\ No newline at end of file
diff --git a/bolt/test/AArch64/conditional-tailcall-tstbr-range.s b/bolt/test/AArch64/conditional-tailcall-tstbr-range.s
new file mode 100644
index 0000000000000..ed37d8201c2c3
--- /dev/null
+++ b/bolt/test/AArch64/conditional-tailcall-tstbr-range.s
@@ -0,0 +1,61 @@
+## Ensure that conditional calls using the TSTBR14 relocation type are only relocated
+## if the target is range. Do not relocate the instruction if the target is out of range.
+## Ensure that this applies in both directions (+/- displacements).
+
+# RUN: llvm-mc -filetype=obj -triple=aarch64-unknown-unknown %s -o %t.o
+# RUN: ld.lld --emit-relocs --section-start=.text=0x3f8ff0 %t.o -o %t.exe
+# RUN: llvm-bolt --skip-funcs=_start --align-text=0x1000 %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED_FORWARD
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED_FORWARD
+
+# NONBOLTED_FORWARD: [[#%x,BAR:]] <bar>:
+# NONBOLTED_FORWARD: {{.*}} <_start>:
+# NONBOLTED_FORWARD: {{.*}} tbz x10, #0x20, 0x[[#BAR]] <bar>
+# NONBOLTED_FORWARD: {{.*}} tbz x10, #0x20, 0x[[#BAR]] <bar>
+
+# BOLTED_FORWARD: [[#%x,BAROLD:]] <bar.org.0>:
+# BOLTED_FORWARD: {{.*}} adrp x16, 0x[[#%x,RELOC:]] <bar>
+# BOLTED_FORWARD: {{.*}} <_start>:
+## The first branch is out of range of the relocated function and hence points to the
+## original function, whereas the second is within range of the relocated function
+## and hence points to the newly relocated function.
+# BOLTED_FORWARD: {{.*}} tbz x10, #0x20, 0x[[#BAROLD]] <bar.org.0>
+# BOLTED_FORWARD: {{.*}} tbz x10, #0x20, 0x[[#RELOC]] <bar>
+# BOLTED_FORWARD: [[#RELOC]] <bar>:
+
+# RUN: ld.lld --emit-relocs --section-start=.text=0x408ff0 %t.o -o %t.exe
+# RUN: llvm-bolt --skip-funcs=_start --align-text=0x1000 \
+# RUN: --custom-allocation-vma=0x400000 %t.exe -o %t.bolt
+# RUN: llvm-objdump -d %t.exe 2>&1 | FileCheck %s --check-prefix=NONBOLTED_BACKWARD
+# RUN: llvm-objdump -d %t.bolt 2>&1 | FileCheck %s --check-prefix=BOLTED_BACKWARD
+
+# NONBOLTED_BACKWARD: [[#%x,BAR:]] <bar>:
+# NONBOLTED_BACKWARD: {{.*}} <_start>:
+# NONBOLTED_BACKWARD: {{.*}} tbz x10, #0x20, 0x[[#BAR]] <bar>
+# NONBOLTED_BACKWARD: {{.*}} tbz x10, #0x20, 0x[[#BAR]] <bar>
+
+# BOLTED_BACKWARD: [[#%x,BAROLD:]] <bar.org.0>:
+# BOLTED_BACKWARD: {{.*}} adrp x16, 0x[[#%x,RELOC:]] <bar>
+# BOLTED_BACKWARD: {{.*}} <_start>:
+## The first branch is in range of the relocated function and hence is relocated to the
+## relocated function, whereas the second is not and so points to the original function.
+# BOLTED_BACKWARD: {{.*}} tbz x10, #0x20, 0x[[#RELOC]] <bar>
+# BOLTED_BACKWARD: {{.*}} tbz x10, #0x20, 0x[[#BAROLD]] <bar.org.0>
+# BOLTED_BACKWARD: [[#RELOC]] <bar>:
+
+ .type bar, at function
+ .globl bar
+bar:
+ .rept 3
+ nop
+ .endr
+ ret
+ .size bar, .-bar
+
+ .type _start, at function
+ .globl _start
+_start:
+ tbz x10, #32, bar
+ tbz x10, #32, bar
+ ret
+ .size _start, .-_start
``````````
</details>
https://github.com/llvm/llvm-project/pull/227693
More information about the llvm-commits
mailing list