[llvm] [BOLT][RISCV] Improve relocations, jump tables, and split-function handling (PR #213919)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Aug 4 05:09:33 PDT 2026
https://github.com/Thrrreeee created https://github.com/llvm/llvm-project/pull/213919
This is a draft PR. I plan to split it into separate PRs by functionality for submission later. The current commits have been reorganized and refactored with the help of GPT-5.6. Before submitting the individual PRs, I will review every change, and validate the implementation with focused regression tests.
Extend BOLT's RISC-V backend to correctly rewrite a broader range of relocation and control-flow patterns.
- adds support for static IFUNC entries in `.iplt` and indirect PLT calls;
- rebuilds moved PC-relative and GOT relocation pairs, including RV32 and
relocation pairs separated by instructions or basic-block boundaries;
- preserves branch and compressed control-flow fixups when rescanning
ignored code;
- improves conditional tail-call handling;
- supports split function fragments using `AUIPC`/`JALR` trampolines;
- recognizes absolute and PC-relative jump-table dispatch sequences;
- tracks jump-table entry size and signedness, and retargets duplicated
jump-table dispatches;
- fixes atomic-add operand ordering.
In my testing (optimization with bb reorder and function reorder), instrumentation-based profiling delivers a 5~8% performance improvement for Clang. In contrast, optimization using profiles collected with perf may cause -1~-3% performance regression (our RISC-V CPU does't support branch record).
In my testing, These changes do not affect the time or memory consumption of BOLT when optimizing Clang binaries on AArch64/x86.
In X86, options: `-infer-stale-profile -reorder-blocks=ext-tsp -reorder-functions=cdsort -split-functions -split-all-cold -split-eh -dyno-stats -icf=safe -use-gnu-stack -inline-
small-functions -simplify-rodata-loads -plt=hot -icp=calls --icp-calls-topn=1 -enable-bat`
branch | wall time (s) | RSS
-- | -- | --
main | 35.73 | 4,320.28 MB
ours | 35.90 | 4,320.31 MB
In AArch64, options: `-infer-stale-profile -reorder-blocks=ext-tsp reorder-functions=cdsort -split-functions -split-all-cold -split-eh -dyno-stats -icf=safe -use-gnu-stack -inline-small-functions -plt=hot -enable-bat`
branch | wall time (s) | RSS
-- | -- | --
main | 71.24 | 5,452.75 MB
ours | 71.39 | 5,448.50 MB
>From 3b67f8fc55b0123acd59038f3c497ee3edb57e5a Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Wed, 15 Jul 2026 20:17:45 +0800
Subject: [PATCH 01/16] [BOLT][RISCV] Handle static IFUNC entries in .iplt
Recognize R_RISCV_IRELATIVE and RISC-V .iplt entries, preserve IFUNC PLT aliases, and register resolver addends as secondary function entry points after function boundaries are finalized.
---
bolt/include/bolt/Rewrite/RewriteInstance.h | 3 +-
bolt/lib/Core/Relocation.cpp | 6 ++-
bolt/lib/Rewrite/RewriteInstance.cpp | 53 +++++++++++++++++++--
bolt/test/RISCV/ifunc.s | 32 +++++++++++++
4 files changed, 88 insertions(+), 6 deletions(-)
create mode 100644 bolt/test/RISCV/ifunc.s
diff --git a/bolt/include/bolt/Rewrite/RewriteInstance.h b/bolt/include/bolt/Rewrite/RewriteInstance.h
index a624c056ada14..53e173fce929f 100644
--- a/bolt/include/bolt/Rewrite/RewriteInstance.h
+++ b/bolt/include/bolt/Rewrite/RewriteInstance.h
@@ -564,7 +564,8 @@ class RewriteInstance {
{".plt"}, {".plt.got"}, {".iplt"}, {nullptr}};
/// RISCV PLT sections.
- const PLTSectionInfo RISCV_PLTSections[2] = {{".plt"}, {nullptr}};
+ const PLTSectionInfo RISCV_PLTSections[3] = {{".plt"}, {".iplt", 16},
+ {nullptr}};
/// Return PLT information for a section with \p SectionName or nullptr
/// if the section is not PLT.
diff --git a/bolt/lib/Core/Relocation.cpp b/bolt/lib/Core/Relocation.cpp
index 6663abcffc7e8..74f60e007ab4c 100644
--- a/bolt/lib/Core/Relocation.cpp
+++ b/bolt/lib/Core/Relocation.cpp
@@ -128,6 +128,7 @@ static bool isSupportedRISCV(uint32_t Type) {
case ELF::R_RISCV_TPREL_ADD:
case ELF::R_RISCV_TPREL_LO12_I:
case ELF::R_RISCV_TPREL_LO12_S:
+ case ELF::R_RISCV_IRELATIVE:
case ELFReserved::R_RISCV_TPREL_I:
case ELFReserved::R_RISCV_TPREL_S:
return true;
@@ -240,6 +241,9 @@ static size_t getSizeForTypeRISCV(uint32_t Type) {
case ELF::R_RISCV_TLS_GD_HI20:
// See extractValueRISCV for why this is necessary.
return 8;
+ case ELF::R_RISCV_IRELATIVE:
+ // R_RISCV_IRELATIVE operates on a wordclass field.
+ return Relocation::Arch == Triple::riscv64 ? 8 : 4;
}
}
@@ -859,7 +863,7 @@ bool Relocation::isIRelative(uint32_t Type) {
return Type == ELF::R_AARCH64_IRELATIVE;
case Triple::riscv64:
case Triple::riscv32:
- llvm_unreachable("not implemented");
+ return Type == ELF::R_RISCV_IRELATIVE;
case Triple::x86_64:
return Type == ELF::R_X86_64_IRELATIVE;
}
diff --git a/bolt/lib/Rewrite/RewriteInstance.cpp b/bolt/lib/Rewrite/RewriteInstance.cpp
index 05ea606bdad7b..50661689a52ca 100644
--- a/bolt/lib/Rewrite/RewriteInstance.cpp
+++ b/bolt/lib/Rewrite/RewriteInstance.cpp
@@ -1385,6 +1385,30 @@ void RewriteInstance::discoverFileObjects() {
adjustFunctionBoundaries(MarkerSymbols);
splitUnmarkedTailFunctions(MarkerSymbols);
+ // R_RISCV_IRELATIVE addends name resolver entry points. LLD may
+ // canonicalize the only IFUNC symbol to the IPLT entry, leaving the resolver
+ // without a symbol. Function sizes are not final when dynamic relocations
+ // are first read, so record these secondary entries after boundary
+ // adjustment.
+ if (BC->isRISCV()) {
+ for (const BinarySection &Section : BC->allocatableSections()) {
+ for (const Relocation &Rel : Section.dynamicRelocations()) {
+ if (!Rel.isIRelative() || !Rel.Addend)
+ continue;
+ BinaryFunction *BF = BC->getBinaryFunctionContainingAddress(Rel.Addend);
+ if (!BF || BF->getAddress() == Rel.Addend)
+ continue;
+ if (BF->isInConstantIsland(Rel.Addend)) {
+ BC->errs() << "BOLT-ERROR: IFUNC resolver at 0x"
+ << Twine::utohexstr(Rel.Addend)
+ << " is in constant island of function " << *BF << '\n';
+ exit(1);
+ }
+ BF->addEntryPointAtOffset(Rel.Addend - BF->getAddress());
+ }
+ }
+ }
+
// Annotate functions with code/data markers in AArch64
for (auto &[Address, Type] : MarkerSymbols) {
auto *BF = BC->getBinaryFunctionContainingAddress(Address,
@@ -1876,7 +1900,7 @@ void RewriteInstance::createPLTBinaryFunction(uint64_t TargetAddress,
MCSymbol *Symbol = Rel->Symbol;
if (!Symbol) {
- if (BC->isRISCV() || !Rel->Addend || !Rel->isIRelative())
+ if (!Rel->Addend || !Rel->isIRelative())
return;
// IFUNC trampoline without symbol
@@ -1900,6 +1924,23 @@ void RewriteInstance::createPLTBinaryFunction(uint64_t TargetAddress,
else
BF->addAlternativeName(Symbol->getName().str() + "@PLT");
setPLTSymbol(BF, Symbol->getName());
+
+ if (Rel->isIRelative()) {
+ auto ResolverSyms = FileSymRefs.equal_range(Rel->Addend);
+ for (const SymbolRef &AliasSymbol : llvm::make_second_range(
+ llvm::make_range(ResolverSyms.first, ResolverSyms.second))) {
+ if (ELFSymbolRef(AliasSymbol).getELFType() != ELF::STT_GNU_IFUNC)
+ continue;
+ StringRef AliasName = cantFail(AliasSymbol.getName());
+ const std::string PLTName = AliasName.str() + "@PLT";
+ if (!BC->getBinaryDataByName(PLTName)) {
+ BF->addAlternativeName(PLTName);
+ BC->registerNameAtAddress(PLTName, EntryAddress, 0, EntrySize,
+ Section->getAlignment());
+ }
+ setPLTSymbol(BF, AliasName);
+ }
+ }
}
void RewriteInstance::disassemblePLTInstruction(const BinarySection &Section,
@@ -1988,8 +2029,9 @@ void RewriteInstance::disassemblePLTSectionRISCV(BinarySection &Section) {
}
};
- // Skip the first special entry since no relocation points to it.
- uint64_t InstrOffset = 32;
+ // Regular .plt has a first special entry with no relocations pointing to it,
+ // while static IFUNC .iplt entries start at the beginning of the section.
+ uint64_t InstrOffset = Section.getName() == ".iplt" ? 0 : 32;
while (InstrOffset < SectionSize) {
InstructionListType Instructions;
@@ -2774,7 +2816,10 @@ bool RewriteInstance::analyzeRelocation(
// Section symbols are marked as ST_Debug.
IsSectionRelocation = (cantFail(Symbol.getType()) == SymbolRef::ST_Debug);
// Check for PLT entry registered with symbol name
- if (!SymbolAddress && !IsWeakReference(Symbol) &&
+ const bool IsRISCVIFuncPLT =
+ BC->isRISCV() && RType == ELF::R_RISCV_CALL_PLT &&
+ ELFSymbolRef(Symbol).getELFType() == ELF::STT_GNU_IFUNC;
+ if ((!SymbolAddress || IsRISCVIFuncPLT) && !IsWeakReference(Symbol) &&
(IsAArch64 || BC->isRISCV())) {
const BinaryData *BD = BC->getPLTBinaryDataByName(SymbolName);
SymbolAddress = BD ? BD->getAddress() : 0;
diff --git a/bolt/test/RISCV/ifunc.s b/bolt/test/RISCV/ifunc.s
new file mode 100644
index 0000000000000..c73eb5cc89fa4
--- /dev/null
+++ b/bolt/test/RISCV/ifunc.s
@@ -0,0 +1,32 @@
+## Check that BOLT recognizes a non-preemptible IFUNC IPLT entry and tracks
+## the resolver when the linker canonicalizes the IFUNC symbol to the entry.
+
+# RUN: llvm-mc -filetype=obj -triple=riscv64 -mattr=+relax -o %t.o %s
+# RUN: ld.lld -pie -q -o %t.exe %t.o
+# RUN: llvm-bolt %t.exe -o %t.bolt --print-disasm --print-only=_start 2>&1 \
+# RUN: | FileCheck --check-prefix=BOLT %s
+# RUN: llvm-readelf -r -s %t.bolt | FileCheck --check-prefix=ELF %s
+
+# BOLT: Binary Function "_start
+# BOLT: auipc a0, %pcrel_hi(__BOLT_PSEUDO_.iplt)
+# BOLT-NOT: unable to get new address corresponding to input address
+# ELF: R_RISCV_IRELATIVE
+# ELF: FUNC{{.*}}ifunc0
+
+ .text
+ .globl _start
+ .type _start, @function
+_start:
+1:
+ auipc a0, %pcrel_hi(ifunc0)
+ addi a0, a0, %pcrel_lo(1b)
+
+ .globl func
+ .type func, @function
+func:
+ ret
+
+ .globl ifunc0
+ .type ifunc0, @gnu_indirect_function
+ifunc0:
+ ret
>From 56308c919ac683b5b81d22e92325957db18ed85d Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Tue, 21 Jul 2026 17:21:28 +0800
Subject: [PATCH 02/16] [BOLT][RISCV] Preserve IRELATIVE secondary entry
offsets
---
bolt/lib/Rewrite/RewriteInstance.cpp | 23 +++++++++++++++++++++++
bolt/test/RISCV/ifunc.s | 10 +++++++++-
2 files changed, 32 insertions(+), 1 deletion(-)
diff --git a/bolt/lib/Rewrite/RewriteInstance.cpp b/bolt/lib/Rewrite/RewriteInstance.cpp
index 50661689a52ca..bf42f8663c512 100644
--- a/bolt/lib/Rewrite/RewriteInstance.cpp
+++ b/bolt/lib/Rewrite/RewriteInstance.cpp
@@ -6460,6 +6460,29 @@ uint64_t RewriteInstance::getNewFunctionAddress(uint64_t OldAddress) {
}
uint64_t RewriteInstance::getNewFunctionOrDataAddress(uint64_t OldAddress) {
+ // Resolve secondary function entry points before the exact-address lookup.
+ // getBinaryFunctionAtAddress() can map a BinaryData symbol at a secondary
+ // entry back to its parent function and would then return the parent's main
+ // output address, losing the entry-point offset.
+ if (const BinaryFunction *BF =
+ BC->getBinaryFunctionContainingAddress(OldAddress)) {
+ if (BF->isEmitted() && BF->isMultiEntry()) {
+ uint64_t EntryAddress = 0;
+ BF->forEachEntryPoint([&](uint64_t Offset, const MCSymbol *Symbol) {
+ if (Offset && BF->getAddress() + Offset == OldAddress) {
+ if (auto SymbolInfo = Linker->lookupSymbolInfo(Symbol->getName()))
+ EntryAddress = SymbolInfo->Address;
+ else
+ EntryAddress = BF->translateInputToOutputAddress(OldAddress);
+ return false;
+ }
+ return true;
+ });
+ if (EntryAddress)
+ return EntryAddress;
+ }
+ }
+
if (uint64_t Function = getNewFunctionAddress(OldAddress))
return Function;
diff --git a/bolt/test/RISCV/ifunc.s b/bolt/test/RISCV/ifunc.s
index c73eb5cc89fa4..587c117ed0943 100644
--- a/bolt/test/RISCV/ifunc.s
+++ b/bolt/test/RISCV/ifunc.s
@@ -6,12 +6,20 @@
# RUN: llvm-bolt %t.exe -o %t.bolt --print-disasm --print-only=_start 2>&1 \
# RUN: | FileCheck --check-prefix=BOLT %s
# RUN: llvm-readelf -r -s %t.bolt | FileCheck --check-prefix=ELF %s
+## RV32 static binaries use a 32-bit wordclass IRELATIVE field. Also verify
+## that a resolver at a secondary entry point retains its +4 offset.
+# RUN: llvm-mc -filetype=obj -triple=riscv32 -mattr=+relax -o %t.32.o %s
+# RUN: ld.lld -q -o %t.32.exe %t.32.o
+# RUN: llvm-bolt %t.32.exe -o %t.32.bolt
+# RUN: llvm-readelf -r -s %t.32.bolt | FileCheck --check-prefix=RV32 %s
# BOLT: Binary Function "_start
# BOLT: auipc a0, %pcrel_hi(__BOLT_PSEUDO_.iplt)
# BOLT-NOT: unable to get new address corresponding to input address
-# ELF: R_RISCV_IRELATIVE
+# ELF: R_RISCV_IRELATIVE{{.*}}400044
# ELF: FUNC{{.*}}ifunc0
+# RV32: R_RISCV_IRELATIVE{{.*}}400044
+# RV32: FUNC{{.*}}ifunc0
.text
.globl _start
>From b6aef0f9f38640cd5aad625aa6e2b2c1d0a5dec3 Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Wed, 15 Jul 2026 20:16:34 +0800
Subject: [PATCH 03/16] [BOLT][RISCV] Rebuild moved PC-relative relocation
pairs
Reconstruct GOT high relocations from the actual matching PCREL_LO12 relocation instead of assuming the low instruction immediately follows AUIPC. Clear instruction-reference addends so moved PC-relative pairs are re-encoded from their new AUIPC location.
---
bolt/lib/Core/BinaryFunction.cpp | 48 ++++++++++++++++++++++-------
bolt/test/RISCV/reloc-got-moved.s | 35 +++++++++++++++++++++
bolt/test/RISCV/reloc-pcrel-moved.s | 31 +++++++++++++++++++
3 files changed, 103 insertions(+), 11 deletions(-)
create mode 100644 bolt/test/RISCV/reloc-got-moved.s
create mode 100644 bolt/test/RISCV/reloc-pcrel-moved.s
diff --git a/bolt/lib/Core/BinaryFunction.cpp b/bolt/lib/Core/BinaryFunction.cpp
index a6722389e5d50..8ed431ec4f6ea 100644
--- a/bolt/lib/Core/BinaryFunction.cpp
+++ b/bolt/lib/Core/BinaryFunction.cpp
@@ -32,6 +32,7 @@
#include "llvm/MC/MCInstPrinter.h"
#include "llvm/MC/MCRegisterInfo.h"
#include "llvm/MC/MCSymbol.h"
+#include "llvm/Object/ELF.h"
#include "llvm/Object/ObjectFile.h"
#include "llvm/Support/CommandLine.h"
#include "llvm/Support/Debug.h"
@@ -1505,11 +1506,38 @@ Error BinaryFunction::disassemble() {
for (auto Itr = Relocations.lower_bound(Offset),
ItrE = Relocations.lower_bound(Offset + Size);
Itr != ItrE; ++Itr) {
- const Relocation &Relocation = Itr->second;
- MCSymbol *Symbol = Relocation.Symbol;
+ const Relocation &Rel = Itr->second;
+ MCSymbol *Symbol = Rel.Symbol;
+ uint64_t Addend = Rel.Addend;
+ int64_t Value = Rel.Value;
+
+ if (Relocation::isGOT(Rel.Type)) {
+ for (const auto &KV : Relocations) {
+ const Relocation &LoRel = KV.second;
+ if (!Relocation::isInstructionReference(LoRel.Type))
+ continue;
+ if (LoRel.Value != getAddress() + Offset)
+ continue;
+
+ ErrorOr<uint64_t> HiContents =
+ BC.getUnsignedValueAtAddress(getAddress() + Offset, 4);
+ ErrorOr<uint64_t> LoContents = BC.getUnsignedValueAtAddress(
+ getAddress() + LoRel.Offset,
+ Relocation::getSizeForType(LoRel.Type));
+ assert(HiContents && LoContents &&
+ "cannot read RISC-V GOT relocation pair");
+
+ Value =
+ Relocation::extractValue(ELF::R_RISCV_PCREL_HI20, *HiContents,
+ getAddress() + Offset) +
+ Relocation::extractValue(LoRel.Type, *LoContents,
+ getAddress() + LoRel.Offset);
+ break;
+ }
+ }
- if (Relocation::isInstructionReference(Relocation.Type)) {
- uint64_t RefOffset = Relocation.Value - getAddress();
+ if (Relocation::isInstructionReference(Rel.Type)) {
+ uint64_t RefOffset = Rel.Value - getAddress();
LabelsMapType::iterator LI = InstructionLabels.find(RefOffset);
if (LI == InstructionLabels.end()) {
@@ -1518,21 +1546,19 @@ Error BinaryFunction::disassemble() {
} else {
Symbol = LI->second;
}
+ Addend = 0;
}
- uint64_t Addend = Relocation.Addend;
-
// For GOT relocations, create a reference against GOT entry ignoring
// the relocation symbol.
- if (Relocation::isGOT(Relocation.Type)) {
- assert(Relocation::isPCRelative(Relocation.Type) &&
+ if (Relocation::isGOT(Rel.Type)) {
+ assert(Relocation::isPCRelative(Rel.Type) &&
"GOT relocation must be PC-relative on RISC-V");
Symbol = BC.registerNameAtAddress("__BOLT_got_zero", 0, 0, 0);
- Addend = Relocation.Value + Relocation.Offset + getAddress();
+ Addend = Value + Rel.Offset + getAddress();
}
- int64_t Value = Relocation.Value;
const bool Result = BC.MIB->replaceImmWithSymbolRef(
- Instruction, Symbol, Addend, Ctx.get(), Value, Relocation.Type);
+ Instruction, Symbol, Addend, Ctx.get(), Value, Rel.Type);
(void)Result;
assert(Result && "cannot replace immediate with relocation");
}
diff --git a/bolt/test/RISCV/reloc-got-moved.s b/bolt/test/RISCV/reloc-got-moved.s
new file mode 100644
index 0000000000000..aa02cb49f8cfd
--- /dev/null
+++ b/bolt/test/RISCV/reloc-got-moved.s
@@ -0,0 +1,35 @@
+## Check that R_RISCV_GOT_HI20 relocations are re-encoded correctly when the
+## matching %pcrel_lo is not in the instruction immediately after the AUIPC.
+
+# RUN: llvm-mc -triple riscv64 -mattr=+c -filetype=obj -o %t.o %s
+# RUN: ld.lld -q -o %t.exe %t.o
+# RUN: llvm-bolt %t.exe -o %t.bolt -reorder-functions=cdsort
+# RUN: llvm-objdump -d %t.bolt | FileCheck %s
+
+# CHECK: Disassembly of section .text:
+# CHECK: <_start>:
+# CHECK-NEXT: auipc a0, 0xffc12
+# CHECK-NEXT: li a1, 0x7
+# CHECK-NEXT: li a2, 0x9
+# CHECK-NEXT: ld a0, 0x1e0(a0)
+# CHECK-NEXT: ret
+
+ .data
+ .p2align 12
+ .globl d
+d:
+ .dword 0
+
+ .text
+ .globl _start
+ .type _start, @function
+_start:
+ nop
+1:
+ auipc a0, %got_pcrel_hi(d)
+ addi a1, zero, 7
+ addi a2, zero, 9
+ ld a0, %pcrel_lo(1b)(a0)
+ ret
+ .reloc 0, R_RISCV_NONE
+ .size _start, .-_start
diff --git a/bolt/test/RISCV/reloc-pcrel-moved.s b/bolt/test/RISCV/reloc-pcrel-moved.s
new file mode 100644
index 0000000000000..f85a6385876a2
--- /dev/null
+++ b/bolt/test/RISCV/reloc-pcrel-moved.s
@@ -0,0 +1,31 @@
+## Check that R_RISCV_PCREL_LO12 relocations are re-encoded relative to the
+## moved AUIPC instruction instead of retaining the input addend.
+
+# RUN: llvm-mc -triple riscv64 -mattr=+c -filetype=obj -o %t.o %s
+# RUN: ld.lld -q -o %t.exe %t.o
+# RUN: llvm-bolt %t.exe -o %t.bolt -reorder-functions=cdsort
+# RUN: llvm-objdump -d %t.bolt | FileCheck %s
+
+# CHECK: Disassembly of section .text:
+# CHECK: <_start>:
+# CHECK-NEXT: auipc a0, 0xffc13
+# CHECK-NEXT: ld a0, 0x0(a0)
+# CHECK-NEXT: ret
+
+ .data
+ .p2align 12
+ .globl d
+d:
+ .dword 0
+
+ .text
+ .globl _start
+ .type _start, @function
+_start:
+ nop
+1:
+ auipc a0, %pcrel_hi(d)
+ ld a0, %pcrel_lo(1b)(a0)
+ ret
+ .reloc 0, R_RISCV_NONE
+ .size _start, .-_start
>From 7f9f86a561a099449bd4d9dbc745f46c5a756261 Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Tue, 21 Jul 2026 17:21:28 +0800
Subject: [PATCH 04/16] [BOLT][RISCV] Handle PC-relative instruction references
on RV32
---
bolt/lib/Core/Relocation.cpp | 2 +-
bolt/test/RISCV/reloc-bb-split-rv32.s | 12 ++++-----
bolt/test/RISCV/reloc-got-moved-rv32.s | 35 ++++++++++++++++++++++++++
3 files changed, 42 insertions(+), 7 deletions(-)
create mode 100644 bolt/test/RISCV/reloc-got-moved-rv32.s
diff --git a/bolt/lib/Core/Relocation.cpp b/bolt/lib/Core/Relocation.cpp
index 74f60e007ab4c..155faed35c0ba 100644
--- a/bolt/lib/Core/Relocation.cpp
+++ b/bolt/lib/Core/Relocation.cpp
@@ -884,7 +884,7 @@ bool Relocation::isTLS(uint32_t Type) {
}
bool Relocation::isInstructionReference(uint32_t Type) {
- if (Arch != Triple::riscv64)
+ if (Arch != Triple::riscv64 && Arch != Triple::riscv32)
return false;
switch (Type) {
diff --git a/bolt/test/RISCV/reloc-bb-split-rv32.s b/bolt/test/RISCV/reloc-bb-split-rv32.s
index 0ad3168fb983d..a434f5c71bd63 100644
--- a/bolt/test/RISCV/reloc-bb-split-rv32.s
+++ b/bolt/test/RISCV/reloc-bb-split-rv32.s
@@ -20,10 +20,10 @@ _start:
/// basic block should start there.
// CHECK-LABEL: {{^}}.LBB00
// CHECK: nop
-// CHECK-LABEL: {{^}}.Ltmp0
-// CHECK: auipc t0, %pcrel_hi(d)
-// CHECK-NEXT: lw t0, %pcrel_lo({{.*}})(t0)
-// CHECK-NEXT: j .Ltmp0
+// CHECK: {{^}}[[BRANCH_LABEL:.Ltmp[0-9]+]]
+// CHECK: auipc t0, %pcrel_hi(d) # Label: [[HI_LABEL:.Ltmp[0-9]+]]
+// CHECK-NEXT: lw t0, %pcrel_lo([[HI_LABEL]])(t0)
+// CHECK-NEXT: j [[BRANCH_LABEL]]
nop
1:
auipc t0, %pcrel_hi(d)
@@ -34,8 +34,8 @@ _start:
/// start there.
// CHECK-LABEL: {{^}}.LFT0
// CHECK: nop
-// CHECK: auipc t0, %pcrel_hi(d)
-// CHECK-NEXT: lw t0, %pcrel_lo({{.*}})(t0)
+// CHECK: auipc t0, %pcrel_hi(d) # Label: [[SECOND_HI:.Ltmp[0-9]+]]
+// CHECK-NEXT: lw t0, %pcrel_lo([[SECOND_HI]])(t0)
// CHECK-NEXT: ret
nop
1:
diff --git a/bolt/test/RISCV/reloc-got-moved-rv32.s b/bolt/test/RISCV/reloc-got-moved-rv32.s
new file mode 100644
index 0000000000000..5dd1d5a0ee48b
--- /dev/null
+++ b/bolt/test/RISCV/reloc-got-moved-rv32.s
@@ -0,0 +1,35 @@
+## Check that the RV32 R_RISCV_GOT_HI20/%pcrel_lo pair is rebuilt when the
+## matching low instruction is not immediately after AUIPC.
+
+# RUN: llvm-mc -triple riscv32 -mattr=+c -filetype=obj -o %t.o %s
+# RUN: ld.lld -q -o %t.exe %t.o
+# RUN: llvm-bolt %t.exe -o %t.bolt -reorder-functions=cdsort
+# RUN: llvm-objdump -d %t.bolt | FileCheck %s
+
+# CHECK: Disassembly of section .text:
+# CHECK: <_start>:
+# CHECK-NEXT: auipc a0, 0xffc12
+# CHECK-NEXT: li a1, 0x7
+# CHECK-NEXT: li a2, 0x9
+# CHECK-NEXT: lw a0, 0x128(a0)
+# CHECK-NEXT: ret
+
+ .data
+ .p2align 12
+ .globl d
+d:
+ .word 0
+
+ .text
+ .globl _start
+ .type _start, @function
+_start:
+ nop
+1:
+ auipc a0, %got_pcrel_hi(d)
+ addi a1, zero, 7
+ addi a2, zero, 9
+ lw a0, %pcrel_lo(1b)(a0)
+ ret
+ .reloc 0, R_RISCV_NONE
+ .size _start, .-_start
>From 9e16c5d24239fc42e31ebcb352d9beaeb851e9dc Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Wed, 15 Jul 2026 20:37:33 +0800
Subject: [PATCH 05/16] [BOLT][RISCV] Preserve branches when rescanning ignored
code
Create relocations for RISC-V MC fixups and re-encode JAL, branch, and compressed control-flow immediates while preserving the original instruction bits. Avoid registering invalid secondary entries in constant islands, and use relaxable PseudoTAIL instructions when rebuilding long tail calls.
---
bolt/include/bolt/Core/Relocation.h | 6 +-
bolt/lib/Core/BinaryContext.cpp | 13 ++--
bolt/lib/Core/BinaryFunction.cpp | 16 +++++
bolt/lib/Core/BinarySection.cpp | 25 +++++--
bolt/lib/Core/Relocation.cpp | 68 +++++++++++++++++-
bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp | 72 ++++++++++++++++++-
.../test/RISCV/constant-island-entry-rescan.s | 44 ++++++++++++
bolt/test/RISCV/ignored-func-short-branch.s | 58 +++++++++++++++
8 files changed, 284 insertions(+), 18 deletions(-)
create mode 100644 bolt/test/RISCV/constant-island-entry-rescan.s
create mode 100644 bolt/test/RISCV/ignored-func-short-branch.s
diff --git a/bolt/include/bolt/Core/Relocation.h b/bolt/include/bolt/Core/Relocation.h
index 9bc94d6484b70..8ee3e2587cb9c 100644
--- a/bolt/include/bolt/Core/Relocation.h
+++ b/bolt/include/bolt/Core/Relocation.h
@@ -94,8 +94,10 @@ class Relocation {
/// Skip relocations that we don't want to handle in BOLT
static bool skipRelocationType(uint32_t Type);
- /// Adjust value depending on relocation type (make it PC relative or not).
- static uint64_t encodeValue(uint32_t Type, uint64_t Value, uint64_t PC);
+ /// Encode \p Value according to the relocation type. \p OldValue is used only
+ /// by RISC-V instruction relocations that preserve non-immediate bits.
+ static uint64_t encodeValue(uint32_t Type, uint64_t Value, uint64_t PC,
+ uint64_t OldValue = 0);
/// Return true if there are enough bits to encode the relocation value.
static bool canEncodeValue(uint32_t Type, uint64_t Value, uint64_t PC);
diff --git a/bolt/lib/Core/BinaryContext.cpp b/bolt/lib/Core/BinaryContext.cpp
index 13d7e4bc1a5d6..5063e957a2be8 100644
--- a/bolt/lib/Core/BinaryContext.cpp
+++ b/bolt/lib/Core/BinaryContext.cpp
@@ -562,12 +562,13 @@ MCSymbol *BinaryContext::handleExternalBranchTarget(uint64_t Address,
<< Twine::utohexstr(Address) << "; ignoring both functions\n";
IsValid = false;
}
- if (Target.isInConstantIsland(Address)) {
- this->errs() << "BOLT-WARNING: ignoring entry point at address 0x"
- << Twine::utohexstr(Address)
- << " in constant island of function " << Target << '\n';
- IsValid = false;
- }
+ }
+
+ if (Target.isInConstantIsland(Address)) {
+ this->errs() << "BOLT-WARNING: ignoring entry point at address 0x"
+ << Twine::utohexstr(Address)
+ << " in constant island of function " << Target << '\n';
+ IsValid = false;
}
if (!IsValid) {
diff --git a/bolt/lib/Core/BinaryFunction.cpp b/bolt/lib/Core/BinaryFunction.cpp
index 8ed431ec4f6ea..53787b68b7695 100644
--- a/bolt/lib/Core/BinaryFunction.cpp
+++ b/bolt/lib/Core/BinaryFunction.cpp
@@ -1909,6 +1909,22 @@ bool BinaryFunction::scanExternalRefs() {
continue;
}
+ if (BC.isRISCV()) {
+ switch (Rel->Type) {
+ default:
+ break;
+ case ELF::R_RISCV_BRANCH:
+ case ELF::R_RISCV_JAL:
+ case ELF::R_RISCV_RVC_BRANCH:
+ case ELF::R_RISCV_RVC_JUMP:
+ if (BinaryFunction *TargetBF = BC.getFunctionForSymbol(Rel->Symbol)) {
+ TargetBF->setNeedsPatch(true);
+ continue;
+ }
+ break;
+ }
+ }
+
if (BC.isAArch64()) {
// Allow the relocation to be skipped in case of the overflow during the
// relocation value encoding.
diff --git a/bolt/lib/Core/BinarySection.cpp b/bolt/lib/Core/BinarySection.cpp
index a8620ba83ebfb..9205523661357 100644
--- a/bolt/lib/Core/BinarySection.cpp
+++ b/bolt/lib/Core/BinarySection.cpp
@@ -191,18 +191,31 @@ void BinarySection::flushPendingRelocations(raw_fd_ostream &OS,
++SkippedPendingRelocations;
continue;
}
+
+ const size_t RelocSize = Relocation::getSizeForType(Reloc.Type);
+ uint64_t OldValue = 0;
+ if (Reloc.Offset + RelocSize <= getContents().size()) {
+ ArrayRef<uint8_t> Bytes(reinterpret_cast<const uint8_t *>(
+ getContents().data() + Reloc.Offset),
+ RelocSize);
+ if (BC.AsmInfo->isLittleEndian()) {
+ for (unsigned I = 0; I < Bytes.size(); ++I)
+ OldValue |= uint64_t(Bytes[I]) << (I * 8);
+ } else {
+ for (uint8_t Byte : Bytes)
+ OldValue = (OldValue << 8) | Byte;
+ }
+ }
Value = Relocation::encodeValue(Reloc.Type, Value,
- SectionAddress + Reloc.Offset);
+ SectionAddress + Reloc.Offset, OldValue);
safePWrite(OS, reinterpret_cast<const char *>(&Value),
- Relocation::getSizeForType(Reloc.Type),
- SectionFileOffset + Reloc.Offset);
+ RelocSize, SectionFileOffset + Reloc.Offset);
LLVM_DEBUG(
dbgs() << "BOLT-DEBUG: writing value 0x" << Twine::utohexstr(Value)
- << " of size " << Relocation::getSizeForType(Reloc.Type)
- << " at section offset 0x" << Twine::utohexstr(Reloc.Offset)
- << " address 0x"
+ << " of size " << RelocSize << " at section offset 0x"
+ << Twine::utohexstr(Reloc.Offset) << " address 0x"
<< Twine::utohexstr(SectionAddress + Reloc.Offset)
<< " file offset 0x"
<< Twine::utohexstr(SectionFileOffset + Reloc.Offset) << '\n';);
diff --git a/bolt/lib/Core/Relocation.cpp b/bolt/lib/Core/Relocation.cpp
index 155faed35c0ba..c1916f5a0967b 100644
--- a/bolt/lib/Core/Relocation.cpp
+++ b/bolt/lib/Core/Relocation.cpp
@@ -339,13 +339,74 @@ static uint64_t canEncodeValueRISCV(uint32_t Type, uint64_t Value,
}
}
-static uint64_t encodeValueRISCV(uint32_t Type, uint64_t Value, uint64_t PC) {
+static uint64_t encodeValueRISCV(uint32_t Type, uint64_t Value, uint64_t PC,
+ uint64_t OldValue) {
+ auto encodePCRel = [&](uint32_t Type, uint64_t Value) -> uint64_t {
+ const int64_t PCRelValue =
+ static_cast<int64_t>(Value) - static_cast<int64_t>(PC);
+ assert((PCRelValue & 0x1) == 0 && "RISC-V branch target is misaligned");
+ const uint64_t EncValue = static_cast<uint64_t>(PCRelValue);
+ switch (Type) {
+ default:
+ llvm_unreachable("unsupported relocation");
+ case ELF::R_RISCV_BRANCH: {
+ assert(isInt<13>(PCRelValue) && "RISC-V branch target out of range");
+ const uint64_t Sbit = (EncValue >> 12) & 0x1;
+ const uint64_t Hi1 = (EncValue >> 11) & 0x1;
+ const uint64_t Mid6 = (EncValue >> 5) & 0x3f;
+ const uint64_t Lo4 = (EncValue >> 1) & 0xf;
+ return (OldValue & 0x01fff07f) | (Sbit << 31) | (Mid6 << 25) |
+ (Lo4 << 8) | (Hi1 << 7);
+ }
+ case ELF::R_RISCV_JAL: {
+ assert(isInt<21>(PCRelValue) && "RISC-V jump target out of range");
+ const uint64_t Sbit = (EncValue >> 20) & 0x1;
+ const uint64_t Hi8 = (EncValue >> 12) & 0xff;
+ const uint64_t Mid1 = (EncValue >> 11) & 0x1;
+ const uint64_t Lo10 = (EncValue >> 1) & 0x3ff;
+ return (OldValue & 0xfff) | (Sbit << 31) | (Lo10 << 21) | (Mid1 << 20) |
+ (Hi8 << 12);
+ }
+ case ELF::R_RISCV_RVC_BRANCH: {
+ assert(isInt<9>(PCRelValue) &&
+ "RISC-V compressed branch target out of range");
+ const uint64_t Bit8 = (EncValue >> 8) & 0x1;
+ const uint64_t Bit7_6 = (EncValue >> 6) & 0x3;
+ const uint64_t Bit5 = (EncValue >> 5) & 0x1;
+ const uint64_t Bit4_3 = (EncValue >> 3) & 0x3;
+ const uint64_t Bit2_1 = (EncValue >> 1) & 0x3;
+ return (OldValue & 0xe383) | (Bit8 << 12) | (Bit4_3 << 10) |
+ (Bit7_6 << 5) | (Bit2_1 << 3) | (Bit5 << 2);
+ }
+ case ELF::R_RISCV_RVC_JUMP: {
+ assert(isInt<12>(PCRelValue) &&
+ "RISC-V compressed jump target out of range");
+ const uint64_t Bit11 = (EncValue >> 11) & 0x1;
+ const uint64_t Bit4 = (EncValue >> 4) & 0x1;
+ const uint64_t Bit9_8 = (EncValue >> 8) & 0x3;
+ const uint64_t Bit10 = (EncValue >> 10) & 0x1;
+ const uint64_t Bit6 = (EncValue >> 6) & 0x1;
+ const uint64_t Bit7 = (EncValue >> 7) & 0x1;
+ const uint64_t Bit3_1 = (EncValue >> 1) & 0x7;
+ const uint64_t Bit5 = (EncValue >> 5) & 0x1;
+ return (OldValue & 0xe003) | (Bit11 << 12) | (Bit4 << 11) |
+ (Bit9_8 << 9) | (Bit10 << 8) | (Bit6 << 7) | (Bit7 << 6) |
+ (Bit3_1 << 3) | (Bit5 << 2);
+ }
+ }
+ };
+
switch (Type) {
default:
llvm_unreachable("unsupported relocation");
case ELF::R_RISCV_32:
case ELF::R_RISCV_64:
break;
+ case ELF::R_RISCV_BRANCH:
+ case ELF::R_RISCV_JAL:
+ case ELF::R_RISCV_RVC_BRANCH:
+ case ELF::R_RISCV_RVC_JUMP:
+ return encodePCRel(Type, Value);
}
return Value;
}
@@ -770,7 +831,8 @@ bool Relocation::skipRelocationType(uint32_t Type) {
}
}
-uint64_t Relocation::encodeValue(uint32_t Type, uint64_t Value, uint64_t PC) {
+uint64_t Relocation::encodeValue(uint32_t Type, uint64_t Value, uint64_t PC,
+ uint64_t OldValue) {
switch (Arch) {
default:
llvm_unreachable("Unsupported architecture");
@@ -778,7 +840,7 @@ uint64_t Relocation::encodeValue(uint32_t Type, uint64_t Value, uint64_t PC) {
return encodeValueAArch64(Type, Value, PC);
case Triple::riscv64:
case Triple::riscv32:
- return encodeValueRISCV(Type, Value, PC);
+ return encodeValueRISCV(Type, Value, PC, OldValue);
case Triple::x86_64:
return encodeValueX86(Type, Value, PC);
}
diff --git a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
index d1a0572277874..30feb994bee8f 100644
--- a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
+++ b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
@@ -11,6 +11,7 @@
//===----------------------------------------------------------------------===//
#include "MCTargetDesc/RISCVMCAsmInfo.h"
+#include "MCTargetDesc/RISCVFixupKinds.h"
#include "MCTargetDesc/RISCVMCTargetDesc.h"
#include "bolt/Core/MCPlusBuilder.h"
#include "llvm/BinaryFormat/ELF.h"
@@ -263,7 +264,8 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
void createLongTailCall(InstructionListType &Seq, const MCSymbol *Target,
MCContext *Ctx) override {
- createShortJmp(Seq, Target, Ctx, /*IsTailCall*/ true);
+ Seq.emplace_back();
+ createTailCall(Seq.back(), Target, Ctx);
}
void createTailCall(MCInst &Inst, const MCSymbol *Target,
@@ -640,6 +642,74 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
return Insts;
}
+ std::optional<Relocation>
+ createRelocation(const MCFixup &Fixup,
+ const MCAsmBackend &MAB) const override {
+ (void)MAB;
+ const uint64_t RelOffset = Fixup.getOffset();
+
+ uint32_t RelType;
+ if (mc::isRelocation(Fixup.getKind())) {
+ RelType = Fixup.getKind();
+ } else if (Fixup.isPCRel()) {
+ switch (Fixup.getKind()) {
+ default:
+ return std::nullopt;
+ case FK_Data_4:
+ RelType = ELF::R_RISCV_32_PCREL;
+ break;
+ case RISCV::fixup_riscv_pcrel_hi20:
+ RelType = ELF::R_RISCV_PCREL_HI20;
+ break;
+ case RISCV::fixup_riscv_pcrel_lo12_i:
+ RelType = ELF::R_RISCV_PCREL_LO12_I;
+ break;
+ case RISCV::fixup_riscv_pcrel_lo12_s:
+ RelType = ELF::R_RISCV_PCREL_LO12_S;
+ break;
+ case RISCV::fixup_riscv_jal:
+ RelType = ELF::R_RISCV_JAL;
+ break;
+ case RISCV::fixup_riscv_branch:
+ RelType = ELF::R_RISCV_BRANCH;
+ break;
+ case RISCV::fixup_riscv_rvc_jump:
+ RelType = ELF::R_RISCV_RVC_JUMP;
+ break;
+ case RISCV::fixup_riscv_rvc_branch:
+ RelType = ELF::R_RISCV_RVC_BRANCH;
+ break;
+ case RISCV::fixup_riscv_call:
+ case RISCV::fixup_riscv_call_plt:
+ RelType = ELF::R_RISCV_CALL_PLT;
+ break;
+ }
+ } else {
+ switch (Fixup.getKind()) {
+ default:
+ return std::nullopt;
+ case FK_Data_4:
+ RelType = ELF::R_RISCV_32;
+ break;
+ case FK_Data_8:
+ RelType = ELF::R_RISCV_64;
+ break;
+ case RISCV::fixup_riscv_hi20:
+ RelType = ELF::R_RISCV_HI20;
+ break;
+ case RISCV::fixup_riscv_lo12_i:
+ RelType = ELF::R_RISCV_LO12_I;
+ break;
+ case RISCV::fixup_riscv_lo12_s:
+ RelType = ELF::R_RISCV_LO12_S;
+ break;
+ }
+ }
+
+ auto [RelSymbol, RelAddend] = extractFixupExpr(Fixup);
+ return Relocation({RelOffset, RelSymbol, RelType, RelAddend, 0});
+ }
+
InstructionListType createInstrIncMemory(const MCSymbol *Target,
MCContext *Ctx, bool IsLeaf,
unsigned CodePointerSize) override {
diff --git a/bolt/test/RISCV/constant-island-entry-rescan.s b/bolt/test/RISCV/constant-island-entry-rescan.s
new file mode 100644
index 0000000000000..7c49ec511e6d8
--- /dev/null
+++ b/bolt/test/RISCV/constant-island-entry-rescan.s
@@ -0,0 +1,44 @@
+# This test verifies that BOLT does not crash while rescanning references from
+# an ignored function when its branch target lies in another function's constant
+# island.
+
+# RUN: llvm-mc -triple riscv64 -mattr=+c -filetype=obj -o %t.o %s
+# RUN: ld.lld -q -o %t.exe %t.o
+# RUN: llvm-bolt %t.exe -o %t.bolt -reorder-blocks=ext-tsp \
+# RUN: -reorder-functions=cdsort -simplify-rodata-loads -plt=hot \
+# RUN: -split-eh -use-gnu-stack 2>&1 | FileCheck %s
+
+# CHECK: BOLT-WARNING: corrupted control flow detected in function source:
+# CHECK-SAME: an external branch/call targets an invalid instruction
+# CHECK-SAME: in function target at address 0x{{[0-9a-f]+}}; ignoring both functions
+# CHECK: BOLT-WARNING: ignoring entry point at address 0x{{[0-9a-f]+}} in constant island of function target
+# CHECK-NOT: cannot add entry point that points to constant data
+
+ .text
+ .globl target
+ .type target, @function
+target:
+ j after_data
+
+data_label:
+ .word 0
+
+after_data:
+ ret
+ .size target, .-target
+
+ .globl source
+ .type source, @function
+source:
+ j data_label
+ ret
+ .size source, .-source
+
+ .globl _start
+ .type _start, @function
+_start:
+ call source
+ ret
+ .size _start, .-_start
+
+ .reloc 0, R_RISCV_NONE
diff --git a/bolt/test/RISCV/ignored-func-short-branch.s b/bolt/test/RISCV/ignored-func-short-branch.s
new file mode 100644
index 0000000000000..f965406732ab2
--- /dev/null
+++ b/bolt/test/RISCV/ignored-func-short-branch.s
@@ -0,0 +1,58 @@
+# This test verifies that rescanning references in an ignored RISC-V function
+# does not try to redirect short branch/jump relocations to moved code.
+
+# RUN: llvm-mc -triple riscv64 -mattr=+c -filetype=obj -o %t.o %s
+# RUN: ld.lld -q -o %t.exe %t.o
+# RUN: llvm-bolt %t.exe -o %t.bolt -reorder-blocks=ext-tsp \
+# RUN: -reorder-functions=cdsort -simplify-rodata-loads -plt=hot \
+# RUN: -split-eh -use-gnu-stack 2>&1 | FileCheck %s
+
+# CHECK: BOLT-WARNING: corrupted control flow detected in function source:
+# CHECK: BOLT-WARNING: ignoring entry point at address 0x{{[0-9a-f]+}} in constant island of function target
+# CHECK-NOT: unsupported relocation
+# CHECK-NOT: could not find corresponding %pcrel_hi
+# CHECK-NOT: target out of range
+
+ .text
+ .globl target
+ .type target, @function
+target:
+ j after_data
+
+data_label:
+ .word 0
+
+after_data:
+ ret
+ .size target, .-target
+
+ .globl callee
+ .type callee, @function
+callee:
+ nop
+ nop
+ nop
+ nop
+ nop
+ nop
+ nop
+ nop
+ ret
+ .size callee, .-callee
+
+ .globl source
+ .type source, @function
+source:
+ j data_label
+ beqz a0, callee
+ ret
+ .size source, .-source
+
+ .globl _start
+ .type _start, @function
+_start:
+ call source
+ ret
+ .size _start, .-_start
+
+ .reloc 0, R_RISCV_NONE
>From 3e47a14b03a841b8ae3bfb7dc6791e9dd3babec9 Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Wed, 15 Jul 2026 20:19:35 +0800
Subject: [PATCH 06/16] [BOLT][RISCV] Handle conditional tail calls
Reject conditional branch opcodes in the unconditional jump-to-tail-call conversion and teach the reverse conversion to clear RISC-V tail-call annotations. Recognize PseudoCALL and PseudoTAIL symbol operands during lowering.
---
bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp | 12 +++++++-
bolt/test/RISCV/conditional-tail-call.s | 30 ++++++++++++++++++++
2 files changed, 41 insertions(+), 1 deletion(-)
create mode 100644 bolt/test/RISCV/conditional-tail-call.s
diff --git a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
index 30feb994bee8f..e8b2700b6f010 100644
--- a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
+++ b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
@@ -216,7 +216,7 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
switch (Inst.getOpcode()) {
default:
- llvm_unreachable("unsupported tail call opcode");
+ return false;
case RISCV::JAL:
case RISCV::JALR:
case RISCV::C_J:
@@ -228,6 +228,14 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
return true;
}
+ bool convertTailCallToJmp(MCInst &Inst) override {
+ removeAnnotation(Inst, MCPlus::MCAnnotation::kTailCall);
+ clearOffset(Inst);
+ if (getConditionalTailCall(Inst))
+ unsetConditionalTailCall(Inst);
+ return true;
+ }
+
void createReturn(MCInst &Inst) const override {
// TODO "c.jr ra" when RVC is enabled
Inst.setOpcode(RISCV::JALR);
@@ -330,6 +338,8 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
default:
return false;
case RISCV::C_J:
+ case RISCV::PseudoCALL:
+ case RISCV::PseudoTAIL:
OpNum = 0;
return true;
case RISCV::AUIPC:
diff --git a/bolt/test/RISCV/conditional-tail-call.s b/bolt/test/RISCV/conditional-tail-call.s
new file mode 100644
index 0000000000000..273a3a40f452f
--- /dev/null
+++ b/bolt/test/RISCV/conditional-tail-call.s
@@ -0,0 +1,30 @@
+## Check that a conditional branch to another function is handled as a
+## conditional tail call. The target-specific jump-to-tail-call conversion must
+## return false for conditional branches so the generic code records the CTC
+## annotation instead of marking the branch as an unconditional tail call.
+
+# RUN: llvm-mc -triple riscv64 -mattr=+c -filetype=obj -o %t.o %s
+# RUN: ld.lld --emit-relocs -o %t.exe %t.o
+# RUN: llvm-bolt %t.exe -o %t.bolt --print-after-lowering --print-only=_start \
+# RUN: 2>&1 | FileCheck %s
+
+# CHECK: Binary Function "_start"
+# CHECK: bnez a0, .Ltmp[[#]]
+# CHECK: tail callee
+# CHECK: End of Function "_start"
+
+ .text
+ .globl callee
+ .type callee, @function
+callee:
+ ret
+ .size callee, .-callee
+
+ .globl _start
+ .type _start, @function
+_start:
+ beq a0, zero, callee
+ ret
+ .size _start, .-_start
+
+ .reloc 0, R_RISCV_NONE
>From c896c40312523613b92b0d514bcaa678c8119056 Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Wed, 15 Jul 2026 20:19:50 +0800
Subject: [PATCH 07/16] [BOLT][RISCV] Support split function fragments
Run long-branch relaxation for split RISC-V functions and route cross-fragment edges through AUIPC/JALR trampolines. Select a dead scratch GPR using liveness analysis, and keep a function unsplit when no ABI-safe register is available.
---
bolt/include/bolt/Core/MCPlusBuilder.h | 3 +-
bolt/lib/Passes/LongJmp.cpp | 243 +++++++++++++++---
bolt/lib/Rewrite/BinaryPassManager.cpp | 4 +
.../Target/AArch64/AArch64MCPlusBuilder.cpp | 4 +-
bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp | 78 ++++++
bolt/test/RISCV/split-functions-long-jump.s | 35 +++
bolt/test/RISCV/split-functions-no-scratch.s | 49 ++++
7 files changed, 375 insertions(+), 41 deletions(-)
create mode 100644 bolt/test/RISCV/split-functions-long-jump.s
create mode 100644 bolt/test/RISCV/split-functions-no-scratch.s
diff --git a/bolt/include/bolt/Core/MCPlusBuilder.h b/bolt/include/bolt/Core/MCPlusBuilder.h
index aae520d5afe54..0a9b283ed6c32 100644
--- a/bolt/include/bolt/Core/MCPlusBuilder.h
+++ b/bolt/include/bolt/Core/MCPlusBuilder.h
@@ -1827,7 +1827,8 @@ class MCPlusBuilder {
}
virtual void createLongJmp(InstructionListType &Seq, const MCSymbol *Target,
- MCContext *Ctx, bool IsTailCall = false) {
+ MCContext *Ctx, bool IsTailCall = false,
+ MCPhysReg ScratchReg = 0) {
llvm_unreachable("not implemented");
}
diff --git a/bolt/lib/Passes/LongJmp.cpp b/bolt/lib/Passes/LongJmp.cpp
index b771e6a8b120a..4670c28197598 100644
--- a/bolt/lib/Passes/LongJmp.cpp
+++ b/bolt/lib/Passes/LongJmp.cpp
@@ -670,14 +670,146 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
if (!BF.isSplit() && BF.estimateSize() < ShortestJumpSpan)
return;
+ DenseMap<const MCInst *, MCPhysReg> ScratchRegs;
+ if (BC.isRISCV()) {
+ const unsigned NumRegs = BC.MRI->getNumRegs();
+ DenseMap<const BinaryBasicBlock *, BitVector> LiveIns;
+ DenseMap<const BinaryBasicBlock *, BitVector> LiveOuts;
+ for (const BinaryBasicBlock &BB : BF) {
+ LiveIns.try_emplace(&BB, NumRegs, false);
+ LiveOuts.try_emplace(&BB, NumRegs, false);
+ }
+
+ BitVector ABIExitState(NumRegs, false);
+ MIB->getDefaultLiveOut(ABIExitState);
+ MIB->getCalleeSavedRegs(ABIExitState);
+
+ auto transfer = [&](const MCInst &Inst, BitVector State) {
+ if (MIB->isCFI(Inst))
+ return State;
+
+ BitVector Written(NumRegs, false);
+ BitVector Used(NumRegs, false);
+ if (MIB->isCall(Inst)) {
+ MIB->getGPRegs(Written, /*IncludeAlias=*/true);
+ BitVector Preserved(NumRegs, false);
+ MIB->getCalleeSavedRegs(Preserved);
+ Preserved.flip();
+ Written &= Preserved;
+ Used |= MIB->getRegsUsedAsParams();
+ } else {
+ MIB->getWrittenRegs(Inst, Written);
+ MIB->getUsedRegs(Inst, Used);
+ }
+ Written.flip();
+ State &= Written;
+ State |= Used;
+ return State;
+ };
+
+ bool Changed;
+ do {
+ Changed = false;
+ for (BinaryBasicBlock &BB : reverse(BF)) {
+ BitVector LiveOut(NumRegs, false);
+ if (BB.succ_size() == 0)
+ LiveOut = ABIExitState;
+ else
+ for (const BinaryBasicBlock *Succ : BB.successors())
+ LiveOut |= LiveIns[Succ];
+
+ BitVector LiveIn = LiveOut;
+ for (const MCInst &Inst : reverse(BB))
+ LiveIn = transfer(Inst, std::move(LiveIn));
+
+ if (LiveOuts[&BB] != LiveOut) {
+ LiveOuts[&BB] = std::move(LiveOut);
+ Changed = true;
+ }
+ if (LiveIns[&BB] != LiveIn) {
+ LiveIns[&BB] = std::move(LiveIn);
+ Changed = true;
+ }
+ }
+ } while (Changed);
+
+ bool CanRelax = true;
+ for (BinaryBasicBlock &BB : BF) {
+ BitVector Live = LiveOuts[&BB];
+ for (MCInst &Inst : reverse(BB)) {
+ if (!MIB->isBranch(Inst) || MIB->isIndirectBranch(Inst))
+ Live = transfer(Inst, std::move(Live));
+ else {
+ const MCSymbol *TargetSymbol = MIB->getTargetSymbol(Inst);
+ BinaryBasicBlock *TargetBB = BB.getSuccessor(TargetSymbol);
+ if (TargetBB && TargetBB->getFragmentNum() != BB.getFragmentNum()) {
+ BitVector Available = Live;
+ Available.flip();
+ BitVector GPRegs(NumRegs, false);
+ MIB->getGPRegs(GPRegs, /*IncludeAlias=*/false);
+ Available &= GPRegs;
+ MIB->removeNonScavengeableRegs(Available);
+ const int Reg = Available.find_first();
+ if (Reg == -1) {
+ CanRelax = false;
+ break;
+ }
+ ScratchRegs[&Inst] = Reg;
+ }
+ Live = transfer(Inst, std::move(Live));
+ }
+ }
+ if (!CanRelax)
+ break;
+ }
+
+ // Unlike AArch64, RISC-V has no ABI-reserved linker scratch register. If
+ // every GPR is live across a cross-fragment edge, keep this function in a
+ // single fragment rather than silently clobbering program state.
+ if (!CanRelax) {
+ BC.errs() << "BOLT-WARNING: keeping " << BF
+ << " unsplit: no dead register for a RISC-V long jump\n";
+ BinaryFunction::BasicBlockOrderType Layout(BF.getLayout().block_begin(),
+ BF.getLayout().block_end());
+ for (BinaryBasicBlock &BB : BF)
+ BB.setFragmentNum(FragmentNum::main());
+ BF.getLayout().update(Layout);
+ BF.fixBranches();
+ return;
+ }
+ }
+
auto isBranchOffsetInRange = [&](const MCInst &Inst, int64_t Offset) {
const unsigned Bits = MIB->getPCRelEncodingSize(Inst);
return isIntN(Bits, Offset);
};
+ // Output address ranges are persistent metadata used later for translating
+ // secondary entry points. Keep RISC-V's temporary relaxation offsets
+ // separate so fragment-relative offsets cannot leak into symbol rewriting.
+ DenseMap<const BinaryBasicBlock *, uint64_t> EstimatedStart;
+ DenseMap<const BinaryBasicBlock *, uint64_t> EstimatedEnd;
+ auto getEstimatedStart = [&](const BinaryBasicBlock *BB) {
+ return BC.isRISCV() ? EstimatedStart.lookup(BB)
+ : BB->getOutputStartAddress();
+ };
+ auto getEstimatedEnd = [&](const BinaryBasicBlock *BB) {
+ return BC.isRISCV() ? EstimatedEnd.lookup(BB) : BB->getOutputEndAddress();
+ };
+ auto setEstimatedRange = [&](BinaryBasicBlock *BB, uint64_t Start,
+ uint64_t End) {
+ if (BC.isRISCV()) {
+ EstimatedStart[BB] = Start;
+ EstimatedEnd[BB] = End;
+ } else {
+ BB->setOutputStartAddress(Start);
+ BB->setOutputEndAddress(End);
+ }
+ };
+
auto isBlockInRange = [&](const MCInst &Inst, uint64_t InstAddress,
const BinaryBasicBlock &BB) {
- const int64_t Offset = BB.getOutputStartAddress() - InstAddress;
+ const int64_t Offset = getEstimatedStart(&BB) - InstAddress;
return isBranchOffsetInRange(Inst, Offset);
};
@@ -688,20 +820,21 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
// Function fragments are relaxed independently.
for (FunctionFragment &FF : BF.getLayout().fragments()) {
- // Fill out code size estimation for the fragment. Use output BB address
- // ranges to store offsets from the start of the function fragment.
+ // Fill out code size estimation for the fragment.
uint64_t CodeSize = 0;
for (BinaryBasicBlock *BB : FF) {
- BB->setOutputStartAddress(CodeSize);
+ const uint64_t Start = CodeSize;
CodeSize += BB->estimateSize();
- BB->setOutputEndAddress(CodeSize);
+ setEstimatedRange(BB, Start, CodeSize);
}
// Dynamically-updated size of the fragment.
uint64_t FragmentSize = CodeSize;
- // Size of the trampoline in bytes.
- constexpr uint64_t TrampolineSize = 4;
+ // AArch64 trampolines start as one direct branch. RISC-V trampolines use
+ // AUIPC+JALR so that split fragments can be placed outside the +/-1 MiB
+ // JAL range.
+ const uint64_t TrampolineSize = BC.isRISCV() ? 8 : 4;
// Trampolines created for the fragment. DestinationBB -> TrampolineBB.
// NB: here we store only the first trampoline created for DestinationBB.
@@ -712,23 +845,31 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
// for basic blocks affected by the insertion of the trampoline.
auto addTrampolineAfter = [&](BinaryBasicBlock *BB,
BinaryBasicBlock *TargetBB, uint64_t Count,
+ MCPhysReg ScratchReg = 0,
bool UpdateOffsets = true) {
FunctionTrampolines.emplace_back(BB ? BB : FF.back(),
BF.createBasicBlock());
BinaryBasicBlock *TrampolineBB = FunctionTrampolines.back().second.get();
- MCInst Inst;
+ InstructionListType Seq;
{
auto L = BC.scopeLock();
- MIB->createUncondBranch(Inst, TargetBB->getLabel(), BC.Ctx.get());
+ if (BC.isRISCV())
+ MIB->createLongJmp(Seq, TargetBB->getLabel(), BC.Ctx.get(),
+ /*IsTailCall=*/false, ScratchReg);
+ else {
+ Seq.emplace_back();
+ MIB->createUncondBranch(Seq.back(), TargetBB->getLabel(),
+ BC.Ctx.get());
+ }
}
- TrampolineBB->addInstruction(Inst);
+ TrampolineBB->addInstructions(Seq.begin(), Seq.end());
TrampolineBB->addSuccessor(TargetBB, Count);
TrampolineBB->setExecutionCount(Count);
const uint64_t TrampolineAddress =
- BB ? BB->getOutputEndAddress() : FragmentSize;
- TrampolineBB->setOutputStartAddress(TrampolineAddress);
- TrampolineBB->setOutputEndAddress(TrampolineAddress + TrampolineSize);
+ BB ? getEstimatedEnd(BB) : FragmentSize;
+ setEstimatedRange(TrampolineBB, TrampolineAddress,
+ TrampolineAddress + TrampolineSize);
TrampolineBB->setFragmentNum(FF.getFragmentNum());
if (!FragmentTrampolines.lookup(TargetBB))
@@ -746,11 +887,10 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
// Update offsets for blocks after BB.
for (BinaryBasicBlock *IBB : FF) {
- if (IBB->getOutputStartAddress() >= TrampolineAddress) {
- IBB->setOutputStartAddress(IBB->getOutputStartAddress() +
- TrampolineSize);
- IBB->setOutputEndAddress(IBB->getOutputEndAddress() + TrampolineSize);
- }
+ const uint64_t Start = getEstimatedStart(IBB);
+ if (Start >= TrampolineAddress)
+ setEstimatedRange(IBB, Start + TrampolineSize,
+ getEstimatedEnd(IBB) + TrampolineSize);
}
// Update offsets for trampolines in this fragment that are placed after
@@ -763,11 +903,10 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
continue;
if (IBB == TrampolineBB)
continue;
- if (IBB->getOutputStartAddress() >= TrampolineAddress) {
- IBB->setOutputStartAddress(IBB->getOutputStartAddress() +
- TrampolineSize);
- IBB->setOutputEndAddress(IBB->getOutputEndAddress() + TrampolineSize);
- }
+ const uint64_t Start = getEstimatedStart(IBB);
+ if (Start >= TrampolineAddress)
+ setEstimatedRange(IBB, Start + TrampolineSize,
+ getEstimatedEnd(IBB) + TrampolineSize);
}
return TrampolineBB;
@@ -781,14 +920,22 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
continue;
const MCSymbol *TargetSymbol = MIB->getTargetSymbol(*Inst);
- BB->eraseInstruction(BB->findInstruction(Inst));
- BB->setOutputEndAddress(BB->getOutputEndAddress() - TrampolineSize);
-
BinaryBasicBlock::BinaryBranchInfo BI;
BinaryBasicBlock *TargetBB = BB->getSuccessor(TargetSymbol, BI);
+ if (!TargetBB ||
+ (BC.isRISCV() && TargetBB->getFragmentNum() == BB->getFragmentNum()))
+ continue;
+
+ const uint64_t BranchSize =
+ BC.isRISCV() ? BC.computeCodeSize(Inst, Inst + 1) : 4;
+ const MCPhysReg ScratchReg = ScratchRegs.lookup(Inst);
+ BB->eraseInstruction(BB->findInstruction(Inst));
+ setEstimatedRange(BB, getEstimatedStart(BB),
+ getEstimatedEnd(BB) - BranchSize);
BinaryBasicBlock *TrampolineBB =
- addTrampolineAfter(BB, TargetBB, BI.Count, /*UpdateOffsets*/ false);
+ addTrampolineAfter(BB, TargetBB, BI.Count, ScratchReg,
+ /*UpdateOffsets=*/false);
BB->replaceSuccessor(TargetBB, TrampolineBB, BI.Count);
}
@@ -806,7 +953,8 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
// Try to reuse an existing trampoline without introducing any new code.
BinaryBasicBlock *TrampolineBB = FragmentTrampolines.lookup(TargetBB);
- if (TrampolineBB && isBlockInRange(Inst, InstAddress, *TrampolineBB)) {
+ if (!BC.isRISCV() && TrampolineBB &&
+ isBlockInRange(Inst, InstAddress, *TrampolineBB)) {
BB->replaceSuccessor(TargetBB, TrampolineBB, Count);
TrampolineBB->setExecutionCount(TrampolineBB->getExecutionCount() +
Count);
@@ -819,9 +967,10 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
// of the fragment that is within the branch reach. Note that such
// trampoline may change address later and become unreachable in which
// case we will need further relaxation.
+ const MCPhysReg ScratchReg = ScratchRegs.lookup(&Inst);
const int64_t OffsetToEnd = FragmentSize - InstAddress;
if (Count == 0 && isBranchOffsetInRange(Inst, OffsetToEnd)) {
- TrampolineBB = addTrampolineAfter(nullptr, TargetBB, Count);
+ TrampolineBB = addTrampolineAfter(nullptr, TargetBB, Count, ScratchReg);
BB->replaceSuccessor(TargetBB, TrampolineBB, Count);
auto L = BC.scopeLock();
MIB->replaceBranchTarget(Inst, TrampolineBB->getLabel(), BC.Ctx.get());
@@ -840,12 +989,12 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
if (ShouldReverseBranch && !IsReversibleBranch) {
const uint64_t NextCount = BB->getBranchInfo(*NextBB).Count;
BinaryBasicBlock *FallThrough =
- addTrampolineAfter(BB, NextBB, NextCount);
+ addTrampolineAfter(BB, NextBB, NextCount, ScratchReg);
BB->replaceSuccessor(NextBB, FallThrough, NextCount);
}
// Create a trampoline basic block for the taken target of the branch.
- TrampolineBB = addTrampolineAfter(BB, TargetBB, Count);
+ TrampolineBB = addTrampolineAfter(BB, TargetBB, Count, ScratchReg);
if (ShouldReverseBranch && IsReversibleBranch) {
BB->swapConditionalSuccessors();
@@ -865,25 +1014,32 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
++NumIterations;
for (auto BBI = FF.begin(); BBI != FF.end(); ++BBI) {
BinaryBasicBlock *BB = *BBI;
- uint64_t NextInstOffset = BB->getOutputStartAddress();
+ uint64_t NextInstOffset = getEstimatedStart(BB);
for (MCInst &Inst : *BB) {
const size_t InstAddress = NextInstOffset;
if (!MIB->isPseudo(Inst))
- NextInstOffset += 4;
+ NextInstOffset +=
+ BC.isRISCV() ? BC.computeCodeSize(&Inst, &Inst + 1) : 4;
if (!mayNeedStub(BF.getBinaryContext(), Inst))
continue;
+ const MCSymbol *TargetSymbol = MIB->getTargetSymbol(Inst);
+ BinaryBasicBlock *TargetBB = BB->getSuccessor(TargetSymbol);
+ if (!TargetBB)
+ continue;
+
const size_t BitsAvailable = MIB->getPCRelEncodingSize(Inst);
- // Span of +/-128MB.
- if (BitsAvailable == LongestJumpBits)
+ // AArch64 compact code model keeps fragments within the range of B.
+ if (!BC.isRISCV() && BitsAvailable == LongestJumpBits)
continue;
- const MCSymbol *TargetSymbol = MIB->getTargetSymbol(Inst);
- BinaryBasicBlock *TargetBB = BB->getSuccessor(TargetSymbol);
- assert(TargetBB &&
- "Basic block target expected for conditional branch.");
+ // Existing intra-fragment RISC-V branches are handled by JITLink's
+ // normal branch relaxation. This pass is responsible for edges that
+ // become unrepresentable specifically because of function split.
+ if (BC.isRISCV() && TargetBB->getFragmentNum() == FF.getFragmentNum())
+ continue;
// Check if the relaxation is needed.
if (TargetBB->getFragmentNum() == FF.getFragmentNum() &&
@@ -930,6 +1086,15 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
}
Error LongJmpPass::runOnFunctions(BinaryContext &BC) {
+ if (BC.isRISCV()) {
+ BC.outs() << "BOLT-INFO: relaxing RISC-V cross-fragment branches\n";
+ for (BinaryFunction *BF : BC.getOutputBinaryFunctions()) {
+ if (!BC.shouldEmit(*BF) || !BF->isSimple() || !BF->isSplit())
+ continue;
+ relaxLocalBranches(*BF);
+ }
+ return Error::success();
+ }
assert((opts::CompactCodeModel ||
opts::SplitStrategy != opts::SplitFunctionsStrategy::CDSplit) &&
diff --git a/bolt/lib/Rewrite/BinaryPassManager.cpp b/bolt/lib/Rewrite/BinaryPassManager.cpp
index 6e3022c491a73..63cb550b3f3a5 100644
--- a/bolt/lib/Rewrite/BinaryPassManager.cpp
+++ b/bolt/lib/Rewrite/BinaryPassManager.cpp
@@ -536,12 +536,16 @@ Error BinaryFunctionPassManager::runAllPasses(BinaryContext &BC) {
if (BC.isAArch64()) {
Manager.registerPass(
std::make_unique<AArch64RelaxationPass>(PrintAArch64Relaxation));
+ }
+ if (BC.isAArch64() || BC.isRISCV()) {
// Tighten branches according to offset differences between branch and
// targets. No extra instructions after this pass, otherwise we may have
// relocations out of range and crash during linking.
Manager.registerPass(std::make_unique<LongJmpPass>(PrintLongJmp));
+ }
+ if (BC.isAArch64()) {
Manager.registerPass(
std::make_unique<PointerAuthCFIFixup>(PrintPAuthCFIFixup));
}
diff --git a/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp b/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
index 26004c94acdd0..1ff06a66a1d57 100644
--- a/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
+++ b/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
@@ -2712,7 +2712,9 @@ class AArch64MCPlusBuilder : public MCPlusBuilder {
}
void createLongJmp(InstructionListType &Seq, const MCSymbol *Target,
- MCContext *Ctx, bool IsTailCall) override {
+ MCContext *Ctx, bool IsTailCall,
+ MCPhysReg ScratchReg) override {
+ (void)ScratchReg;
// ip0 (r16) is reserved to the linker (refer to 5.3.1.1 of "Procedure Call
// Standard for the ARM 64-bit Architecture (AArch64)".
// The sequence of instructions we create here is the following:
diff --git a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
index e8b2700b6f010..34f7f30e15c01 100644
--- a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
+++ b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
@@ -40,6 +40,13 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
public:
using MCPlusBuilder::MCPlusBuilder;
+ BitVector getRegsUsedAsParams() const override {
+ BitVector Regs(RegInfo->getNumRegs(), false);
+ for (MCPhysReg Reg = RISCV::X10; Reg <= RISCV::X17; ++Reg)
+ Regs |= getAliases(Reg);
+ return Regs;
+ }
+
bool equals(const MCSpecifierExpr &A, const MCSpecifierExpr &B,
CompFuncTy Comp) const override {
const auto &RISCVExprA = cast<MCSpecifierExpr>(A);
@@ -67,6 +74,31 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
Regs |= getAliases(RISCV::X27);
}
+ void getDefaultLiveOut(BitVector &Regs) const override {
+ Regs |= getAliases(RISCV::X10);
+ Regs |= getAliases(RISCV::X11);
+ }
+
+ void getGPRegs(BitVector &Regs, bool IncludeAlias = true) const override {
+ for (MCPhysReg Reg = RISCV::X1; Reg <= RISCV::X31; ++Reg) {
+ if (IncludeAlias)
+ Regs |= getAliases(Reg);
+ else
+ Regs.set(Reg);
+ }
+ }
+
+ void removeNonScavengeableRegs(BitVector &Regs) const override {
+ BitVector ExclusionMask(RegInfo->getNumRegs(), false);
+ ExclusionMask |= getAliases(RISCV::X1); // return address
+ ExclusionMask |= getAliases(RISCV::X2); // stack pointer
+ ExclusionMask |= getAliases(RISCV::X3); // global pointer
+ ExclusionMask |= getAliases(RISCV::X4); // thread pointer
+ ExclusionMask |= getAliases(RISCV::X8); // frame pointer
+ ExclusionMask.flip();
+ Regs &= ExclusionMask;
+ }
+
bool shouldRecordCodeRelocation(uint32_t RelType) const override {
switch (RelType) {
case ELF::R_RISCV_JAL:
@@ -170,6 +202,29 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
replaceBranchTarget(Inst, TBB, Ctx);
}
+ int getPCRelEncodingSize(const MCInst &Inst) const override {
+ switch (Inst.getOpcode()) {
+ default:
+ llvm_unreachable("Failed to get RISC-V PC-relative encoding size");
+ case RISCV::C_BEQZ:
+ case RISCV::C_BNEZ:
+ return 9;
+ case RISCV::C_J:
+ return 12;
+ case RISCV::BEQ:
+ case RISCV::BNE:
+ case RISCV::BLT:
+ case RISCV::BGE:
+ case RISCV::BLTU:
+ case RISCV::BGEU:
+ return 13;
+ case RISCV::JAL:
+ return 21;
+ }
+ }
+
+ int getUncondBranchEncodingSize() const override { return 21; }
+
void replaceBranchTarget(MCInst &Inst, const MCSymbol *TBB,
MCContext *Ctx) const override {
assert((isCall(Inst) || isBranch(Inst)) && !isIndirectBranch(Inst) &&
@@ -613,6 +668,29 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
Seq.swap(Insts);
}
+ void createLongJmp(InstructionListType &Seq, const MCSymbol *Target,
+ MCContext *Ctx, bool IsTailCall,
+ MCPhysReg ScratchReg) override {
+ assert(ScratchReg && "RISC-V long jump requires a scratch register");
+ MCSymbol *AuipcLabel = Ctx->createNamedTempSymbol("long_jmp");
+
+ MCInst Inst = MCInstBuilder(RISCV::AUIPC).addReg(ScratchReg).addImm(0);
+ setOperandToSymbolRef(Inst, /*OpNum=*/1, Target, /*Addend=*/0, Ctx,
+ ELF::R_RISCV_PCREL_HI20);
+ setInstLabel(Inst, AuipcLabel);
+ Seq.emplace_back(std::move(Inst));
+
+ Inst = MCInstBuilder(RISCV::JALR)
+ .addReg(RISCV::X0)
+ .addReg(ScratchReg)
+ .addImm(0);
+ setOperandToSymbolRef(Inst, /*OpNum=*/2, AuipcLabel, /*Addend=*/0, Ctx,
+ ELF::R_RISCV_PCREL_LO12_I);
+ if (IsTailCall)
+ setTailCall(Inst);
+ Seq.emplace_back(std::move(Inst));
+ }
+
InstructionListType createGetter(MCContext *Ctx, const char *name) const {
InstructionListType Insts(4);
MCSymbol *Locs = Ctx->getOrCreateSymbol(name);
diff --git a/bolt/test/RISCV/split-functions-long-jump.s b/bolt/test/RISCV/split-functions-long-jump.s
new file mode 100644
index 0000000000000..eb1282d0928c9
--- /dev/null
+++ b/bolt/test/RISCV/split-functions-long-jump.s
@@ -0,0 +1,35 @@
+## Check that a branch crossing a split-function fragment is redirected through
+## a local trampoline. The trampoline uses AUIPC+JALR instead of JAL so the
+## cold fragment can be placed outside the +/-1 MiB JAL range.
+
+# RUN: llvm-mc -triple riscv64 -mattr=+c -filetype=obj -o %t.o %s
+# RUN: ld.lld --emit-relocs -e _start -o %t.exe %t.o
+# RUN: llvm-bolt %t.exe -o %t.bolt -split-functions \
+# RUN: -split-strategy=random2 -bolt-seed=1
+# RUN: llvm-objdump -d %t.bolt | FileCheck %s
+# RUN: llvm-readelf -s %t.bolt | FileCheck --check-prefix=SYMBOLS %s
+
+# CHECK: Disassembly of section .text:
+# CHECK-LABEL: <_start>:
+# CHECK: auipc [[REG:[a-z0-9]+]],
+# CHECK-NEXT: {{(jalr zero,|jr)}} {{.*}}([[REG]])
+# CHECK: Disassembly of section .text.cold:
+# CHECK-LABEL: <secondary>:
+# SYMBOLS: FUNC GLOBAL DEFAULT {{[0-9]+}} secondary
+
+ .text
+ .globl _start
+ .type _start, @function
+_start:
+ beq a0, zero, .Lcold
+ .globl secondary
+ .type secondary, @function
+secondary:
+ li a0, 1
+ ret
+.Lcold:
+ li a0, 2
+ ret
+ .size _start, .-_start
+
+ .reloc 0, R_RISCV_NONE
diff --git a/bolt/test/RISCV/split-functions-no-scratch.s b/bolt/test/RISCV/split-functions-no-scratch.s
new file mode 100644
index 0000000000000..b3841268a318c
--- /dev/null
+++ b/bolt/test/RISCV/split-functions-no-scratch.s
@@ -0,0 +1,49 @@
+## RISC-V has no ABI-reserved linker scratch register. If every usable GPR is
+## live across a split edge, check that BOLT keeps that function unsplit rather
+## than clobbering state in an AUIPC+JALR trampoline.
+
+# RUN: llvm-mc -triple riscv64 -filetype=obj -o %t.o %s
+# RUN: ld.lld --emit-relocs -e _start -o %t.exe %t.o
+# RUN: llvm-bolt %t.exe -o %t.bolt -split-functions \
+# RUN: -split-strategy=random2 -bolt-seed=1 2>&1 | FileCheck %s
+# RUN: llvm-readelf -S %t.bolt | FileCheck --check-prefix=SECTIONS %s
+
+# CHECK: BOLT-WARNING: keeping _start unsplit: no dead register for a RISC-V long jump
+# SECTIONS-NOT: .text.cold
+
+ .text
+ .globl _start
+ .type _start, @function
+_start:
+ beq a0, zero, .Lcold
+ ret
+.Lcold:
+ add a0, a0, t0
+ add a0, a0, t1
+ add a0, a0, t2
+ add a0, a0, s1
+ add a0, a0, a1
+ add a0, a0, a2
+ add a0, a0, a3
+ add a0, a0, a4
+ add a0, a0, a5
+ add a0, a0, a6
+ add a0, a0, a7
+ add a0, a0, s2
+ add a0, a0, s3
+ add a0, a0, s4
+ add a0, a0, s5
+ add a0, a0, s6
+ add a0, a0, s7
+ add a0, a0, s8
+ add a0, a0, s9
+ add a0, a0, s10
+ add a0, a0, s11
+ add a0, a0, t3
+ add a0, a0, t4
+ add a0, a0, t5
+ add a0, a0, t6
+ ret
+ .size _start, .-_start
+
+ .reloc 0, R_RISCV_NONE
>From ec1727d842cccaa4d17ce07187dc7cf842d7d9b2 Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Tue, 21 Jul 2026 17:21:28 +0800
Subject: [PATCH 08/16] [BOLT][RISCV] Account for explicit call operands in
liveness
---
bolt/lib/Passes/LongJmp.cpp | 11 ++---
.../split-functions-indirect-call-scratch.s | 43 +++++++++++++++++++
bolt/test/RISCV/split-functions-long-jump.s | 6 +++
3 files changed, 55 insertions(+), 5 deletions(-)
create mode 100644 bolt/test/RISCV/split-functions-indirect-call-scratch.s
diff --git a/bolt/lib/Passes/LongJmp.cpp b/bolt/lib/Passes/LongJmp.cpp
index 4670c28197598..1d60bfc2fc3cd 100644
--- a/bolt/lib/Passes/LongJmp.cpp
+++ b/bolt/lib/Passes/LongJmp.cpp
@@ -690,16 +690,17 @@ void LongJmpPass::relaxLocalBranches(BinaryFunction &BF) {
BitVector Written(NumRegs, false);
BitVector Used(NumRegs, false);
+ MIB->getWrittenRegs(Inst, Written);
+ MIB->getUsedRegs(Inst, Used);
if (MIB->isCall(Inst)) {
- MIB->getGPRegs(Written, /*IncludeAlias=*/true);
+ BitVector CallClobbered(NumRegs, false);
+ MIB->getGPRegs(CallClobbered, /*IncludeAlias=*/true);
BitVector Preserved(NumRegs, false);
MIB->getCalleeSavedRegs(Preserved);
Preserved.flip();
- Written &= Preserved;
+ CallClobbered &= Preserved;
+ Written |= CallClobbered;
Used |= MIB->getRegsUsedAsParams();
- } else {
- MIB->getWrittenRegs(Inst, Written);
- MIB->getUsedRegs(Inst, Used);
}
Written.flip();
State &= Written;
diff --git a/bolt/test/RISCV/split-functions-indirect-call-scratch.s b/bolt/test/RISCV/split-functions-indirect-call-scratch.s
new file mode 100644
index 0000000000000..a87027571ef53
--- /dev/null
+++ b/bolt/test/RISCV/split-functions-indirect-call-scratch.s
@@ -0,0 +1,43 @@
+## Check that register scavenging for a cross-fragment long jump accounts for
+## the explicit target register of an indirect call in the destination block.
+## The trampoline must not clobber t0 before the cold block calls through it.
+
+# RUN: llvm-mc -triple riscv64 -mattr=+c -filetype=obj -o %t.o %s
+# RUN: ld.lld --emit-relocs -e _start -o %t.exe %t.o
+# RUN: llvm-bolt %t.exe -o %t.bolt -split-functions \
+# RUN: -split-strategy=random2 -bolt-seed=1
+# RUN: llvm-objdump -d --no-show-raw-insn %t.bolt | FileCheck %s
+
+# CHECK-LABEL: <_start>:
+# CHECK: auipc t0,
+# CHECK-NEXT: addi t0, t0,
+# CHECK: auipc t1,
+# CHECK-NEXT: {{(jalr zero,|jr)}} {{.*}}(t1)
+# CHECK: auipc t1,
+# CHECK-NEXT: {{(jalr zero,|jr)}} {{.*}}(t1)
+# CHECK-LABEL: <_start.cold.0>:
+# CHECK: jalr t0
+
+ .text
+ .globl _start
+ .type _start, @function
+_start:
+1:
+ auipc t0, %pcrel_hi(callee)
+ addi t0, t0, %pcrel_lo(1b)
+ beq a0, zero, .Lcold
+ li a0, 1
+ ret
+.Lcold:
+ jalr ra, t0, 0
+ ret
+ .size _start, .-_start
+
+ .globl callee
+ .type callee, @function
+callee:
+ li a0, 0
+ ret
+ .size callee, .-callee
+
+ .reloc 0, R_RISCV_NONE
diff --git a/bolt/test/RISCV/split-functions-long-jump.s b/bolt/test/RISCV/split-functions-long-jump.s
index eb1282d0928c9..15a3ea8314e5e 100644
--- a/bolt/test/RISCV/split-functions-long-jump.s
+++ b/bolt/test/RISCV/split-functions-long-jump.s
@@ -8,6 +8,12 @@
# RUN: -split-strategy=random2 -bolt-seed=1
# RUN: llvm-objdump -d %t.bolt | FileCheck %s
# RUN: llvm-readelf -s %t.bolt | FileCheck --check-prefix=SYMBOLS %s
+# RUN: llvm-mc -triple riscv32 -mattr=+c -filetype=obj -o %t.32.o %s
+# RUN: ld.lld --emit-relocs -e _start -o %t.32.exe %t.32.o
+# RUN: llvm-bolt %t.32.exe -o %t.32.bolt -split-functions \
+# RUN: -split-strategy=random2 -bolt-seed=1
+# RUN: llvm-objdump -d %t.32.bolt | FileCheck %s
+# RUN: llvm-readelf -s %t.32.bolt | FileCheck --check-prefix=SYMBOLS %s
# CHECK: Disassembly of section .text:
# CHECK-LABEL: <_start>:
>From cff0239f71752cea9f6a089c8eaa0aa86827d580 Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Tue, 21 Jul 2026 17:27:34 +0800
Subject: [PATCH 09/16] [BOLT][RISCV] Respect the RVE register set in long
jumps
---
bolt/lib/Rewrite/RewriteInstance.cpp | 6 ++++
bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp | 9 +++--
.../RISCV/split-functions-no-scratch-rve.s | 33 +++++++++++++++++++
3 files changed, 46 insertions(+), 2 deletions(-)
create mode 100644 bolt/test/RISCV/split-functions-no-scratch-rve.s
diff --git a/bolt/lib/Rewrite/RewriteInstance.cpp b/bolt/lib/Rewrite/RewriteInstance.cpp
index bf42f8663c512..fde853e25ae08 100644
--- a/bolt/lib/Rewrite/RewriteInstance.cpp
+++ b/bolt/lib/Rewrite/RewriteInstance.cpp
@@ -443,6 +443,12 @@ RewriteInstance::RewriteInstance(ELFObjectFileBase *File, const int Argc,
return;
} else {
Features.reset(new SubtargetFeatures(*FeaturesOrErr));
+ // EF_RISCV_RVE selects the E ABI even when the input has no
+ // .riscv.attributes architecture string. ObjectFile::getFeatures()
+ // currently derives RVC from e_flags but not RVE, so preserve this ABI
+ // constraint explicitly for register analysis and code generation.
+ if (File->getPlatformFlags() & ELF::EF_RISCV_RVE)
+ Features->AddFeature("e");
}
}
diff --git a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
index 34f7f30e15c01..7953d3a4fd7a9 100644
--- a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
+++ b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
@@ -30,6 +30,7 @@ namespace {
class RISCVMCPlusBuilder : public MCPlusBuilder {
bool isRV64() const { return STI->hasFeature(RISCV::Feature64Bit); }
+ bool isRVE() const { return STI->hasFeature(RISCV::FeatureStdExtE); }
unsigned regSize() const { return isRV64() ? 8 : 4; }
unsigned loadOpc() const { return isRV64() ? RISCV::LD : RISCV::LW; }
unsigned storeOpc() const { return isRV64() ? RISCV::SD : RISCV::SW; }
@@ -42,7 +43,8 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
BitVector getRegsUsedAsParams() const override {
BitVector Regs(RegInfo->getNumRegs(), false);
- for (MCPhysReg Reg = RISCV::X10; Reg <= RISCV::X17; ++Reg)
+ const MCPhysReg LastArgReg = isRVE() ? RISCV::X15 : RISCV::X17;
+ for (MCPhysReg Reg = RISCV::X10; Reg <= LastArgReg; ++Reg)
Regs |= getAliases(Reg);
return Regs;
}
@@ -62,6 +64,8 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
Regs |= getAliases(RISCV::X2);
Regs |= getAliases(RISCV::X8);
Regs |= getAliases(RISCV::X9);
+ if (isRVE())
+ return;
Regs |= getAliases(RISCV::X18);
Regs |= getAliases(RISCV::X19);
Regs |= getAliases(RISCV::X20);
@@ -80,7 +84,8 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
}
void getGPRegs(BitVector &Regs, bool IncludeAlias = true) const override {
- for (MCPhysReg Reg = RISCV::X1; Reg <= RISCV::X31; ++Reg) {
+ const MCPhysReg LastGPR = isRVE() ? RISCV::X15 : RISCV::X31;
+ for (MCPhysReg Reg = RISCV::X1; Reg <= LastGPR; ++Reg) {
if (IncludeAlias)
Regs |= getAliases(Reg);
else
diff --git a/bolt/test/RISCV/split-functions-no-scratch-rve.s b/bolt/test/RISCV/split-functions-no-scratch-rve.s
new file mode 100644
index 0000000000000..3cf385916ad53
--- /dev/null
+++ b/bolt/test/RISCV/split-functions-no-scratch-rve.s
@@ -0,0 +1,33 @@
+## RVE only has x0-x15. If every usable RVE GPR is live across a split edge,
+## check that BOLT does not select an unavailable x16-x31 register for the
+## AUIPC+JALR trampoline and keeps the function unsplit instead.
+
+# RUN: llvm-mc -triple riscv32 -mattr=+e -filetype=obj -o %t.o %s
+# RUN: ld.lld --emit-relocs -e _start -o %t.exe %t.o
+# RUN: llvm-bolt %t.exe -o %t.bolt -split-functions \
+# RUN: -split-strategy=random2 -bolt-seed=1 2>&1 | FileCheck %s
+# RUN: llvm-readelf -S %t.bolt | FileCheck --check-prefix=SECTIONS %s
+
+# CHECK: BOLT-WARNING: keeping _start unsplit: no dead register for a RISC-V long jump
+# SECTIONS-NOT: .text.cold
+
+ .text
+ .globl _start
+ .type _start, @function
+_start:
+ beq a0, zero, .Lcold
+ ret
+.Lcold:
+ add a0, a0, t0
+ add a0, a0, t1
+ add a0, a0, t2
+ add a0, a0, s1
+ add a0, a0, a1
+ add a0, a0, a2
+ add a0, a0, a3
+ add a0, a0, a4
+ add a0, a0, a5
+ ret
+ .size _start, .-_start
+
+ .reloc 0, R_RISCV_NONE
>From a8e699aa02949ed2b8d313d7e248446175fdc28d Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Tue, 21 Jul 2026 17:30:53 +0800
Subject: [PATCH 10/16] [BOLT][RISCV] Do not suffix mapping symbols for split
fragments
---
bolt/lib/Rewrite/RewriteInstance.cpp | 15 +++++++++++----
bolt/test/RISCV/split-functions-long-jump.s | 7 +++++++
2 files changed, 18 insertions(+), 4 deletions(-)
diff --git a/bolt/lib/Rewrite/RewriteInstance.cpp b/bolt/lib/Rewrite/RewriteInstance.cpp
index fde853e25ae08..ad2ab39f50313 100644
--- a/bolt/lib/Rewrite/RewriteInstance.cpp
+++ b/bolt/lib/Rewrite/RewriteInstance.cpp
@@ -5666,6 +5666,17 @@ void RewriteInstance::updateELFSymbolTable(
Expected<StringRef> SymbolName = Symbol.getName(StringSection);
assert(SymbolName && "cannot get symbol name");
+ // Mapping symbols can share the exact address of a function entry, but
+ // they are code/data metadata rather than function aliases. Let the
+ // marker-specific path below handle them; otherwise addExtraSymbols()
+ // creates invalid split names such as "$xrv64i...cold.0".
+ auto IsMarkerSymbol = [&]() {
+ return BC->getMarkerType(Symbol.getType(), Symbol.st_size,
+ *SymbolName) != MarkerSymType::NONE;
+ };
+ if (Function && IsMarkerSymbol())
+ Function = nullptr;
+
auto updateSymbolValue = [&](const StringRef Name,
std::optional<uint64_t> Value = std::nullopt) {
NewSymbol.st_value = Value ? *Value : getNewValueForSymbol(Name);
@@ -5717,10 +5728,6 @@ void RewriteInstance::updateELFSymbolTable(
// update their addresses to reflect the output layout.
// Skip AArch64/RISC-V marker symbols ($d, $x) inside functions —
// BOLT generates its own via addExtraSymbols.
- auto IsMarkerSymbol = [&]() {
- return BC->getMarkerType(Symbol.getType(), Symbol.st_size,
- *SymbolName) != MarkerSymType::NONE;
- };
const bool IsLocalLabel = Symbol.getType() == ELF::STT_NOTYPE &&
Symbol.getBinding() == ELF::STB_LOCAL &&
Symbol.st_size == 0 && !IsMarkerSymbol();
diff --git a/bolt/test/RISCV/split-functions-long-jump.s b/bolt/test/RISCV/split-functions-long-jump.s
index 15a3ea8314e5e..9cf576d2f8038 100644
--- a/bolt/test/RISCV/split-functions-long-jump.s
+++ b/bolt/test/RISCV/split-functions-long-jump.s
@@ -14,6 +14,12 @@
# RUN: -split-strategy=random2 -bolt-seed=1
# RUN: llvm-objdump -d %t.32.bolt | FileCheck %s
# RUN: llvm-readelf -s %t.32.bolt | FileCheck --check-prefix=SYMBOLS %s
+# RUN: llvm-mc -triple riscv32 -mattr=+e -filetype=obj -o %t.e.o %s
+# RUN: ld.lld --emit-relocs -e _start -o %t.e.exe %t.e.o
+# RUN: llvm-bolt %t.e.exe -o %t.e.bolt -split-functions \
+# RUN: -split-strategy=random2 -bolt-seed=1
+# RUN: llvm-objdump -d %t.e.bolt 2>&1 | FileCheck %s
+# RUN: llvm-readelf -s %t.e.bolt | FileCheck --check-prefix=SYMBOLS %s
# CHECK: Disassembly of section .text:
# CHECK-LABEL: <_start>:
@@ -22,6 +28,7 @@
# CHECK: Disassembly of section .text.cold:
# CHECK-LABEL: <secondary>:
# SYMBOLS: FUNC GLOBAL DEFAULT {{[0-9]+}} secondary
+# SYMBOLS-NOT: $x{{.*}}.cold
.text
.globl _start
>From 21cd33c82784d4dfb8010e522564197ab13c8d8b Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Wed, 29 Jul 2026 20:17:18 +0800
Subject: [PATCH 11/16] [BOLT][RISCV] Implement indirect PLT calls
---
bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp | 53 ++++++++++++++++++++
bolt/test/RISCV/plt-call.test | 44 ++++++++++++++++
2 files changed, 97 insertions(+)
create mode 100644 bolt/test/RISCV/plt-call.test
diff --git a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
index 7953d3a4fd7a9..bbc37cc674534 100644
--- a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
+++ b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
@@ -341,6 +341,59 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
return createCall(RISCV::PseudoTAIL, Inst, Target, Ctx);
}
+ InstructionListType createIndirectPLTCall(MCInst &&DirectCall,
+ const MCSymbol *TargetLocation,
+ MCContext *Ctx) override {
+ const bool IsTailCall = isTailCall(DirectCall);
+ assert(((DirectCall.getOpcode() == RISCV::PseudoCALL && !IsTailCall) ||
+ (DirectCall.getOpcode() == RISCV::PseudoTAIL && IsTailCall)) &&
+ "RISC-V direct (tail) call instruction expected");
+
+ // Load the resolved function address directly from its GOT slot:
+ //
+ // auipc t3, %pcrel_hi(TargetLocation)
+ // l[dw] t3, %pcrel_lo(.Lpcrel_hi)(t3)
+ // jalr ra, t3, 0
+ //
+ // A tail call uses zero instead of ra as the JALR destination.
+ InstructionListType Code;
+ // Use t3 (x28), the scratch register used by linker-generated RISC-V
+ // PLT/IPLT entries. It is caller-saved, is not an argument register, and
+ // the original call through the PLT already clobbers it.
+ const MCPhysReg PLTScratchReg = RISCV::X28;
+ MCSymbol *AUIPCLabel = Ctx->createNamedTempSymbol("pcrel_hi");
+
+ MCInst InstAUIPC =
+ MCInstBuilder(RISCV::AUIPC).addReg(PLTScratchReg).addImm(0);
+ // TargetLocation is already registered at the existing GOT slot, so use a
+ // direct PC-relative relocation to that slot instead of R_RISCV_GOT_HI20,
+ // which is used when starting from the referenced function symbol.
+ setOperandToSymbolRef(InstAUIPC, /*OpNum=*/1, TargetLocation,
+ /*Addend=*/0, Ctx, ELF::R_RISCV_PCREL_HI20);
+ setInstLabel(InstAUIPC, AUIPCLabel);
+ Code.emplace_back(std::move(InstAUIPC));
+
+ // Load the call target from the GOT slot using LD on RV64 or LW on RV32.
+ MCInst InstLoad = MCInstBuilder(loadOpc())
+ .addReg(PLTScratchReg)
+ .addReg(PLTScratchReg)
+ .addImm(0);
+ // Pair the I-type LD/LW immediate with the label on AUIPC. RISC-V
+ // R_RISCV_PCREL_LO12_I relocations name the corresponding HI20 location.
+ setOperandToSymbolRef(InstLoad, /*OpNum=*/2, AUIPCLabel,
+ /*Addend=*/0, Ctx, ELF::R_RISCV_PCREL_LO12_I);
+ Code.emplace_back(std::move(InstLoad));
+
+ MCInst InstCall = MCInstBuilder(RISCV::JALR)
+ .addReg(IsTailCall ? RISCV::X0 : RISCV::X1)
+ .addReg(PLTScratchReg)
+ .addImm(0);
+ moveAnnotations(std::move(DirectCall), InstCall);
+ Code.emplace_back(std::move(InstCall));
+
+ return Code;
+ }
+
bool analyzeBranch(InstructionIterator Begin, InstructionIterator End,
const MCSymbol *&TBB, const MCSymbol *&FBB,
MCInst *&CondBranch,
diff --git a/bolt/test/RISCV/plt-call.test b/bolt/test/RISCV/plt-call.test
new file mode 100644
index 0000000000000..76f850e892e59
--- /dev/null
+++ b/bolt/test/RISCV/plt-call.test
@@ -0,0 +1,44 @@
+// Verify that PLTCall optimization works on RISC-V.
+
+// RUN: split-file %s %t.dir
+// RUN: llvm-mc -triple=riscv64 -filetype=obj -o %t.dir/main.o %t.dir/main.s
+// RUN: llvm-mc -triple=riscv64 -filetype=obj -o %t.dir/lib.o %t.dir/lib.s
+// RUN: ld.lld -shared -soname libplt.so -o %t.dir/libplt.so %t.dir/lib.o
+// RUN: ld.lld --no-pie --emit-relocs -z now \
+// RUN: -dynamic-linker /lib/ld.so.1 %t.dir/main.o %t.dir/libplt.so \
+// RUN: -o %t.exe
+// RUN: llvm-bolt %t.exe -o %t.bolt --plt=all --print-plt \
+// RUN: --print-only=_start | FileCheck %s
+
+// Call to foo.
+// CHECK: auipc t3, %pcrel_hi(foo at GOT)
+// CHECK-NEXT: ld t3, %pcrel_lo({{.*}})(t3)
+// CHECK-NEXT: jalr t3 # PLTCall: 1
+
+// Tail call to bar.
+// CHECK: auipc t3, %pcrel_hi(bar at GOT)
+// CHECK-NEXT: ld t3, %pcrel_lo({{.*}})(t3)
+// CHECK-NEXT: jr t3 # TAILCALL # PLTCall: 1
+
+//--- main.s
+ .text
+ .globl _start
+ .type _start, @function
+_start:
+ call foo
+ tail bar
+ .size _start, .-_start
+
+//--- lib.s
+ .text
+ .globl foo
+ .type foo, @function
+foo:
+ ret
+ .size foo, .-foo
+
+ .globl bar
+ .type bar, @function
+bar:
+ ret
+ .size bar, .-bar
>From 107c1b61734ebe849f516aa2940327f9d33af16e Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Mon, 3 Aug 2026 15:58:27 +0800
Subject: [PATCH 12/16] [BOLT] Track jump-table entry size and signedness
---
bolt/include/bolt/Core/BinaryContext.h | 16 +++--
bolt/include/bolt/Core/JumpTable.h | 6 +-
bolt/include/bolt/Core/MCPlusBuilder.h | 6 +-
bolt/include/bolt/Core/Relocation.h | 3 +
bolt/lib/Core/BinaryContext.cpp | 66 +++++++++++++------
bolt/lib/Core/BinaryFunction.cpp | 44 +++++++++++--
bolt/lib/Core/JumpTable.cpp | 15 +++--
bolt/lib/Core/Relocation.cpp | 14 ++++
bolt/lib/Passes/IndirectCallPromotion.cpp | 6 +-
.../Target/AArch64/AArch64MCPlusBuilder.cpp | 17 +++--
bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp | 5 +-
bolt/lib/Target/X86/X86MCPlusBuilder.cpp | 15 +++--
12 files changed, 154 insertions(+), 59 deletions(-)
diff --git a/bolt/include/bolt/Core/BinaryContext.h b/bolt/include/bolt/Core/BinaryContext.h
index 240e5a75d1de5..273a5d7a76396 100644
--- a/bolt/include/bolt/Core/BinaryContext.h
+++ b/bolt/include/bolt/Core/BinaryContext.h
@@ -602,7 +602,9 @@ class BinaryContext {
/// element of the pair.
const MCSymbol *getOrCreateJumpTable(BinaryFunction &Function,
uint64_t Address,
- JumpTable::JumpTableType Type);
+ JumpTable::JumpTableType Type,
+ uint64_t EntrySize = 0,
+ bool EntriesAreSigned = false);
/// Analyze a possible jump table of type \p Type at a given \p Address.
/// \p BF is a function referencing the jump table.
@@ -614,12 +616,12 @@ class BinaryContext {
///
/// Optionally, populate \p Address from jump table entries. The entries
/// could be partially populated if the jump table detection fails.
- bool analyzeJumpTable(const uint64_t Address,
- const JumpTable::JumpTableType Type,
- const BinaryFunction &BF,
- const uint64_t NextJTAddress = 0,
- JumpTable::AddressesType *EntriesAsAddress = nullptr,
- bool *HasEntryInFragment = nullptr) const;
+ bool
+ analyzeJumpTable(const uint64_t Address, const JumpTable::JumpTableType Type,
+ const BinaryFunction &BF, const uint64_t NextJTAddress = 0,
+ JumpTable::AddressesType *EntriesAsAddress = nullptr,
+ bool *HasEntryInFragment = nullptr, uint64_t EntrySize = 0,
+ bool EntriesAreSigned = false) const;
/// After jump table locations are established, this function will populate
/// their EntriesAsAddress based on memory contents.
diff --git a/bolt/include/bolt/Core/JumpTable.h b/bolt/include/bolt/Core/JumpTable.h
index 52b9ccee1f7e1..0f44252dd47e3 100644
--- a/bolt/include/bolt/Core/JumpTable.h
+++ b/bolt/include/bolt/Core/JumpTable.h
@@ -66,6 +66,9 @@ class JumpTable : public BinaryData {
/// The type of this jump table.
JumpTableType Type;
+ /// Whether entries are sign-extended when loaded by the dispatch sequence.
+ bool EntriesAreSigned;
+
/// Whether this jump table has entries pointing to multiple functions.
bool IsSplit{false};
@@ -95,7 +98,8 @@ class JumpTable : public BinaryData {
private:
/// Constructor should only be called by a BinaryContext.
JumpTable(MCSymbol &Symbol, uint64_t Address, size_t EntrySize,
- JumpTableType Type, LabelMapType &&Labels, BinarySection &Section);
+ bool EntriesAreSigned, JumpTableType Type, LabelMapType &&Labels,
+ BinarySection &Section);
public:
/// Return the size of the jump table.
diff --git a/bolt/include/bolt/Core/MCPlusBuilder.h b/bolt/include/bolt/Core/MCPlusBuilder.h
index 0a9b283ed6c32..b92d57724ed79 100644
--- a/bolt/include/bolt/Core/MCPlusBuilder.h
+++ b/bolt/include/bolt/Core/MCPlusBuilder.h
@@ -1773,11 +1773,15 @@ class MCPlusBuilder {
/// will be set to the different components of the branch. \p MemLocInstr
/// is the instruction that loads up the indirect function pointer. It may
/// or may not be same as \p Instruction.
+ /// \p EntrySize and \p EntrySigned describe the jump-table entry loaded by
+ /// the matched instruction sequence. A zero entry size requests the target's
+ /// default for the detected jump-table type.
virtual IndirectBranchType analyzeIndirectBranch(
MCInst &Instruction, InstructionIterator Begin, InstructionIterator End,
const unsigned PtrSize, MCInst *&MemLocInstr, unsigned &BaseRegNum,
unsigned &IndexRegNum, int64_t &DispValue, const MCExpr *&DispExpr,
- MCInst *&PCRelBaseOut, MCInst *&FixedEntryLoadInst) const {
+ uint64_t &EntrySize, bool &EntrySigned, MCInst *&PCRelBaseOut,
+ MCInst *&FixedEntryLoadInst) const {
llvm_unreachable("not implemented");
return IndirectBranchType::UNKNOWN;
}
diff --git a/bolt/include/bolt/Core/Relocation.h b/bolt/include/bolt/Core/Relocation.h
index 8ee3e2587cb9c..d3f232055f8c8 100644
--- a/bolt/include/bolt/Core/Relocation.h
+++ b/bolt/include/bolt/Core/Relocation.h
@@ -150,6 +150,9 @@ class Relocation {
/// Return code for a PC-relative 8-byte relocation
static uint32_t getPC64();
+ /// Return code for an ABS 4-byte relocation
+ static uint32_t getAbs32();
+
/// Return code for a ABS 8-byte relocation
static uint32_t getAbs64();
diff --git a/bolt/lib/Core/BinaryContext.cpp b/bolt/lib/Core/BinaryContext.cpp
index 5063e957a2be8..39acd40670454 100644
--- a/bolt/lib/Core/BinaryContext.cpp
+++ b/bolt/lib/Core/BinaryContext.cpp
@@ -619,12 +619,11 @@ MemoryContentsType BinaryContext::analyzeMemoryAt(uint64_t Address,
return MemoryContentsType::UNKNOWN;
}
-bool BinaryContext::analyzeJumpTable(const uint64_t Address,
- const JumpTable::JumpTableType Type,
- const BinaryFunction &BF,
- const uint64_t NextJTAddress,
- JumpTable::AddressesType *EntriesAsAddress,
- bool *HasEntryInFragment) const {
+bool BinaryContext::analyzeJumpTable(
+ const uint64_t Address, const JumpTable::JumpTableType Type,
+ const BinaryFunction &BF, const uint64_t NextJTAddress,
+ JumpTable::AddressesType *EntriesAsAddress, bool *HasEntryInFragment,
+ uint64_t EntrySize, bool EntriesAreSigned) const {
// Target address of __builtin_unreachable.
const uint64_t UnreachableAddress = BF.getAddress() + BF.getSize();
@@ -685,7 +684,11 @@ bool BinaryContext::analyzeJumpTable(const uint64_t Address,
Address, BF.getPrintName(),
Type == JTT::JTT_PIC ? "PIC" : "Normal");
});
- const uint64_t EntrySize = getJumpTableEntrySize(Type);
+ if (!EntrySize)
+ EntrySize = getJumpTableEntrySize(Type);
+ EntriesAreSigned |= Type == JumpTable::JTT_PIC;
+ if (UpperBound < Address || UpperBound - Address < EntrySize)
+ return false;
for (uint64_t EntryAddress = Address; EntryAddress <= UpperBound - EntrySize;
EntryAddress += EntrySize) {
LLVM_DEBUG(dbgs() << " * Checking 0x" << Twine::utohexstr(EntryAddress)
@@ -706,10 +709,22 @@ bool BinaryContext::analyzeJumpTable(const uint64_t Address,
}
}
- const uint64_t Value =
- (Type == JumpTable::JTT_PIC)
- ? Address + *getSignedValueAtAddress(EntryAddress, EntrySize)
- : *getPointerAtAddress(EntryAddress);
+ uint64_t Value;
+ if (EntriesAreSigned) {
+ ErrorOr<int64_t> SignedValue =
+ getSignedValueAtAddress(EntryAddress, EntrySize);
+ if (!SignedValue)
+ break;
+ Value = static_cast<uint64_t>(*SignedValue);
+ if (Type == JumpTable::JTT_PIC)
+ Value += Address;
+ } else {
+ ErrorOr<uint64_t> UnsignedValue =
+ getUnsignedValueAtAddress(EntryAddress, EntrySize);
+ if (!UnsignedValue)
+ break;
+ Value = *UnsignedValue;
+ }
// __builtin_unreachable() case.
if (Value == UnreachableAddress) {
@@ -781,7 +796,8 @@ void BinaryContext::populateJumpTables() {
const bool Success =
analyzeJumpTable(JT->getAddress(), JT->Type, *(JT->Parents[0]),
- NextJTAddress, &JT->EntriesAsAddress, &JT->IsSplit);
+ NextJTAddress, &JT->EntriesAsAddress, &JT->IsSplit,
+ JT->EntrySize, JT->EntriesAreSigned);
if (!Success) {
// Re-analysis here is stricter than during disassembly (the referenced
// function is now disassembled), so it may fail on a table we accepted
@@ -825,7 +841,7 @@ void BinaryContext::populateJumpTables() {
for (uint64_t Address = JT->getAddress();
Address < JT->getAddress() + JT->getSize();
Address += JT->EntrySize) {
- DataPCRelocations.erase(DataPCRelocations.find(Address));
+ DataPCRelocations.erase(Address);
}
}
@@ -915,11 +931,18 @@ BinaryFunction *BinaryContext::createBinaryFunction(
const MCSymbol *
BinaryContext::getOrCreateJumpTable(BinaryFunction &Function, uint64_t Address,
- JumpTable::JumpTableType Type) {
+ JumpTable::JumpTableType Type,
+ uint64_t EntrySize, bool EntriesAreSigned) {
+ EntriesAreSigned |= Type == JumpTable::JTT_PIC;
+
// Two fragments of same function access same jump table
if (JumpTable *JT = getJumpTableContainingAddress(Address)) {
assert(JT->Type == Type && "jump table types have to match");
assert(Address == JT->getAddress() && "unexpected non-empty jump table");
+ assert((!EntrySize || JT->EntrySize == EntrySize) &&
+ "jump table entry sizes have to match");
+ assert((!EntrySize || JT->EntriesAreSigned == EntriesAreSigned) &&
+ "jump table entry signedness has to match");
if (llvm::is_contained(JT->Parents, &Function))
return JT->getFirstLabel();
@@ -951,7 +974,8 @@ BinaryContext::getOrCreateJumpTable(BinaryFunction &Function, uint64_t Address,
JTLabel = Object->getSymbol();
}
- const uint64_t EntrySize = getJumpTableEntrySize(Type);
+ if (!EntrySize)
+ EntrySize = getJumpTableEntrySize(Type);
if (!JTLabel) {
const std::string JumpTableName = generateJumpTableName(Function, Address);
JTLabel = registerNameAtAddress(JumpTableName, Address, 0, EntrySize);
@@ -960,8 +984,8 @@ BinaryContext::getOrCreateJumpTable(BinaryFunction &Function, uint64_t Address,
LLVM_DEBUG(dbgs() << "BOLT-DEBUG: creating jump table " << JTLabel->getName()
<< " in function " << Function << '\n');
- JumpTable *JT = new JumpTable(*JTLabel, Address, EntrySize, Type,
- JumpTable::LabelMapType{{0, JTLabel}},
+ JumpTable *JT = new JumpTable(*JTLabel, Address, EntrySize, EntriesAreSigned,
+ Type, JumpTable::LabelMapType{{0, JTLabel}},
*getSectionForAddress(Address));
JT->Parents.push_back(&Function);
if (opts::Verbosity > 2)
@@ -989,10 +1013,10 @@ BinaryContext::duplicateJumpTable(BinaryFunction &Function, JumpTable *JT,
assert(Found && "Label not found");
(void)Found;
MCSymbol *NewLabel = Ctx->createNamedTempSymbol("duplicatedJT");
- JumpTable *NewJT =
- new JumpTable(*NewLabel, JT->getAddress(), JT->EntrySize, JT->Type,
- JumpTable::LabelMapType{{Offset, NewLabel}},
- *getSectionForAddress(JT->getAddress()));
+ JumpTable *NewJT = new JumpTable(*NewLabel, JT->getAddress(), JT->EntrySize,
+ JT->EntriesAreSigned, JT->Type,
+ JumpTable::LabelMapType{{Offset, NewLabel}},
+ *getSectionForAddress(JT->getAddress()));
NewJT->Parents = JT->Parents;
NewJT->Entries = JT->Entries;
NewJT->Counts = JT->Counts;
diff --git a/bolt/lib/Core/BinaryFunction.cpp b/bolt/lib/Core/BinaryFunction.cpp
index 53787b68b7695..6899f746cddd8 100644
--- a/bolt/lib/Core/BinaryFunction.cpp
+++ b/bolt/lib/Core/BinaryFunction.cpp
@@ -824,6 +824,8 @@ BinaryFunction::processIndirectBranch(MCInst &Instruction, unsigned Size,
unsigned BaseRegNum, IndexRegNum;
int64_t DispValue;
const MCExpr *DispExpr;
+ uint64_t EntrySize;
+ bool EntrySigned;
// In AArch, identify the instruction adding the PC-relative offset to
// jump table entries to correctly decode it.
@@ -847,12 +849,13 @@ BinaryFunction::processIndirectBranch(MCInst &Instruction, unsigned Size,
IndirectBranchType BranchType = BC.MIB->analyzeIndirectBranch(
Instruction, Begin, Instructions.end(), PtrSize, MemLocInstr, BaseRegNum,
- IndexRegNum, DispValue, DispExpr, PCRelBaseInstr, FixedEntryLoadInstr);
+ IndexRegNum, DispValue, DispExpr, EntrySize, EntrySigned, PCRelBaseInstr,
+ FixedEntryLoadInstr);
if (BranchType == IndirectBranchType::UNKNOWN && !MemLocInstr)
return BranchType;
- if (MemLocInstr != &Instruction)
+ if (MemLocInstr && MemLocInstr != &Instruction)
IndexRegNum = BC.MIB->getNoRegister();
if (BC.isAArch64()) {
@@ -910,7 +913,8 @@ BinaryFunction::processIndirectBranch(MCInst &Instruction, unsigned Size,
ArrayStart = static_cast<uint64_t>(DispValue);
}
- if (BaseRegNum == BC.MRI->getProgramCounter())
+ if (BaseRegNum != BC.MIB->getNoRegister() &&
+ BaseRegNum == BC.MRI->getProgramCounter())
ArrayStart += getAddress() + Offset + Size;
if (FixedEntryLoadInstr) {
@@ -991,6 +995,16 @@ BinaryFunction::processIndirectBranch(MCInst &Instruction, unsigned Size,
// Check if there's already a jump table registered at this address.
MemoryContentsType MemType;
if (JumpTable *JT = BC.getJumpTableContainingAddress(ArrayStart)) {
+ if (BC.isRISCV()) {
+ const bool IsRelated = llvm::all_of(JT->Parents, [&](BinaryFunction *BF) {
+ return BC.areRelatedFragments(this, BF);
+ });
+ if (!IsRelated)
+ return IndirectBranchType::UNKNOWN;
+ }
+
+ EntrySize = JT->EntrySize;
+ EntrySigned = JT->EntriesAreSigned;
switch (JT->Type) {
case JumpTable::JTT_NORMAL:
MemType = MemoryContentsType::POSSIBLE_JUMP_TABLE;
@@ -999,6 +1013,18 @@ BinaryFunction::processIndirectBranch(MCInst &Instruction, unsigned Size,
MemType = MemoryContentsType::POSSIBLE_PIC_JUMP_TABLE;
break;
}
+ } else if (EntrySize) {
+ const JumpTable::JumpTableType ExpectedType =
+ BranchType == IndirectBranchType::POSSIBLE_PIC_JUMP_TABLE
+ ? JumpTable::JTT_PIC
+ : JumpTable::JTT_NORMAL;
+ const bool IsJumpTable =
+ BC.analyzeJumpTable(ArrayStart, ExpectedType, *this, 0, nullptr,
+ nullptr, EntrySize, EntrySigned);
+ MemType = IsJumpTable ? ExpectedType == JumpTable::JTT_PIC
+ ? MemoryContentsType::POSSIBLE_PIC_JUMP_TABLE
+ : MemoryContentsType::POSSIBLE_JUMP_TABLE
+ : MemoryContentsType::UNKNOWN;
} else {
MemType = BC.analyzeMemoryAt(ArrayStart, *this);
}
@@ -1021,8 +1047,10 @@ BinaryFunction::processIndirectBranch(MCInst &Instruction, unsigned Size,
}
// Convert the instruction into jump table branch.
- const MCSymbol *JTLabel = BC.getOrCreateJumpTable(*this, ArrayStart, JTType);
- BC.MIB->replaceMemOperandDisp(*MemLocInstr, JTLabel, BC.Ctx.get());
+ const MCSymbol *JTLabel = BC.getOrCreateJumpTable(*this, ArrayStart, JTType,
+ EntrySize, EntrySigned);
+ if (MemLocInstr)
+ BC.MIB->replaceMemOperandDisp(*MemLocInstr, JTLabel, BC.Ctx.get());
BC.MIB->setJumpTable(Instruction, ArrayStart, IndexRegNum);
JTSites.emplace_back(Offset, ArrayStart);
@@ -2239,12 +2267,14 @@ bool BinaryFunction::postProcessIndirectBranches(
unsigned BaseRegNum, IndexRegNum;
int64_t DispValue;
const MCExpr *DispExpr;
+ uint64_t EntrySize;
+ bool EntrySigned;
MCInst *PCRelBaseInstr;
MCInst *FixedEntryLoadInstr;
IndirectBranchType Type = BC.MIB->analyzeIndirectBranch(
Instr, BB.begin(), II, PtrSize, MemLocInstr, BaseRegNum,
- IndexRegNum, DispValue, DispExpr, PCRelBaseInstr,
- FixedEntryLoadInstr);
+ IndexRegNum, DispValue, DispExpr, EntrySize, EntrySigned,
+ PCRelBaseInstr, FixedEntryLoadInstr);
if (Type != IndirectBranchType::UNKNOWN || MemLocInstr != nullptr)
continue;
diff --git a/bolt/lib/Core/JumpTable.cpp b/bolt/lib/Core/JumpTable.cpp
index 6f588d2b95fd6..cd83b847ad25d 100644
--- a/bolt/lib/Core/JumpTable.cpp
+++ b/bolt/lib/Core/JumpTable.cpp
@@ -13,6 +13,7 @@
#include "bolt/Core/JumpTable.h"
#include "bolt/Core/BinaryFunction.h"
#include "bolt/Core/BinarySection.h"
+#include "bolt/Core/Relocation.h"
#include "llvm/Support/CommandLine.h"
#define DEBUG_TYPE "bolt"
@@ -28,10 +29,11 @@ extern cl::opt<unsigned> Verbosity;
} // namespace opts
bolt::JumpTable::JumpTable(MCSymbol &Symbol, uint64_t Address, size_t EntrySize,
- JumpTableType Type, LabelMapType &&Labels,
- BinarySection &Section)
+ bool EntriesAreSigned, JumpTableType Type,
+ LabelMapType &&Labels, BinarySection &Section)
: BinaryData(Symbol, Address, 0, EntrySize, Section), EntrySize(EntrySize),
- OutputEntrySize(EntrySize), Type(Type), Labels(Labels) {}
+ OutputEntrySize(EntrySize), Type(Type),
+ EntriesAreSigned(EntriesAreSigned), Labels(Labels) {}
std::pair<size_t, size_t>
bolt::JumpTable::getEntriesForAddress(const uint64_t Addr) const {
@@ -84,8 +86,11 @@ void bolt::JumpTable::updateOriginal() {
const uint64_t BaseOffset = getAddress() - getSection().getAddress();
uint64_t EntryOffset = BaseOffset;
for (MCSymbol *Entry : Entries) {
- const uint32_t RelType =
- Type == JTT_NORMAL ? ELF::R_X86_64_64 : ELF::R_X86_64_PC32;
+ assert((Type == JTT_PIC || EntrySize == 4 || EntrySize == 8) &&
+ "unsupported absolute jump-table entry size");
+ const uint32_t RelType = Type == JTT_PIC ? Relocation::getPC32()
+ : EntrySize == 4 ? Relocation::getAbs32()
+ : Relocation::getAbs64();
const uint64_t RelAddend =
Type == JTT_NORMAL ? 0 : EntryOffset - BaseOffset;
// Replace existing relocation with the new one to allow any modifications
diff --git a/bolt/lib/Core/Relocation.cpp b/bolt/lib/Core/Relocation.cpp
index c1916f5a0967b..262bc0e19f534 100644
--- a/bolt/lib/Core/Relocation.cpp
+++ b/bolt/lib/Core/Relocation.cpp
@@ -1000,6 +1000,20 @@ uint32_t Relocation::getPC64() {
}
}
+uint32_t Relocation::getAbs32() {
+ switch (Arch) {
+ default:
+ llvm_unreachable("Unsupported architecture");
+ case Triple::aarch64:
+ return ELF::R_AARCH64_ABS32;
+ case Triple::riscv64:
+ case Triple::riscv32:
+ return ELF::R_RISCV_32;
+ case Triple::x86_64:
+ return ELF::R_X86_64_32;
+ }
+}
+
uint32_t Relocation::getType(const object::RelocationRef &Rel) {
uint64_t RelType = Rel.getType();
assert(isUInt<32>(RelType) && "BOLT relocation types are 32 bits");
diff --git a/bolt/lib/Passes/IndirectCallPromotion.cpp b/bolt/lib/Passes/IndirectCallPromotion.cpp
index 39ae4cda145c4..0e891733182ce 100644
--- a/bolt/lib/Passes/IndirectCallPromotion.cpp
+++ b/bolt/lib/Passes/IndirectCallPromotion.cpp
@@ -387,11 +387,13 @@ IndirectCallPromotion::maybeGetHotJumpTableTargets(BinaryBasicBlock &BB,
unsigned BaseReg, IndexReg;
int64_t DispValue;
const MCExpr *DispExpr;
+ uint64_t EntrySize;
+ bool EntrySigned;
MutableArrayRef<MCInst> Insts(&BB.front(), &CallInst);
const IndirectBranchType Type = BC.MIB->analyzeIndirectBranch(
CallInst, Insts.begin(), Insts.end(), BC.AsmInfo->getCodePointerSize(),
- MemLocInstr, BaseReg, IndexReg, DispValue, DispExpr, PCRelBaseOut,
- FixedEntryLoadInstr);
+ MemLocInstr, BaseReg, IndexReg, DispValue, DispExpr, EntrySize,
+ EntrySigned, PCRelBaseOut, FixedEntryLoadInstr);
assert(MemLocInstr && "There should always be a load for jump tables");
if (!MemLocInstr)
diff --git a/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp b/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
index 1ff06a66a1d57..136e679443858 100644
--- a/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
+++ b/bolt/lib/Target/AArch64/AArch64MCPlusBuilder.cpp
@@ -1782,18 +1782,19 @@ class AArch64MCPlusBuilder : public MCPlusBuilder {
return Uses;
}
- IndirectBranchType
- analyzeIndirectBranch(MCInst &Instruction, InstructionIterator Begin,
- InstructionIterator End, const unsigned PtrSize,
- MCInst *&MemLocInstrOut, unsigned &BaseRegNumOut,
- unsigned &IndexRegNumOut, int64_t &DispValueOut,
- const MCExpr *&DispExprOut, MCInst *&PCRelBaseOut,
- MCInst *&FixedEntryLoadInstr) const override {
+ IndirectBranchType analyzeIndirectBranch(
+ MCInst &Instruction, InstructionIterator Begin, InstructionIterator End,
+ const unsigned PtrSize, MCInst *&MemLocInstrOut, unsigned &BaseRegNumOut,
+ unsigned &IndexRegNumOut, int64_t &DispValueOut,
+ const MCExpr *&DispExprOut, uint64_t &EntrySizeOut, bool &EntrySignedOut,
+ MCInst *&PCRelBaseOut, MCInst *&FixedEntryLoadInstr) const override {
MemLocInstrOut = nullptr;
BaseRegNumOut = AArch64::NoRegister;
IndexRegNumOut = AArch64::NoRegister;
DispValueOut = 0;
DispExprOut = nullptr;
+ EntrySizeOut = 0;
+ EntrySignedOut = false;
FixedEntryLoadInstr = nullptr;
// An instruction referencing memory used by jump instruction (directly or
@@ -1815,6 +1816,8 @@ class AArch64MCPlusBuilder : public MCPlusBuilder {
MemLocInstrOut = MemLocInstr;
DispValueOut = DispValue;
DispExprOut = DispExpr;
+ EntrySizeOut = ScaleValue;
+ EntrySignedOut = true;
PCRelBaseOut = PCRelBase;
return IndirectBranchType::POSSIBLE_PIC_JUMP_TABLE;
}
diff --git a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
index bbc37cc674534..9fc220e689449 100644
--- a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
+++ b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
@@ -248,12 +248,15 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
MCInst &Instruction, InstructionIterator Begin, InstructionIterator End,
const unsigned PtrSize, MCInst *&MemLocInstr, unsigned &BaseRegNum,
unsigned &IndexRegNum, int64_t &DispValue, const MCExpr *&DispExpr,
- MCInst *&PCRelBaseOut, MCInst *&FixedEntryLoadInst) const override {
+ uint64_t &EntrySize, bool &EntrySigned, MCInst *&PCRelBaseOut,
+ MCInst *&FixedEntryLoadInst) const override {
MemLocInstr = nullptr;
BaseRegNum = 0;
IndexRegNum = 0;
DispValue = 0;
DispExpr = nullptr;
+ EntrySize = 0;
+ EntrySigned = false;
PCRelBaseOut = nullptr;
FixedEntryLoadInst = nullptr;
diff --git a/bolt/lib/Target/X86/X86MCPlusBuilder.cpp b/bolt/lib/Target/X86/X86MCPlusBuilder.cpp
index 9fd3cdb909ce6..5b2cdaef9bf3e 100644
--- a/bolt/lib/Target/X86/X86MCPlusBuilder.cpp
+++ b/bolt/lib/Target/X86/X86MCPlusBuilder.cpp
@@ -2008,13 +2008,12 @@ class X86MCPlusBuilder : public MCPlusBuilder {
SecondInstr, nullptr);
}
- IndirectBranchType
- analyzeIndirectBranch(MCInst &Instruction, InstructionIterator Begin,
- InstructionIterator End, const unsigned PtrSize,
- MCInst *&MemLocInstrOut, unsigned &BaseRegNumOut,
- unsigned &IndexRegNumOut, int64_t &DispValueOut,
- const MCExpr *&DispExprOut, MCInst *&PCRelBaseOut,
- MCInst *&FixedEntryLoadInst) const override {
+ IndirectBranchType analyzeIndirectBranch(
+ MCInst &Instruction, InstructionIterator Begin, InstructionIterator End,
+ const unsigned PtrSize, MCInst *&MemLocInstrOut, unsigned &BaseRegNumOut,
+ unsigned &IndexRegNumOut, int64_t &DispValueOut,
+ const MCExpr *&DispExprOut, uint64_t &EntrySizeOut, bool &EntrySignedOut,
+ MCInst *&PCRelBaseOut, MCInst *&FixedEntryLoadInst) const override {
// Try to find a (base) memory location from where the address for
// the indirect branch is loaded. For X86-64 the memory will be specified
// in the following format:
@@ -2041,6 +2040,8 @@ class X86MCPlusBuilder : public MCPlusBuilder {
IndexRegNumOut = X86::NoRegister;
DispValueOut = 0;
DispExprOut = nullptr;
+ EntrySizeOut = 0;
+ EntrySignedOut = false;
FixedEntryLoadInst = nullptr;
std::reverse_iterator<InstructionIterator> II(End);
>From 2d917f5881079bfb892adde36bc3eac1d409907a Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Mon, 3 Aug 2026 15:59:27 +0800
Subject: [PATCH 13/16] [BOLT][RISCV] Recognize jump-table dispatch sequences
---
bolt/lib/Core/BinaryContext.cpp | 8 +-
bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp | 352 ++++++++++++++++++-
bolt/test/RISCV/jump-table-shared-anchor.s | 92 +++++
bolt/test/RISCV/jump-table-use-def.s | 120 +++++++
4 files changed, 554 insertions(+), 18 deletions(-)
create mode 100644 bolt/test/RISCV/jump-table-shared-anchor.s
create mode 100644 bolt/test/RISCV/jump-table-use-def.s
diff --git a/bolt/lib/Core/BinaryContext.cpp b/bolt/lib/Core/BinaryContext.cpp
index 39acd40670454..20bb6049b0fb5 100644
--- a/bolt/lib/Core/BinaryContext.cpp
+++ b/bolt/lib/Core/BinaryContext.cpp
@@ -17,6 +17,7 @@
#include "bolt/Utils/Utils.h"
#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/Twine.h"
+#include "llvm/BinaryFormat/ELF.h"
#include "llvm/DebugInfo/DWARF/DWARFCompileUnit.h"
#include "llvm/DebugInfo/DWARF/DWARFFormValue.h"
#include "llvm/DebugInfo/DWARF/DWARFUnit.h"
@@ -695,13 +696,16 @@ bool BinaryContext::analyzeJumpTable(
<< " -> ");
// Check if there's a proper relocation against the jump table entry.
if (HasRelocations) {
+ const Relocation *Rel = getRelocationAt(EntryAddress);
+ const bool HasRISCVLabelDifference =
+ isRISCV() && Rel && Rel->Type == ELF::R_RISCV_ADD32;
if (Type == JumpTable::JTT_PIC &&
- !DataPCRelocations.count(EntryAddress)) {
+ !DataPCRelocations.count(EntryAddress) && !HasRISCVLabelDifference) {
LLVM_DEBUG(
dbgs() << "FAIL: JTT_PIC table, no relocation for this address\n");
break;
}
- if (Type == JumpTable::JTT_NORMAL && !getRelocationAt(EntryAddress)) {
+ if (Type == JumpTable::JTT_NORMAL && !Rel) {
LLVM_DEBUG(
dbgs()
<< "FAIL: JTT_NORMAL table, no relocation for this address\n");
diff --git a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
index 9fc220e689449..5163aab88de5e 100644
--- a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
+++ b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
@@ -10,10 +10,13 @@
//
//===----------------------------------------------------------------------===//
-#include "MCTargetDesc/RISCVMCAsmInfo.h"
#include "MCTargetDesc/RISCVFixupKinds.h"
+#include "MCTargetDesc/RISCVMCAsmInfo.h"
#include "MCTargetDesc/RISCVMCTargetDesc.h"
#include "bolt/Core/MCPlusBuilder.h"
+#include "llvm/ADT/DenseMap.h"
+#include "llvm/ADT/SmallPtrSet.h"
+#include "llvm/ADT/SmallVector.h"
#include "llvm/BinaryFormat/ELF.h"
#include "llvm/MC/MCContext.h"
#include "llvm/MC/MCInst.h"
@@ -29,6 +32,20 @@ using namespace bolt;
namespace {
class RISCVMCPlusBuilder : public MCPlusBuilder {
+ using LocalUDChain = DenseMap<const MCInst *, SmallVector<MCInst *, 4>>;
+
+ struct JumpTableLoad {
+ const MCInst *Inst{nullptr};
+ uint64_t EntrySize{0};
+ bool EntrySigned{false};
+ int64_t Offset{0};
+ };
+
+ struct ScaledAddress {
+ const MCInst *BaseDef{nullptr};
+ MCPhysReg IndexReg{MCRegister::NoRegister};
+ };
+
bool isRV64() const { return STI->hasFeature(RISCV::Feature64Bit); }
bool isRVE() const { return STI->hasFeature(RISCV::FeatureStdExtE); }
unsigned regSize() const { return isRV64() ? 8 : 4; }
@@ -38,6 +55,226 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
return isRV64() ? RISCV::AMOADD_D : RISCV::AMOADD_W;
}
+ LocalUDChain computeLocalUDChain(const MCInst *CurInstr,
+ InstructionIterator Begin,
+ InstructionIterator End) const {
+ DenseMap<int, MCInst *> RegAliasTable;
+ LocalUDChain Uses;
+
+ auto addInstrOperands = [&](const MCInst &Instr) {
+ for (const MCOperand &Operand : MCPlus::primeOperands(Instr)) {
+ if (!Operand.isReg())
+ continue;
+ Uses[&Instr].push_back(RegAliasTable[Operand.getReg()]);
+ }
+ };
+
+ bool TerminatorSeen = false;
+ for (auto II = Begin; II != End; ++II) {
+ MCInst &Instr = *II;
+ if (isPseudo(Instr) || isNoop(Instr))
+ continue;
+ if (TerminatorSeen) {
+ RegAliasTable.clear();
+ Uses.clear();
+ }
+
+ addInstrOperands(Instr);
+
+ BitVector Regs(RegInfo->getNumRegs(), false);
+ getWrittenRegs(Instr, Regs);
+ for (int Idx : Regs.set_bits())
+ RegAliasTable[Idx] = &Instr;
+
+ TerminatorSeen = isTerminator(Instr);
+ }
+
+ if (CurInstr)
+ addInstrOperands(*CurInstr);
+
+ return Uses;
+ }
+
+ const MCInst *getOperandDef(const MCInst &Inst, unsigned OperandIndex,
+ const LocalUDChain &UDChain) const {
+ if (OperandIndex >= MCPlus::getNumPrimeOperands(Inst) ||
+ !Inst.getOperand(OperandIndex).isReg())
+ return nullptr;
+
+ const auto UsesIt = UDChain.find(&Inst);
+ if (UsesIt == UDChain.end())
+ return nullptr;
+
+ unsigned RegOperandIndex = 0;
+ for (unsigned Index = 0; Index < OperandIndex; ++Index)
+ RegOperandIndex += Inst.getOperand(Index).isReg();
+
+ if (RegOperandIndex >= UsesIt->second.size())
+ return nullptr;
+ return UsesIt->second[RegOperandIndex];
+ }
+
+ const MCInst *followCopies(const MCInst *Def,
+ const LocalUDChain &UDChain) const {
+ SmallPtrSet<const MCInst *, 4> Visited;
+ while (Def && Visited.insert(Def).second) {
+ unsigned SourceOperand = 0;
+ switch (Def->getOpcode()) {
+ default:
+ return Def;
+ case RISCV::ADDI:
+ case RISCV::ORI:
+ if (!Def->getOperand(2).isImm() || Def->getOperand(2).getImm() != 0)
+ return Def;
+ SourceOperand = 1;
+ break;
+ case RISCV::ADD:
+ case RISCV::OR:
+ if (Def->getOperand(1).getReg() == RISCV::X0)
+ SourceOperand = 2;
+ else if (Def->getOperand(2).getReg() == RISCV::X0)
+ SourceOperand = 1;
+ else
+ return Def;
+ break;
+ case RISCV::C_MV:
+ SourceOperand = 1;
+ break;
+ }
+ Def = getOperandDef(*Def, SourceOperand, UDChain);
+ }
+ return Def;
+ }
+
+ static const MCExpr *stripSpecifier(const MCExpr *Expr) {
+ while (const auto *Specifier = dyn_cast_or_null<MCSpecifierExpr>(Expr))
+ Expr = Specifier->getSubExpr();
+ return Expr;
+ }
+
+ const MCExpr *matchJumpTableBase(const MCInst *Def,
+ const LocalUDChain &UDChain) const {
+ Def = followCopies(Def, UDChain);
+ if (!Def)
+ return nullptr;
+
+ if (Def->getOpcode() == RISCV::ADDI || Def->getOpcode() == RISCV::C_ADDI) {
+ Def = followCopies(getOperandDef(*Def, 1, UDChain), UDChain);
+ if (!Def)
+ return nullptr;
+ }
+
+ switch (Def->getOpcode()) {
+ default:
+ return nullptr;
+ case RISCV::LUI:
+ case RISCV::AUIPC:
+ case RISCV::C_LUI:
+ break;
+ }
+
+ if (!Def->getOperand(1).isExpr())
+ return nullptr;
+ const MCExpr *Expr = stripSpecifier(Def->getOperand(1).getExpr());
+ return getTargetSymbolInfo(Expr).first ? Expr : nullptr;
+ }
+
+ bool matchJumpTableLoad(const MCInst *Def, const LocalUDChain &UDChain,
+ JumpTableLoad &Load) const {
+ Def = followCopies(Def, UDChain);
+ if (!Def)
+ return false;
+
+ switch (Def->getOpcode()) {
+ default:
+ return false;
+ case RISCV::LW:
+ case RISCV::C_LW:
+ Load.EntrySize = 4;
+ Load.EntrySigned = isRV64();
+ break;
+ case RISCV::LWU:
+ Load.EntrySize = 4;
+ Load.EntrySigned = false;
+ break;
+ case RISCV::LD:
+ case RISCV::C_LD:
+ // GCC uses full-width label-address arrays as a family of sub-tables
+ // relative to one shared anchor. BOLT cannot move those safely until it
+ // can retarget every LUI/AUIPC + ADDI reference to an interior label.
+ return false;
+ }
+
+ if (!Def->getOperand(2).isImm())
+ return false;
+ Load.Inst = Def;
+ Load.Offset = Def->getOperand(2).getImm();
+ // A non-zero displacement can select an embedded table relative to a
+ // larger anchor object. Moving that table requires retargeting the whole
+ // LUI/AUIPC + ADDI pair to a new interior label, which is not represented
+ // by MemLocInstr today. Reject it instead of moving the wrong sub-table.
+ return Load.Offset == 0;
+ }
+
+ static unsigned getSHXADDScale(unsigned Opcode) {
+ switch (Opcode) {
+ default:
+ return 0;
+ case RISCV::SH1ADD:
+ case RISCV::SH1ADD_UW:
+ return 2;
+ case RISCV::SH2ADD:
+ case RISCV::SH2ADD_UW:
+ return 4;
+ case RISCV::SH3ADD:
+ case RISCV::SH3ADD_UW:
+ return 8;
+ }
+ }
+
+ bool matchScaledAddress(const MCInst *Def, uint64_t EntrySize,
+ const LocalUDChain &UDChain,
+ ScaledAddress &Address) const {
+ Def = followCopies(Def, UDChain);
+ if (!Def)
+ return false;
+
+ if (getSHXADDScale(Def->getOpcode()) == EntrySize) {
+ Address.IndexReg = Def->getOperand(1).getReg();
+ Address.BaseDef = followCopies(getOperandDef(*Def, 2, UDChain), UDChain);
+ return Address.BaseDef != nullptr;
+ }
+
+ if (Def->getOpcode() != RISCV::ADD && Def->getOpcode() != RISCV::C_ADD)
+ return false;
+
+ for (unsigned ShiftOperand : {1U, 2U}) {
+ const unsigned BaseOperand = ShiftOperand == 1 ? 2 : 1;
+ const MCInst *Shift =
+ followCopies(getOperandDef(*Def, ShiftOperand, UDChain), UDChain);
+ if (!Shift)
+ continue;
+ if (Shift->getOpcode() != RISCV::SLLI &&
+ Shift->getOpcode() != RISCV::SLLI_UW &&
+ Shift->getOpcode() != RISCV::C_SLLI)
+ continue;
+ if (!Shift->getOperand(2).isImm() ||
+ (1ULL << Shift->getOperand(2).getImm()) != EntrySize)
+ continue;
+
+ Address.IndexReg = Shift->getOperand(1).getReg();
+ Address.BaseDef =
+ followCopies(getOperandDef(*Def, BaseOperand, UDChain), UDChain);
+ if (Address.BaseDef)
+ return true;
+ }
+ return false;
+ }
+
+ bool areSameJumpTable(const MCExpr *LHS, const MCExpr *RHS) const {
+ return getTargetSymbolInfo(LHS) == getTargetSymbolInfo(RHS);
+ }
+
public:
using MCPlusBuilder::MCPlusBuilder;
@@ -260,17 +497,99 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
PCRelBaseOut = nullptr;
FixedEntryLoadInst = nullptr;
- // Check for the following long tail call sequence:
- // 1: auipc xi, %pcrel_hi(sym)
- // jalr zero, %pcrel_lo(1b)(xi)
- if (Instruction.getOpcode() == RISCV::JALR && Begin != End) {
- MCInst &PrevInst = *std::prev(End);
- if (isRISCVCall(PrevInst, Instruction) &&
- Instruction.getOperand(0).getReg() == RISCV::X0)
- return IndirectBranchType::POSSIBLE_TAIL_CALL;
+ (void)PtrSize;
+
+ unsigned TargetOperand;
+ switch (Instruction.getOpcode()) {
+ default:
+ return IndirectBranchType::UNKNOWN;
+ case RISCV::JALR:
+ if (Instruction.getOperand(0).getReg() != RISCV::X0)
+ return IndirectBranchType::UNKNOWN;
+ TargetOperand = 1;
+ break;
+ case RISCV::C_JR:
+ TargetOperand = 0;
+ break;
}
- return IndirectBranchType::UNKNOWN;
+ LocalUDChain UDChain = computeLocalUDChain(&Instruction, Begin, End);
+ const MCInst *TargetDef =
+ getOperandDef(Instruction, TargetOperand, UDChain);
+
+ // Check for a long tail call. The local use-def chain makes this robust
+ // against unrelated instructions between AUIPC and JALR.
+ if (Instruction.getOpcode() == RISCV::JALR && TargetDef &&
+ isRISCVCall(*TargetDef, Instruction))
+ return IndirectBranchType::POSSIBLE_TAIL_CALL;
+
+ // Jump-table dispatches use an unmodified register as the JALR target.
+ if (Instruction.getOpcode() == RISCV::JALR &&
+ (!Instruction.getOperand(2).isImm() ||
+ Instruction.getOperand(2).getImm() != 0))
+ return IndirectBranchType::UNKNOWN;
+
+ const MCInst *Root = followCopies(TargetDef, UDChain);
+ if (!Root)
+ return IndirectBranchType::UNKNOWN;
+
+ // PIC tables contain signed 32-bit offsets. Match
+ // add target, loaded-offset, table-base
+ // before the absolute-address form, which branches directly to the load.
+ if (Root->getOpcode() == RISCV::ADD || Root->getOpcode() == RISCV::C_ADD) {
+ for (unsigned LoadOperand : {1U, 2U}) {
+ const unsigned BaseOperand = LoadOperand == 1 ? 2 : 1;
+ JumpTableLoad Load;
+ if (!matchJumpTableLoad(getOperandDef(*Root, LoadOperand, UDChain),
+ UDChain, Load) ||
+ Load.EntrySize != 4)
+ continue;
+
+ const MCExpr *TargetBase = matchJumpTableBase(
+ getOperandDef(*Root, BaseOperand, UDChain), UDChain);
+ if (!TargetBase)
+ continue;
+
+ ScaledAddress Address;
+ if (!matchScaledAddress(getOperandDef(*Load.Inst, 1, UDChain),
+ Load.EntrySize, UDChain, Address))
+ continue;
+ const MCExpr *LoadBase = matchJumpTableBase(Address.BaseDef, UDChain);
+ if (!LoadBase || !areSameJumpTable(TargetBase, LoadBase))
+ continue;
+
+ IndexRegNum = Address.IndexReg;
+ DispValue = Load.Offset;
+ DispExpr = LoadBase;
+ EntrySize = Load.EntrySize;
+ EntrySigned = true;
+ return IndirectBranchType::POSSIBLE_PIC_JUMP_TABLE;
+ }
+ }
+
+ // Absolute-address tables branch directly to a loaded 32/64-bit entry:
+ // load target, (table-base + index * entry-size)
+ // jr target
+ JumpTableLoad Load;
+ if (!matchJumpTableLoad(Root, UDChain, Load))
+ return IndirectBranchType::UNKNOWN;
+
+ ScaledAddress Address;
+ if (!matchScaledAddress(getOperandDef(*Load.Inst, 1, UDChain),
+ Load.EntrySize, UDChain, Address))
+ return IndirectBranchType::UNKNOWN;
+
+ const MCExpr *LoadBase = matchJumpTableBase(Address.BaseDef, UDChain);
+ if (!LoadBase)
+ return IndirectBranchType::UNKNOWN;
+
+ BaseRegNum = getNoRegister();
+ IndexRegNum = Address.IndexReg;
+ DispValue = Load.Offset;
+ DispExpr = LoadBase;
+ EntrySize = Load.EntrySize;
+ EntrySigned = Load.EntrySigned;
+ return IndirectBranchType::POSSIBLE_JUMP_TABLE;
}
bool convertJmpToTailCall(MCInst &Inst) override {
@@ -376,13 +695,13 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
setInstLabel(InstAUIPC, AUIPCLabel);
Code.emplace_back(std::move(InstAUIPC));
- // Load the call target from the GOT slot using LD on RV64 or LW on RV32.
MCInst InstLoad = MCInstBuilder(loadOpc())
.addReg(PLTScratchReg)
.addReg(PLTScratchReg)
.addImm(0);
// Pair the I-type LD/LW immediate with the label on AUIPC. RISC-V
- // R_RISCV_PCREL_LO12_I relocations name the corresponding HI20 location.
+ // R_RISCV_PCREL_LO12_I relocations name the corresponding HI20 location,
+ // not the final GOT-slot symbol.
setOperandToSymbolRef(InstLoad, /*OpNum=*/2, AUIPCLabel,
/*Addend=*/0, Ctx, ELF::R_RISCV_PCREL_LO12_I);
Code.emplace_back(std::move(InstLoad));
@@ -414,6 +733,11 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
if (!isTerminator(*I) || isTailCall(*I) || !isBranch(*I))
break;
+ // An indirect jump has no symbolic TBB operand. It may be a recognized
+ // jump-table dispatch and must not enter the direct unconditional path.
+ if (isIndirectBranch(*I))
+ return false;
+
// Handle unconditional branches.
if (isUnconditionalBranch(*I)) {
// If any code was seen after this unconditional branch, we've seen
@@ -427,10 +751,6 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
continue;
}
- // Handle conditional branches and ignore indirect branches
- if (isIndirectBranch(*I))
- return false;
-
if (CondBranch == nullptr) {
const MCSymbol *TargetBB = getTargetSymbol(*I);
if (TargetBB == nullptr) {
diff --git a/bolt/test/RISCV/jump-table-shared-anchor.s b/bolt/test/RISCV/jump-table-shared-anchor.s
new file mode 100644
index 0000000000000..ca2a9a93e0f3f
--- /dev/null
+++ b/bolt/test/RISCV/jump-table-shared-anchor.s
@@ -0,0 +1,92 @@
+// REQUIRES: system-linux,target=riscv64{{.*}}
+
+// Do not treat RV64 full-width label-address arrays as movable jump tables.
+// GCC can place multiple arrays at offsets from one shared anchor. Moving the
+// array at offset zero changes the anchor while an unrecognized reference to a
+// later array still relies on the original layout.
+
+// RUN: %clang %cflags64 -march=rv64gc -no-pie \
+// RUN: -Wl,--no-relax,--image-base=0x10000,--section-start=.text=0x20000,--section-start=.rodata=0x30000 \
+// RUN: -o %t %s
+// RUN: %t
+// RUN: llvm-bolt %t -o %t.bolt --jump-tables=move --print-cfg \
+// RUN: --print-only=shared_first,shared_second 2>&1 | FileCheck %s
+// RUN: %t.bolt
+
+// CHECK-LABEL: Binary Function "shared_first"
+// CHECK-NOT: JUMPTABLE
+// CHECK: jr a1 # UNKNOWN CONTROL FLOW
+// CHECK-LABEL: Binary Function "shared_second"
+// CHECK-NOT: JUMPTABLE
+// CHECK: jr a1 # UNKNOWN CONTROL FLOW
+
+ .text
+ .globl _start
+ .type _start, @function
+ .p2align 2
+_start:
+ li a0, 1
+ call shared_second
+ li t0, 41
+ bne a0, t0, .Lfail
+
+ li a0, 0
+ li a7, 93
+ ecall
+.Lfail:
+ li a0, 1
+ li a7, 93
+ ecall
+ .size _start, .-_start
+
+ .globl shared_first
+ .type shared_first, @function
+ .p2align 2
+shared_first:
+ lui a4, %hi(SHARED_ANCHOR)
+ addi a4, a4, %lo(SHARED_ANCHOR)
+ slli a1, a0, 3
+ add a1, a1, a4
+ ld a1, 0(a1)
+ jr a1
+.Lfirst0:
+ li a0, 30
+ ret
+.Lfirst1:
+ li a0, 31
+ ret
+ .size shared_first, .-shared_first
+
+ .globl shared_second
+ .type shared_second, @function
+ .p2align 2
+shared_second:
+ lui a4, %hi(SHARED_ANCHOR)
+ addi a4, a4, %lo(SHARED_ANCHOR)
+ slli a1, a0, 3
+ add a1, a1, a4
+ ld a1, 16(a1)
+ jr a1
+.Lsecond0:
+ li a0, 40
+ ret
+.Lsecond1:
+ li a0, 41
+ ret
+ .size shared_second, .-shared_second
+
+ .section .rodata,"a", at progbits
+ .globl SHARED_ANCHOR
+ .type SHARED_ANCHOR, @object
+ .p2align 3
+SHARED_ANCHOR:
+ .dword .Lfirst0
+ .dword .Lfirst1
+ .size SHARED_ANCHOR, .-SHARED_ANCHOR
+
+ .globl SECOND_JT
+ .type SECOND_JT, @object
+SECOND_JT:
+ .dword .Lsecond0
+ .dword .Lsecond1
+ .size SECOND_JT, .-SECOND_JT
diff --git a/bolt/test/RISCV/jump-table-use-def.s b/bolt/test/RISCV/jump-table-use-def.s
new file mode 100644
index 0000000000000..af1abf52530d7
--- /dev/null
+++ b/bolt/test/RISCV/jump-table-use-def.s
@@ -0,0 +1,120 @@
+// REQUIRES: system-linux,target=riscv64{{.*}}
+
+// Verify that RISC-V jump-table recognition follows the local register
+// use-def chain instead of relying on adjacent instructions. Exercise both
+// GCC-style absolute 32-bit entries and PIC-relative 32-bit entries.
+
+// RUN: %clang %cflags64 -march=rv64gc_zba -no-pie \
+// RUN: -Wl,--no-relax,--image-base=0x10000,--section-start=.text=0x20000,--section-start=.rodata=0x30000 \
+// RUN: -o %t %s
+// RUN: %t
+// RUN: llvm-bolt %t -o %t.bolt --jump-tables=move --print-cfg \
+// RUN: --print-jump-tables --print-only=abs_dispatch,pic_dispatch 2>&1 | \
+// RUN: FileCheck %s
+// RUN: %t.bolt
+
+// RUN: %clang %cflags32 -march=rv32imac_zba -no-pie \
+// RUN: -Wl,--no-relax,--image-base=0x10000,--section-start=.text=0x20000,--section-start=.rodata=0x30000 \
+// RUN: -o %t.rv32 %s
+// RUN: llvm-bolt %t.rv32 -o %t.rv32.bolt --jump-tables=move --print-cfg \
+// RUN: --print-jump-tables --print-only=abs_dispatch,pic_dispatch 2>&1 | \
+// RUN: FileCheck %s
+
+// CHECK-LABEL: Binary Function "abs_dispatch"
+// CHECK: jr a1 # JUMPTABLE @0x30000
+// CHECK-LABEL: Binary Function "pic_dispatch"
+// CHECK: jr a1 # JUMPTABLE @0x3000c
+// CHECK: Jump table ABS_JT for function abs_dispatch
+// CHECK: PIC Jump table PIC_JT for function pic_dispatch
+
+ .text
+ .globl _start
+ .type _start, @function
+ .p2align 2
+_start:
+ li a0, 1
+ call abs_dispatch
+ li t0, 11
+ bne a0, t0, .Lfail
+
+ li a0, 2
+ call pic_dispatch
+ li t0, 22
+ bne a0, t0, .Lfail
+
+ li a0, 0
+ li a7, 93
+ ecall
+.Lfail:
+ li a0, 1
+ li a7, 93
+ ecall
+ .size _start, .-_start
+
+ .globl abs_dispatch
+ .type abs_dispatch, @function
+ .p2align 2
+abs_dispatch:
+ lui a4, %hi(ABS_JT)
+ addi a4, a4, %lo(ABS_JT)
+ li t0, 7 // Unrelated instruction in the def chain.
+ sh2add a1, a0, a4
+ li t1, 8 // Unrelated instruction in the def chain.
+ lw a1, 0(a1)
+ li t2, 9 // The load need not be adjacent to JR.
+ jr a1
+.Labs0:
+ li a0, 10
+ ret
+.Labs1:
+ li a0, 11
+ ret
+.Labs2:
+ li a0, 12
+ ret
+ .size abs_dispatch, .-abs_dispatch
+
+ .globl pic_dispatch
+ .type pic_dispatch, @function
+ .p2align 2
+pic_dispatch:
+.Lpcrel_hi:
+ auipc a4, %pcrel_hi(PIC_JT)
+ addi a4, a4, %pcrel_lo(.Lpcrel_hi)
+ li t0, 7 // Unrelated instruction in the def chain.
+ slli a1, a0, 2
+ add a1, a1, a4
+ li t1, 8 // Unrelated instruction in the def chain.
+ lw a1, 0(a1)
+ li t2, 9 // Separate the load, ADD, and JR.
+ add a1, a1, a4
+ jr a1
+.Lpic0:
+ li a0, 20
+ ret
+.Lpic1:
+ li a0, 21
+ ret
+.Lpic2:
+ li a0, 22
+ ret
+ .size pic_dispatch, .-pic_dispatch
+
+ .section .rodata,"a", at progbits
+ .globl ABS_JT
+ .type ABS_JT, @object
+ .p2align 2
+ABS_JT:
+ .word .Labs0
+ .word .Labs1
+ .word .Labs2
+ .size ABS_JT, .-ABS_JT
+
+ .globl PIC_JT
+ .type PIC_JT, @object
+ .p2align 2
+PIC_JT:
+ .word .Lpic0 - PIC_JT
+ .word .Lpic1 - PIC_JT
+ .word .Lpic2 - PIC_JT
+ .size PIC_JT, .-PIC_JT
>From fa45200eb64c8bfe3e00a93f13dd937989f9b64b Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Mon, 3 Aug 2026 15:59:47 +0800
Subject: [PATCH 14/16] [BOLT][RISCV] Rewrite call pairs across basic-block
boundaries
---
bolt/lib/Passes/FixRISCVCallsPass.cpp | 66 +++++++++++++++++++--------
bolt/test/RISCV/relax.s | 4 +-
bolt/test/RISCV/reloc-call-split-bb.s | 42 +++++++++++++++++
3 files changed, 89 insertions(+), 23 deletions(-)
create mode 100644 bolt/test/RISCV/reloc-call-split-bb.s
diff --git a/bolt/lib/Passes/FixRISCVCallsPass.cpp b/bolt/lib/Passes/FixRISCVCallsPass.cpp
index 6b73bd6854c9d..f1d6a749786fd 100644
--- a/bolt/lib/Passes/FixRISCVCallsPass.cpp
+++ b/bolt/lib/Passes/FixRISCVCallsPass.cpp
@@ -9,8 +9,6 @@
#include "bolt/Passes/FixRISCVCallsPass.h"
#include "bolt/Core/ParallelUtilities.h"
-#include <iterator>
-
using namespace llvm;
namespace llvm {
@@ -21,8 +19,17 @@ void FixRISCVCallsPass::runOnFunction(BinaryFunction &BF) {
auto &MIB = BC.MIB;
auto *Ctx = BC.Ctx.get();
+ MCInst *Previous = nullptr;
+ BinaryBasicBlock *PreviousBB = nullptr;
for (auto &BB : BF) {
for (auto II = BB.begin(); II != BB.end();) {
+ // CFI and other zero-sized pseudo instructions do not break an
+ // AUIPC/JALR pair in the input instruction stream.
+ if (MIB->isPseudo(*II)) {
+ ++II;
+ continue;
+ }
+
if (MIB->isCall(*II) && !MIB->isIndirectCall(*II)) {
auto *Target = MIB->getTargetSymbol(*II);
assert(Target && "Cannot find call target");
@@ -36,35 +43,54 @@ void FixRISCVCallsPass::runOnFunction(BinaryFunction &BF) {
MIB->createCall(*II, Target, Ctx);
MIB->moveAnnotations(std::move(OldCall), *II);
+ Previous = &*II;
+ PreviousBB = &BB;
++II;
continue;
}
- auto NextII = std::next(II);
-
- if (NextII == BB.end())
- break;
-
- if (MIB->isRISCVCall(*II, *NextII)) {
- auto *Target = MIB->getTargetSymbol(*II);
+ // A label, secondary entry point, or CFG boundary may split an
+ // AUIPC/JALR call pair across two basic blocks. Keep the previous real
+ // instruction across block boundaries so that the pair is still
+ // rewritten atomically. Otherwise the old JALR immediate remains in the
+ // encoding and can be ORed with the new R_RISCV_CALL_PLT fixup.
+ if (Previous && MIB->isRISCVCall(*Previous, *II)) {
+ auto *Target = MIB->getTargetSymbol(*Previous);
assert(Target && "Cannot find call target");
- MCInst OldCall = *NextII;
+ MCInst OldCall = *II;
auto L = BC.scopeLock();
- MIB->createNoop(*II);
-
- if (MIB->isTailCall(*NextII))
- MIB->createTailCall(*NextII, Target, Ctx);
- else
- MIB->createCall(*NextII, Target, Ctx);
-
- MIB->moveAnnotations(std::move(OldCall), *NextII);
-
- II = std::next(NextII);
+ if (PreviousBB == &BB) {
+ // Keep the original JALR offset annotation on the combined call, but
+ // emit the pseudo at the AUIPC position and remove the JALR. This
+ // preserves profile attribution without adding an executed NOP to
+ // every long call.
+ if (MIB->isTailCall(*II))
+ MIB->createTailCall(*Previous, Target, Ctx);
+ else
+ MIB->createCall(*Previous, Target, Ctx);
+ MIB->moveAnnotations(std::move(OldCall), *Previous);
+ II = BB.eraseInstruction(II);
+ } else {
+ // Keep split pairs in their original basic blocks. Moving the call
+ // across a CFG boundary would invalidate block-level control-flow
+ // information.
+ MIB->createNoop(*Previous);
+ if (MIB->isTailCall(*II))
+ MIB->createTailCall(*II, Target, Ctx);
+ else
+ MIB->createCall(*II, Target, Ctx);
+ MIB->moveAnnotations(std::move(OldCall), *II);
+ ++II;
+ }
+ Previous = nullptr;
+ PreviousBB = nullptr;
continue;
}
+ Previous = &*II;
+ PreviousBB = &BB;
++II;
}
}
diff --git a/bolt/test/RISCV/relax.s b/bolt/test/RISCV/relax.s
index 41124751f38e8..74f049b8f8dd9 100644
--- a/bolt/test/RISCV/relax.s
+++ b/bolt/test/RISCV/relax.s
@@ -12,15 +12,13 @@
// CHECK: Binary Function "_start" after fix-riscv-calls {
// CHECK: call near_f
-// CHECK-NEXT: nop
// CHECK-NEXT: call far_f
// CHECK-NEXT: tail near_f
// OBJDUMP: 0000000000600000 <_start>:
// OBJDUMP-NEXT: jal 0x600040 <near_f>
-// OBJDUMP-NEXT: nop
// OBJDUMP-NEXT: auipc ra, 0x200
-// OBJDUMP-NEXT: jalr 0x78(ra)
+// OBJDUMP-NEXT: jalr 0x7c(ra)
// OBJDUMP-NEXT: j 0x600040 <near_f>
// OBJDUMP: 0000000000600040 <near_f>:
// OBJDUMP: 0000000000800080 <far_f>:
diff --git a/bolt/test/RISCV/reloc-call-split-bb.s b/bolt/test/RISCV/reloc-call-split-bb.s
new file mode 100644
index 0000000000000..fffdc78c408c6
--- /dev/null
+++ b/bolt/test/RISCV/reloc-call-split-bb.s
@@ -0,0 +1,42 @@
+// Check that FixRISCVCalls rewrites an AUIPC/JALR pair even when a branch
+// target splits the pair across two basic blocks.
+
+// RUN: llvm-mc -triple riscv64 -mattr=+c -filetype=obj -o %t.o %s
+// RUN: ld.lld --emit-relocs -o %t %t.o
+// RUN: llvm-bolt --print-cfg --print-fix-riscv-calls --print-only=_start \
+// RUN: -o %t.bolt %t | FileCheck %s
+// RUN: llvm-objdump -d %t.bolt | FileCheck --check-prefix=OBJDUMP %s
+
+ .text
+ .option norvc
+
+ .globl target
+ .p2align 2
+target:
+ ret
+ .size target, .-target
+
+ .globl _start
+ .p2align 2
+_start:
+ // This branch is never taken, but makes .Ljalr a basic-block entry.
+ bne zero, zero, .Ljalr
+.Lcall:
+ auipc ra, 0
+ .reloc .Lcall, R_RISCV_CALL_PLT, target
+.Ljalr:
+ jalr ra, ra, 0
+ ret
+ .size _start, .-_start
+
+// CHECK-LABEL: Binary Function "_start" after building cfg {
+// CHECK: auipc ra, target
+// CHECK: jalr
+
+// CHECK-LABEL: Binary Function "_start" after fix-riscv-calls {
+// CHECK: nop
+// CHECK: call target
+
+// OBJDUMP-LABEL: <_start>:
+// OBJDUMP: nop
+// OBJDUMP-NEXT: jal
>From 9947a2fd499cc3f5faf2509a643a3fe51c19f381 Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Mon, 3 Aug 2026 15:59:54 +0800
Subject: [PATCH 15/16] [BOLT][RISCV] Retarget duplicated jump-table dispatches
---
bolt/include/bolt/Core/MCPlusBuilder.h | 9 ++++
bolt/lib/Core/BinaryFunction.cpp | 32 +++++++++++++
bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp | 47 ++++++++++++++++++++
3 files changed, 88 insertions(+)
diff --git a/bolt/include/bolt/Core/MCPlusBuilder.h b/bolt/include/bolt/Core/MCPlusBuilder.h
index b92d57724ed79..77def0b4f2b22 100644
--- a/bolt/include/bolt/Core/MCPlusBuilder.h
+++ b/bolt/include/bolt/Core/MCPlusBuilder.h
@@ -1154,6 +1154,15 @@ class MCPlusBuilder {
return nullptr;
}
+ /// Retarget the reference used by the jump-table dispatch at the end of
+ /// \p InstrWindow from \p OldTarget to \p NewTarget. Targets that need
+ /// architecture-specific multi-instruction matching can override this hook.
+ virtual bool replaceJumpTableReference(
+ MutableArrayRef<MCInst> InstrWindow, const MCSymbol *OldTarget,
+ const MCSymbol *NewTarget, MCContext *Ctx) const {
+ return false;
+ }
+
/// \brief Given a branch instruction try to get the address the branch
/// targets. Return true on success, and the address in Target.
virtual bool evaluateBranch(const MCInst &Inst, uint64_t Addr, uint64_t Size,
diff --git a/bolt/lib/Core/BinaryFunction.cpp b/bolt/lib/Core/BinaryFunction.cpp
index 6899f746cddd8..488a7ef388a42 100644
--- a/bolt/lib/Core/BinaryFunction.cpp
+++ b/bolt/lib/Core/BinaryFunction.cpp
@@ -4284,6 +4284,38 @@ void BinaryFunction::disambiguateJumpTables(
continue;
if (JumpTables.insert(JT).second)
continue;
+
+ if (BC.isRISCV()) {
+ const uint64_t JTAddress = BC.MIB->getJumpTable(Inst);
+ const uint64_t JTOffset = JTAddress - JT->getAddress();
+ const auto LabelIt = JT->Labels.find(JTOffset);
+ if (LabelIt == JT->Labels.end()) {
+ BC.errs() << "BOLT-ERROR: failed to find RISC-V jump table label at "
+ << "offset 0x" << Twine::utohexstr(JTOffset)
+ << " in function " << *this << '\n';
+ exit(1);
+ }
+
+ const MCSymbol *OldJTLabel = LabelIt->second;
+ uint64_t NewJumpTableID = 0;
+ const MCSymbol *NewJTLabel;
+ std::tie(NewJumpTableID, NewJTLabel) =
+ BC.duplicateJumpTable(*this, JT, OldJTLabel);
+
+ MutableArrayRef<MCInst> InstrWindow(&*BB->begin(), &Inst + 1);
+ if (!BC.MIB->replaceJumpTableReference(
+ InstrWindow, OldJTLabel, NewJTLabel, BC.Ctx.get())) {
+ BC.errs() << "BOLT-ERROR: failed to retarget duplicated RISC-V jump "
+ "table in function "
+ << *this << '\n';
+ exit(1);
+ }
+
+ const uint16_t IndexReg = BC.MIB->getJumpTableIndexReg(Inst);
+ BC.MIB->setJumpTable(Inst, NewJumpTableID, IndexReg, AllocId);
+ continue;
+ }
+
// This instruction is an indirect jump using a jump table, but it is
// using the same jump table of another jump. Try all our tricks to
// extract the jump table symbol and make it point to a new, duplicated JT
diff --git a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
index 5163aab88de5e..802e8d7aaabe3 100644
--- a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
+++ b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
@@ -275,6 +275,35 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
return getTargetSymbolInfo(LHS) == getTargetSymbolInfo(RHS);
}
+ bool replaceJumpTableSymbol(MCInst &Inst, const MCSymbol *OldTarget,
+ const MCSymbol *NewTarget,
+ MCContext *Ctx) const {
+ for (unsigned OpIndex = 0;
+ OpIndex < MCPlus::getNumPrimeOperands(Inst); ++OpIndex) {
+ MCOperand &Operand = Inst.getOperand(OpIndex);
+ if (!Operand.isExpr())
+ continue;
+
+ const MCExpr *Expr = Operand.getExpr();
+ const auto *Specifier = dyn_cast<MCSpecifierExpr>(Expr);
+ const MCExpr *SubExpr = Specifier ? Specifier->getSubExpr() : Expr;
+ const auto [Symbol, Addend] = getTargetSymbolInfo(SubExpr);
+ if (Symbol != OldTarget)
+ continue;
+
+ const MCExpr *NewExpr = MCSymbolRefExpr::create(NewTarget, *Ctx);
+ if (Addend)
+ NewExpr = MCBinaryExpr::createAdd(
+ NewExpr, MCConstantExpr::create(Addend, *Ctx), *Ctx);
+ if (Specifier)
+ NewExpr =
+ MCSpecifierExpr::create(NewExpr, Specifier->getSpecifier(), *Ctx);
+ Operand = MCOperand::createExpr(NewExpr);
+ return true;
+ }
+ return false;
+ }
+
public:
using MCPlusBuilder::MCPlusBuilder;
@@ -592,6 +621,24 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
return IndirectBranchType::POSSIBLE_JUMP_TABLE;
}
+ bool replaceJumpTableReference(
+ MutableArrayRef<MCInst> InstrWindow, const MCSymbol *OldTarget,
+ const MCSymbol *NewTarget, MCContext *Ctx) const override {
+ for (MCInst &Inst : llvm::reverse(InstrWindow)) {
+ switch (Inst.getOpcode()) {
+ default:
+ continue;
+ case RISCV::AUIPC:
+ case RISCV::LUI:
+ case RISCV::C_LUI:
+ break;
+ }
+ if (replaceJumpTableSymbol(Inst, OldTarget, NewTarget, Ctx))
+ return true;
+ }
+ return false;
+ }
+
bool convertJmpToTailCall(MCInst &Inst) override {
if (isTailCall(Inst))
return false;
>From b41b31506d0cc12c4bfb8392a0d9c130fe3f0a9e Mon Sep 17 00:00:00 2001
From: Thrrreeeee <1379998393 at qq.com>
Date: Mon, 3 Aug 2026 16:00:45 +0800
Subject: [PATCH 16/16] [BOLT][RISCV] Fix atomic-add operand order
---
bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp | 5 +++--
1 file changed, 3 insertions(+), 2 deletions(-)
diff --git a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
index 802e8d7aaabe3..5465306c36b69 100644
--- a/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
+++ b/bolt/lib/Target/RISCV/RISCVMCPlusBuilder.cpp
@@ -1050,10 +1050,11 @@ class RISCVMCPlusBuilder : public MCPlusBuilder {
void atomicAdd(MCInst &Inst, MCPhysReg RegAtomic, MCPhysReg RegTo,
MCPhysReg RegCnt) const {
+ // AMO operands are ordered as rd, rs2 (value), rs1 (address).
Inst = MCInstBuilder(atomicAddOpc())
.addReg(RegAtomic)
- .addReg(RegTo)
- .addReg(RegCnt);
+ .addReg(RegCnt)
+ .addReg(RegTo);
}
InstructionListType createRegCmpJE(MCPhysReg RegNo, const MCSymbol *Target,
More information about the llvm-commits
mailing list