[llvm] [BOLT][AArch64] Relax calls and branches with fragment clusters (PR #215825)
Alexandros Lamprineas via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 02:16:41 PDT 2026
https://github.com/labrinea updated https://github.com/llvm/llvm-project/pull/215825
>From e989c9aea5e047c1eaf8db273a3033e8ffce6d8d Mon Sep 17 00:00:00 2001
From: Alexandros Lamprineas <alexandros.lamprineas at arm.com>
Date: Sun, 23 Aug 2026 19:20:43 +0000
Subject: [PATCH 1/2] [BOLT][AArch64] Relax calls and branches with fragment
clusters
Build relaxation clusters from emitted function fragments instead of
whole functions, so split hot/cold code is modeled in the same order
it will be emitted. Use those clusters to place short and long call
thunks, and to build forward/backward branch thunk chains for
cross-cluster unconditional branches. This lets the experimental
AArch64 relaxation path handle large split binaries where direct
call/branch relocations can cross the +/-128MiB branch range after
BOLT layout changes.
Tested with stage2-clang-bolt and Chromium using Speedometer-generated
profile data, with lite mode disabled and without assuming PLT entries
are nearby.
---
bolt/include/bolt/Core/BinaryContext.h | 3 +
bolt/include/bolt/Core/BinaryFunction.h | 12 +
bolt/include/bolt/Passes/LongJmp.h | 75 ++-
bolt/lib/Core/BinaryContext.cpp | 102 ++-
bolt/lib/Passes/LongJmp.cpp | 597 +++++++++++-------
bolt/lib/Rewrite/RewriteInstance.cpp | 104 +--
.../AArch64/relax-branches-with-thunk-chain.s | 309 +++++++++
.../AArch64/{relax-exp.s => relax-calls.s} | 12 +-
.../test/AArch64/relax-cross-fragment-calls.s | 216 +++++++
9 files changed, 1070 insertions(+), 360 deletions(-)
create mode 100644 bolt/test/AArch64/relax-branches-with-thunk-chain.s
rename bolt/test/AArch64/{relax-exp.s => relax-calls.s} (86%)
create mode 100644 bolt/test/AArch64/relax-cross-fragment-calls.s
diff --git a/bolt/include/bolt/Core/BinaryContext.h b/bolt/include/bolt/Core/BinaryContext.h
index 92cd853870cae..f0bd997effa3a 100644
--- a/bolt/include/bolt/Core/BinaryContext.h
+++ b/bolt/include/bolt/Core/BinaryContext.h
@@ -1199,6 +1199,9 @@ class BinaryContext {
return ".text.injected.cold";
}
+ /// Return true if \p A should be emitted before \p B in code section order.
+ bool compareSectionNames(StringRef A, StringRef B) const;
+
ErrorOr<BinarySection &> getGdbIndexSection() const {
return getUniqueSectionByName(".gdb_index");
}
diff --git a/bolt/include/bolt/Core/BinaryFunction.h b/bolt/include/bolt/Core/BinaryFunction.h
index 14d7f9b5b5359..3865e87dc25ef 100644
--- a/bolt/include/bolt/Core/BinaryFunction.h
+++ b/bolt/include/bolt/Core/BinaryFunction.h
@@ -394,6 +394,9 @@ class BinaryFunction {
/// True if the function is used for patching code at a fixed address.
bool IsPatch{false};
+ /// True if the function is a synthetic branch/call thunk.
+ bool IsThunk{false};
+
/// True if the original entry point of the function may get called, but the
/// original body cannot be executed and needs to be patched with code that
/// redirects execution to the new function body.
@@ -1504,6 +1507,9 @@ class BinaryFunction {
/// Return true if this function is used for patching existing code.
bool isPatch() const { return IsPatch; }
+ /// Return true if this function is a synthetic branch/call thunk.
+ bool isThunk() const { return IsThunk; }
+
/// Return true if the function requires a patch.
bool needsPatch() const { return NeedsPatch; }
@@ -1941,6 +1947,12 @@ class BinaryFunction {
IsPatch = V;
}
+ /// Indicate that this function is a synthetic branch/call thunk.
+ void setIsThunk(bool V) {
+ assert(isInjected() && "Only injected functions can be used as thunks");
+ IsThunk = V;
+ }
+
/// Mark the function for patching.
void setNeedsPatch(bool V) { NeedsPatch = V; }
diff --git a/bolt/include/bolt/Passes/LongJmp.h b/bolt/include/bolt/Passes/LongJmp.h
index fa8cf67a95d73..37c7e1452d5bf 100644
--- a/bolt/include/bolt/Passes/LongJmp.h
+++ b/bolt/include/bolt/Passes/LongJmp.h
@@ -10,6 +10,8 @@
#define BOLT_PASSES_LONGJMP_H
#include "bolt/Passes/BinaryPasses.h"
+#include "llvm/ADT/SmallString.h"
+#include "llvm/ADT/SmallVector.h"
namespace llvm {
namespace bolt {
@@ -80,46 +82,71 @@ class LongJmpPass : public BinaryFunctionPass {
bool relaxLocalBranches(BinaryFunction &BF,
const BranchLivenessInfo *BLI = nullptr);
- /// A group of functions that are located within the longest direct
- /// branch/call instruction distance. Functions within the cluster do not
- /// require a thunk for calls in the same cluster. The cluster may include
- /// a set of thunks for covering calls to functions outside.
- struct FunctionCluster {
- /// All functions in this cluster.
- DenseSet<BinaryFunction *> Functions;
-
- /// Symbols corresponding to entry points of functions that this cluster
- /// calls. Note that it excludes all functions in the cluster itself.
- DenseSet<const MCSymbol *> Callees;
+ /// A group of function fragments that are located within the longest direct
+ /// branch/call instruction distance. Jumps within the cluster do not require
+ /// a thunk. The cluster may include thunks for jumps to targets outside.
+ struct FragmentCluster {
+ /// Output code section containing fragments in this cluster.
+ SmallString<32> SectionName;
/// Estimated size of the cluster in bytes.
uint64_t Size{0};
- /// The index of the last function in the cluster. Used as an insertion
- /// point for adding thunks to the output function list.
- size_t LastFunctionIndex = -1;
+ /// Number of function fragments in the cluster.
+ size_t NumFragments{0};
- /// When placing hot code at the end of the binary, track the first function
- /// for insertion purposes.
+ /// The indices of the first and last functions contributing fragments to
+ /// this cluster. Used as insertion points for adding thunks to the output
+ /// function list.
size_t FirstFunctionIndex = -1;
+ size_t LastFunctionIndex = -1;
- /// Thunks located at the end of this cluster.
- BinaryFunctionListType ThunkList;
+ /// Thunks located after this cluster.
+ BinaryFunctionListType ForwardThunkList;
- /// Thunks used by this cluster. Some could be in a ThunkList of the
- /// preceding cluster.
+ /// Thunks located before this cluster.
+ BinaryFunctionListType BackwardThunkList;
+
+ /// Call thunks used by this cluster.
///
/// <Function Symbol> -> <Thunk Function>.
- DenseMap<const MCSymbol *, BinaryFunction *> Thunks;
+ DenseMap<const MCSymbol *, BinaryFunction *> CallThunks;
+
+ /// <Destination Symbol> -> <Thunk Function>.
+ DenseMap<const MCSymbol *, BinaryFunction *> ForwardBranchThunks;
+ DenseMap<const MCSymbol *, BinaryFunction *> BackwardBranchThunks;
};
/// Maximum size of combined regular functions in the cluster. Note that it's
/// less than 128MB, because the size of the cluster plus its thunks should be
/// less than 128MB.
- static constexpr uint64_t MaxClusterSize = 125 * 1024 * 1024;
+ static constexpr uint64_t MaxClusterSize = 120 * 1024 * 1024;
+
+ struct FragmentClusterLayout {
+ SmallVector<FragmentCluster, 4> Clusters;
+ DenseMap<const BinaryBasicBlock *, unsigned> BBToCluster;
+ DenseMap<const MCSymbol *, unsigned> SymToCluster;
+ };
+
+ FragmentClusterLayout
+ buildClusterLayout(BinaryContext &BC,
+ const BinaryFunctionListType &OutputFunctions);
+
+ /// Relax calls using function fragment clusters.
+ void relaxCalls(BinaryContext &BC, BinaryFunctionListType &OutputFunctions,
+ FragmentClusterLayout &Layout);
+
+ /// Relax direct unconditional branches using function fragment clusters.
+ void relaxUnconditionalBranches(BinaryContext &BC,
+ BinaryFunctionListType &OutputFunctions,
+ FragmentClusterLayout &Layout);
+
+ /// Insert all thunks owned by the fragment cluster layout.
+ void insertClusterThunks(BinaryFunctionListType &OutputFunctions,
+ FragmentClusterLayout &Layout);
- /// Relax calls using function cluster approach.
- void relaxCalls(BinaryContext &BC);
+ /// Relax calls and direct unconditional branches using one cluster layout.
+ void relaxWithClusters(BinaryContext &BC);
/// -- Layout estimation methods --
/// Try to do layout before running the emitter, by looking at BinaryFunctions
diff --git a/bolt/lib/Core/BinaryContext.cpp b/bolt/lib/Core/BinaryContext.cpp
index d911f8191f791..94d8770dae493 100644
--- a/bolt/lib/Core/BinaryContext.cpp
+++ b/bolt/lib/Core/BinaryContext.cpp
@@ -54,6 +54,7 @@ namespace opts {
extern cl::opt<bool> LargeCodeModel;
extern cl::opt<bool> UpdateDebugSections;
+extern cl::opt<bool> HotFunctionsAtEnd;
static cl::opt<bool>
NoHugePages("no-huge-pages",
@@ -353,6 +354,103 @@ bool BinaryContext::forceSymbolRelocations(StringRef SymbolName) const {
return false;
}
+namespace {
+
+/// Defines the strict weak ordering for BOLT-produced code sections.
+class CodeSectionOrder {
+public:
+ CodeSectionOrder(StringRef ColdSectionName, StringRef HotTextMoverSectionName,
+ StringRef MainSectionName, StringRef WarmSectionName,
+ bool HotText, bool HotFunctionsAtEnd)
+ : ColdSectionName(ColdSectionName),
+ HotTextMoverSectionName(HotTextMoverSectionName),
+ MainSectionName(MainSectionName), WarmSectionName(WarmSectionName),
+ HotText(HotText), HotFunctionsAtEnd(HotFunctionsAtEnd) {}
+
+ bool operator()(StringRef AName, StringRef BName) const {
+ const SectionKind AKind = getKind(AName);
+ const SectionKind BKind = getKind(BName);
+ const unsigned ARank = getRank(AKind);
+ const unsigned BRank = getRank(BKind);
+ if (ARank != BRank)
+ return ARank < BRank;
+
+ if (AKind == SectionKind::Cold) {
+ if (AName.size() != BName.size())
+ return HotFunctionsAtEnd ? AName.size() > BName.size()
+ : AName.size() < BName.size();
+ if (AName != BName)
+ return HotFunctionsAtEnd ? AName > BName : AName < BName;
+ }
+
+ return false;
+ }
+
+private:
+ enum class SectionKind { Mover, Main, Warm, Cold, Other };
+
+ SectionKind getKind(StringRef Name) const {
+ if (HotText && Name == HotTextMoverSectionName)
+ return SectionKind::Mover;
+ if (Name == MainSectionName)
+ return SectionKind::Main;
+ if (Name == WarmSectionName)
+ return SectionKind::Warm;
+ if (Name.starts_with(ColdSectionName))
+ return SectionKind::Cold;
+ return SectionKind::Other;
+ }
+
+ unsigned getRank(SectionKind Kind) const {
+ if (Kind == SectionKind::Mover)
+ return 0;
+ if (HotFunctionsAtEnd) {
+ switch (Kind) {
+ case SectionKind::Other:
+ return 1;
+ case SectionKind::Cold:
+ return 2;
+ case SectionKind::Warm:
+ return 3;
+ case SectionKind::Main:
+ return 4;
+ case SectionKind::Mover:
+ llvm_unreachable("handled above");
+ }
+ }
+ switch (Kind) {
+ case SectionKind::Main:
+ return 1;
+ case SectionKind::Warm:
+ return 2;
+ case SectionKind::Cold:
+ return 3;
+ case SectionKind::Other:
+ return 4;
+ case SectionKind::Mover:
+ llvm_unreachable("handled above");
+ }
+ llvm_unreachable("unknown section kind");
+ }
+
+ StringRef ColdSectionName;
+ StringRef HotTextMoverSectionName;
+ StringRef MainSectionName;
+ StringRef WarmSectionName;
+ bool HotText;
+ bool HotFunctionsAtEnd;
+};
+
+} // namespace
+
+bool BinaryContext::compareSectionNames(StringRef A, StringRef B) const {
+ const CodeSectionOrder CompareSections(
+ getColdCodeSectionName(), getHotTextMoverSectionName(),
+ getMainCodeSectionName(), getWarmCodeSectionName(), opts::HotText,
+ opts::HotFunctionsAtEnd);
+ return CompareSections(A, B);
+}
+
std::unique_ptr<MCObjectWriter>
BinaryContext::createObjectWriter(raw_pwrite_stream &OS) {
return MAB->createObjectWriter(OS);
@@ -2816,7 +2914,9 @@ BinaryContext::createInstructionPatch(uint64_t Address,
BinaryFunction *
BinaryContext::createThunkBinaryFunction(const std::string &Name) {
static NameResolver NR;
- return createInjectedBinaryFunction(NR.uniquify(Name));
+ BinaryFunction *BF = createInjectedBinaryFunction(NR.uniquify(Name));
+ BF->setIsThunk(true);
+ return BF;
}
std::pair<size_t, size_t>
diff --git a/bolt/lib/Passes/LongJmp.cpp b/bolt/lib/Passes/LongJmp.cpp
index 30ff9dc4ddc71..ccdb2ef37f77f 100644
--- a/bolt/lib/Passes/LongJmp.cpp
+++ b/bolt/lib/Passes/LongJmp.cpp
@@ -1023,297 +1023,434 @@ bool LongJmpPass::relaxLocalBranches(BinaryFunction &BF,
return true;
}
-void LongJmpPass::relaxCalls(BinaryContext &BC) {
- // Operate on a copy of binary functions. We are going to manually insert new
- // thunks and update the list.
- BinaryFunctionListType OutputFunctions = BC.getOutputBinaryFunctions();
+static uint64_t estimateFragmentSize(const BinaryFunction &BF,
+ const FunctionFragment &FF) {
+ uint64_t Size = 0;
+ for (const BinaryBasicBlock *BB : FF)
+ Size += BB->estimateSize();
+
+ if (BF.hasIslandsInfo()) {
+ Size += BF.estimateConstantIslandSize();
+ if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
+ Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
+ }
- // Conservatively estimate emitted function size. Assume the worst case
- // alignment.
- auto estimateFunctionSize = [&](const BinaryFunction &BF) -> uint64_t {
- if (!BC.shouldEmit(BF))
- return 0;
- uint64_t Size = BF.estimateSize() + BF.getMaxAlignmentBytes();
-
- // Each additional fragment can attribute extra bytes due to its alignment
- // requirements.
- for ([[maybe_unused]] const FunctionFragment &FF :
- BF.getLayout().getSplitFragments())
- Size += BF.getMaxColdAlignmentBytes();
-
- if (BF.hasIslandsInfo()) {
- Size += BF.estimateConstantIslandSize();
- if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
- Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
- }
+ Size += FF.isSplitFragment() ? BF.getMaxColdAlignmentBytes()
+ : BF.getMaxAlignmentBytes();
+ return Size;
+}
- return Size;
+LongJmpPass::FragmentClusterLayout
+LongJmpPass::buildClusterLayout(BinaryContext &BC,
+ const BinaryFunctionListType &OutputFunctions) {
+ struct OutputFragment {
+ const FunctionFragment *FF;
+ size_t FunctionIndex;
+ SmallString<32> SectionName;
};
- // Map every function to its direct callees. Note that this is different from
- // the regular call graph as here we completely ignore indirect calls.
+ FragmentClusterLayout Layout;
+ SmallVector<OutputFragment> OrderedFragments;
uint64_t EstimatedSize = 0;
- DenseMap<BinaryFunction *, std::set<const MCSymbol *>> CallMap;
- for (BinaryFunction *BF : OutputFunctions) {
+ for (size_t I = 0; I < OutputFunctions.size(); ++I) {
+ BinaryFunction *BF = OutputFunctions[I];
if (!BC.shouldEmit(*BF) || BF->isPatch())
continue;
- EstimatedSize += estimateFunctionSize(*BF);
-
- for (const BinaryBasicBlock &BB : *BF) {
- for (const MCInst &Inst : BB) {
- if (!BC.MIB->isCall(Inst) || BC.MIB->isIndirectCall(Inst) ||
- BC.MIB->isIndirectBranch(Inst))
- continue;
- const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
- assert(TargetSymbol);
-
- // Ignore internal calls that use basic block labels as a destination.
- if (!BC.getFunctionForSymbol(TargetSymbol))
- continue;
+ for (const FunctionFragment &FF : BF->getLayout().fragments()) {
+ if (FF.empty() && !BF->hasConstantIsland())
+ continue;
- CallMap[BF].insert(TargetSymbol);
- }
+ OrderedFragments.push_back(
+ {&FF, I, BF->getCodeSectionName(FF.getFragmentNum())});
}
}
- LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << EstimatedSize
- << '\n');
-
- // Build clusters in the order the functions will appear in the output.
- std::vector<FunctionCluster> Clusters;
- for (size_t Index = 0, NumFuncs = OutputFunctions.size(); Index < NumFuncs;
- ++Index) {
- const size_t BFIndex =
- opts::HotFunctionsAtEnd ? NumFuncs - Index - 1 : Index;
- BinaryFunction *BF = OutputFunctions[BFIndex];
- if (!BC.shouldEmit(*BF) || BF->isPatch())
- continue;
+ // Model final output layout by grouping function fragments in output section
+ // order. Within each section, fragments remain in OutputFunctions order.
+ llvm::stable_sort(
+ OrderedFragments, [&](const OutputFragment &A, const OutputFragment &B) {
+ return BC.compareSectionNames(A.SectionName, B.SectionName);
+ });
- const uint64_t BFSize = estimateFunctionSize(*BF);
- if (Clusters.empty() || Clusters.back().Size + BFSize > MaxClusterSize) {
- Clusters.emplace_back(FunctionCluster());
- Clusters.back().FirstFunctionIndex = BFIndex;
+ auto addFragmentToCluster = [&](const OutputFragment &Fragment) {
+ BinaryFunction &BF = *OutputFunctions[Fragment.FunctionIndex];
+ const FunctionFragment &FF = *Fragment.FF;
+ const uint64_t FFSize = estimateFragmentSize(BF, FF);
+
+ if (Layout.Clusters.empty() ||
+ Layout.Clusters.back().SectionName != Fragment.SectionName ||
+ Layout.Clusters.back().Size + FFSize > MaxClusterSize) {
+ Layout.Clusters.emplace_back(FragmentCluster());
+ FragmentCluster &FC = Layout.Clusters.back();
+ FC.SectionName = Fragment.SectionName;
+ FC.FirstFunctionIndex = Fragment.FunctionIndex;
}
- FunctionCluster &FC = Clusters.back();
- FC.Functions.insert(BF);
-
- // When a function is added to the cluster, we have to remove all of its
- // symbols from the cluster callee list. These include alternative symbols
- // (e.g. after ICF) and secondary entry point symbols.
- for (const MCSymbol *Symbol : BF->getSymbols()) {
- auto It = FC.Callees.find(Symbol);
- if (It != FC.Callees.end())
- FC.Callees.erase(It);
- }
- BF->forEachEntryPoint(
- [&FC](uint64_t Offset, const MCSymbol *EntrySymbol) -> bool {
- auto It = FC.Callees.find(EntrySymbol);
- if (It != FC.Callees.end())
- FC.Callees.erase(It);
- return true;
- });
-
- // Update cluster callee list with added function callees.
- for (const MCSymbol *CalleeSymbol : CallMap[BF]) {
- BinaryFunction *Callee = BC.getFunctionForSymbol(CalleeSymbol);
- if (!FC.Functions.count(Callee)) {
- FC.Callees.insert(CalleeSymbol);
- }
+ FragmentCluster &FC = Layout.Clusters.back();
+ FC.LastFunctionIndex = Fragment.FunctionIndex;
+ ++FC.NumFragments;
+ EstimatedSize += FFSize;
+ const unsigned ClusterNum = Layout.Clusters.size() - 1;
+
+ // Map primary entry points.
+ if (FF.isMainFragment())
+ for (const MCSymbol *Symbol : BF.getSymbols())
+ Layout.SymToCluster[Symbol] = ClusterNum;
+
+ for (const BinaryBasicBlock *BB : FF) {
+ Layout.BBToCluster[BB] = ClusterNum;
+ if (const MCSymbol *Label = BB->getLabel())
+ Layout.SymToCluster[Label] = ClusterNum;
+
+ // Map secondary entry points.
+ if (MCSymbol *EntrySymbol = BF.getSecondaryEntryPointSymbol(*BB))
+ Layout.SymToCluster[EntrySymbol] = ClusterNum;
}
- FC.Size += BFSize;
- FC.LastFunctionIndex = BFIndex;
- }
+ FC.Size += FFSize;
+ };
- if (opts::HotFunctionsAtEnd) {
- std::reverse(Clusters.begin(), Clusters.end());
- llvm::for_each(Clusters, [](FunctionCluster &FC) {
- std::swap(FC.LastFunctionIndex, FC.FirstFunctionIndex);
- });
- }
+ for (const OutputFragment &Fragment : OrderedFragments)
+ addFragmentToCluster(Fragment);
- if (Clusters.empty())
- return;
+ if (Layout.Clusters.empty())
+ return Layout;
+
+ LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << EstimatedSize
+ << '\n');
// Print cluster stats.
- BC.outs() << "BOLT-INFO: built " << Clusters.size()
- << " function cluster(s)\n";
- uint64_t ClusterIndex = 0;
- for (const FunctionCluster &FC : Clusters) {
- BC.outs() << "BOLT-INFO: cluster: " << ClusterIndex++ << '\n'
- << "BOLT-INFO: " << FC.Functions.size() << " function(s)\n"
- << "BOLT-INFO: " << FC.Callees.size() << " callee(s)\n"
+ BC.outs() << "BOLT-INFO: built " << Layout.Clusters.size()
+ << " function fragment cluster(s)\n";
+ for (size_t I = 0; I < Layout.Clusters.size(); ++I) {
+ const FragmentCluster &FC = Layout.Clusters[I];
+ BC.outs() << "BOLT-INFO: cluster: " << I << '\n'
+ << "BOLT-INFO: " << FC.NumFragments << " fragment(s)\n"
<< "BOLT-INFO: " << FC.Size << " estimated bytes\n";
}
if (opts::RelaxPLT) {
// Populate one of the clusters with PLT functions based on the proximity of
// the PLT section to avoid unneeded thunk redirection.
- const size_t PLTClusterNum = opts::UseOldText ? Clusters.size() - 1 : 0;
- auto &PLTCluster = Clusters[PLTClusterNum];
- for (BinaryFunction &BF :
- llvm::make_second_range(BC.getBinaryFunctions())) {
+ const unsigned PLTCluster =
+ opts::UseOldText ? Layout.Clusters.size() - 1 : 0;
+ for (BinaryFunction &BF : llvm::make_second_range(BC.getBinaryFunctions()))
if (BF.isPLTFunction()) {
- PLTCluster.Functions.insert(&BF);
- auto It = PLTCluster.Callees.find(BF.getSymbol());
- if (It != PLTCluster.Callees.end())
- PLTCluster.Callees.erase(It);
+ for (const MCSymbol *Symbol : BF.getSymbols())
+ Layout.SymToCluster[Symbol] = PLTCluster;
+ BF.forEachEntryPoint(
+ [&](uint64_t, const MCSymbol *EntrySymbol) -> bool {
+ Layout.SymToCluster[EntrySymbol] = PLTCluster;
+ return true;
+ });
}
- }
}
- // Create a thunk with +-128MB span.
- size_t NumShortThunks = 0;
- auto createShortThunk = [&](const MCSymbol *TargetSymbol) {
- ++NumShortThunks;
- BinaryFunction *ThunkBF = BC.createThunkBinaryFunction(
- "__AArch64Thunk_" + TargetSymbol->getName().str());
- MCInst Inst;
- BC.MIB->createTailCall(Inst, TargetSymbol, BC.Ctx.get());
- ThunkBF->addBasicBlock()->addInstruction(Inst);
+ return Layout;
+}
- return ThunkBF;
+void LongJmpPass::relaxCalls(BinaryContext &BC,
+ BinaryFunctionListType &OutputFunctions,
+ FragmentClusterLayout &Layout) {
+ auto &Clusters = Layout.Clusters;
+ auto &BBToCluster = Layout.BBToCluster;
+ auto &SymToCluster = Layout.SymToCluster;
+
+ struct CrossClusterCall {
+ MCInst *Inst;
+ const MCSymbol *TargetSymbol;
+ unsigned SourceCluster;
+ unsigned TargetCluster;
};
- // Create a thunk with +-4GB span.
- size_t NumLongThunks = 0;
- auto createLongThunk = [&](const MCSymbol *TargetSymbol) {
- ++NumLongThunks;
- BinaryFunction *ThunkBF = BC.createThunkBinaryFunction(
- "__AArch64ADRPThunk_" + TargetSymbol->getName().str());
- InstructionListType Instructions;
- BC.MIB->createLongTailCall(Instructions, TargetSymbol, BC.Ctx.get());
- ThunkBF->addBasicBlock()->addInstructions(Instructions);
+ SmallVector<CrossClusterCall> CrossClusterCalls;
+ for (BinaryFunction *BF : OutputFunctions) {
+ if (!BC.shouldEmit(*BF) || BF->isPatch())
+ continue;
- return ThunkBF;
- };
+ for (BinaryBasicBlock &BB : *BF) {
+ auto SourceIt = BBToCluster.find(&BB);
+ if (SourceIt == BBToCluster.end())
+ continue;
+ const unsigned SourceCluster = SourceIt->second;
- for (unsigned ClusterNum = 0; ClusterNum < Clusters.size(); ++ClusterNum) {
- FunctionCluster &FC = Clusters[ClusterNum];
- SmallVector<const MCSymbol *, 16> Callees(FC.Callees.begin(),
- FC.Callees.end());
-
- // Generate thunks in deterministic order.
- llvm::sort(Callees, [&BC](const MCSymbol *A, const MCSymbol *B) {
- uint64_t EntryA;
- uint64_t EntryB;
- BinaryFunction *BFA = BC.getFunctionForSymbol(A, &EntryA);
- BinaryFunction *BFB = BC.getFunctionForSymbol(B, &EntryB);
- if (BFA == BFB) {
- if (EntryA != EntryB)
- return EntryA < EntryB;
-
- // Use lexicographical order for ICF'ed symbols.
- return A->getName() < B->getName();
- }
- return compareBinaryFunctionByIndex(BFA, BFB);
- });
+ for (MCInst &Inst : BB) {
+ if (!BC.MIB->isCall(Inst) && !BC.MIB->isUnconditionalBranch(Inst))
+ continue;
- // Return index of adjacent cluster containing the function.
- auto getAdjClusterWithFunction =
- [&](const BinaryFunction *BF) -> std::optional<unsigned> {
- if (ClusterNum > 0 && Clusters[ClusterNum - 1].Functions.count(BF))
- return ClusterNum - 1;
- if (ClusterNum + 1 < Clusters.size() &&
- Clusters[ClusterNum + 1].Functions.count(BF))
- return ClusterNum + 1;
- return std::nullopt;
- };
+ const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
+ if (!TargetSymbol)
+ continue;
- const FunctionCluster *PrevCluster =
- ClusterNum ? &Clusters[ClusterNum - 1] : nullptr;
+ auto It = SymToCluster.find(TargetSymbol);
+ const bool Found = It != SymToCluster.end();
+ // Known plain branches use B-only thunk chains.
+ // Unknown address-only branches may use tail-call thunks.
+ if (BC.MIB->isUnconditionalBranch(Inst) && Found)
+ continue;
- // Create short thunks for callees in adjacent clusters and long thunks
- // for callees outside.
- for (const MCSymbol *Callee : Callees) {
- if (FC.Thunks.count(Callee))
- continue;
+ if (!BC.getFunctionForSymbol(TargetSymbol) &&
+ !BC.getSymbolValue(*TargetSymbol))
+ continue;
- BinaryFunction *Thunk = 0;
- std::optional<unsigned> AdjCluster =
- getAdjClusterWithFunction(BC.getFunctionForSymbol(Callee));
- if (AdjCluster) {
- Thunk = createShortThunk(Callee);
- } else {
- // Previous cluster may already have a long thunk that can be reused.
- if (PrevCluster) {
- auto It = PrevCluster->Thunks.find(Callee);
- // Reuse only if previous cluster hosts this thunk.
- if (It != PrevCluster->Thunks.end() &&
- llvm::is_contained(PrevCluster->ThunkList, It->second)) {
- FC.Thunks[Callee] = It->second;
- continue;
- }
- }
- Thunk = createLongThunk(Callee);
- }
+ // If not found TargetCluster becomes UINT_MAX.
+ unsigned TargetCluster = Found ? It->second : -1;
+ if (TargetCluster == SourceCluster)
+ continue;
- // The cluster that will host this thunk. If the current cluster is the
- // last one, try to use the previous one. Matters when we want to have hot
- // functions at higher addresses under HotFunctionsAtEnd.
- FunctionCluster *ThunkCluster = &Clusters[ClusterNum];
- if ((AdjCluster && *AdjCluster == ClusterNum - 1) ||
- (ClusterNum && ClusterNum == Clusters.size() - 1))
- ThunkCluster = &Clusters[ClusterNum - 1];
- ThunkCluster->ThunkList.push_back(Thunk);
-
- // Register thunks for all symbols associated with the function.
- uint64_t EntryID = 0;
- const BinaryFunction *BF = BC.getFunctionForSymbol(Callee, &EntryID);
- if (EntryID != 0) {
- FC.Thunks[Callee] = Thunk;
- } else {
- for (const MCSymbol *Symbol : BF->getSymbols()) {
- FC.Thunks[Symbol] = Thunk;
- }
+ CrossClusterCalls.push_back(
+ {&Inst, TargetSymbol, SourceCluster, TargetCluster});
}
}
}
+ size_t NumShortThunks = 0;
+ size_t NumLongThunks = 0;
+ auto createCallThunk = [&](const MCSymbol *TargetSymbol, bool IsShort) {
+ BinaryFunction *Thunk = nullptr;
+ if (IsShort) {
+ ++NumShortThunks;
+ Thunk = BC.createThunkBinaryFunction("__AArch64Thunk_" +
+ TargetSymbol->getName().str());
+ MCInst Inst;
+ BC.MIB->createTailCall(Inst, TargetSymbol, BC.Ctx.get());
+ Thunk->addBasicBlock()->addInstruction(Inst);
+ } else {
+ ++NumLongThunks;
+ Thunk = BC.createThunkBinaryFunction("__AArch64ADRPThunk_" +
+ TargetSymbol->getName().str());
+ InstructionListType Instructions;
+ BC.MIB->createLongTailCall(Instructions, TargetSymbol, BC.Ctx.get());
+ Thunk->addBasicBlock()->addInstructions(Instructions);
+ }
+ return Thunk;
+ };
+
+ auto registerCallThunk = [&](FragmentCluster &FC, const MCSymbol *Callee,
+ BinaryFunction *Thunk) {
+ uint64_t EntryID = 0;
+ const BinaryFunction *BF = BC.getFunctionForSymbol(Callee, &EntryID);
+ const bool IsPrimaryEntry = EntryID == 0;
+ if (BF && IsPrimaryEntry)
+ for (const MCSymbol *Symbol : BF->getSymbols())
+ FC.CallThunks[Symbol] = Thunk;
+ else
+ FC.CallThunks[Callee] = Thunk;
+ };
+
+ auto getOrCreateCallThunk = [&](const MCSymbol *TargetSymbol,
+ unsigned SourceCluster, bool IsShort,
+ bool IsForward) {
+ FragmentCluster &FC = Clusters[SourceCluster];
+ if (auto It = FC.CallThunks.find(TargetSymbol); It != FC.CallThunks.end())
+ return It->second;
+
+ BinaryFunction *Thunk = createCallThunk(TargetSymbol, IsShort);
+
+ FragmentCluster &ThunkCluster = Clusters[SourceCluster];
+ Thunk->setCodeSectionName(ThunkCluster.SectionName);
+ BinaryFunctionListType &ThunkList = IsForward
+ ? ThunkCluster.ForwardThunkList
+ : ThunkCluster.BackwardThunkList;
+ ThunkList.push_back(Thunk);
+
+ // Register thunks for all symbols associated with the function.
+ registerCallThunk(FC, TargetSymbol, Thunk);
+ return Thunk;
+ };
+
+ auto areAdjacent = [](unsigned A, unsigned B) {
+ return A > B ? A - B == 1 : B - A == 1;
+ };
+
+ for (CrossClusterCall &Call : CrossClusterCalls) {
+ const bool IsForward = Call.SourceCluster < Call.TargetCluster;
+ const bool Adjacent = areAdjacent(Call.SourceCluster, Call.TargetCluster);
+
+ BinaryFunction *Thunk = getOrCreateCallThunk(
+ Call.TargetSymbol, Call.SourceCluster, Adjacent, IsForward);
+ BC.MIB->replaceBranchTarget(*Call.Inst, Thunk->getSymbol(), BC.Ctx.get());
+ }
+
+ if (!CrossClusterCalls.empty())
+ BC.outs() << "BOLT-INFO: relaxed " << CrossClusterCalls.size()
+ << " calls with thunks\n";
+
if (NumShortThunks)
BC.outs() << "BOLT-INFO: " << NumShortThunks << " short thunks created\n";
if (NumLongThunks)
BC.outs() << "BOLT-INFO: " << NumLongThunks << " long thunks created\n";
+}
- // Replace callees with thunks.
- for (FunctionCluster &FC : Clusters) {
- for (BinaryFunction *BF : FC.Functions) {
- if (!CallMap.count(BF))
+void LongJmpPass::relaxUnconditionalBranches(
+ BinaryContext &BC, BinaryFunctionListType &OutputFunctions,
+ FragmentClusterLayout &Layout) {
+ auto &Clusters = Layout.Clusters;
+ auto &BBToCluster = Layout.BBToCluster;
+ auto &SymToCluster = Layout.SymToCluster;
+
+ struct CrossClusterBranch {
+ MCInst *Inst;
+ const MCSymbol *TargetSymbol;
+ unsigned SourceCluster;
+ unsigned TargetCluster;
+ };
+
+ SmallVector<CrossClusterBranch> CrossClusterBranches;
+ for (BinaryFunction *BF : OutputFunctions) {
+ if (!BC.shouldEmit(*BF) || BF->isPatch())
+ continue;
+
+ for (BinaryBasicBlock &BB : *BF) {
+ auto SourceIt = BBToCluster.find(&BB);
+ if (SourceIt == BBToCluster.end())
continue;
+ const unsigned SourceCluster = SourceIt->second;
- for (BinaryBasicBlock &BB : *BF) {
- for (MCInst &Inst : BB) {
- if (!BC.MIB->isCall(Inst) || BC.MIB->isIndirectCall(Inst) ||
- BC.MIB->isIndirectBranch(Inst))
- continue;
- const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
- assert(TargetSymbol);
+ for (MCInst &Inst : BB) {
+ if (!BC.MIB->isUnconditionalBranch(Inst))
+ continue;
- auto It = FC.Thunks.find(TargetSymbol);
- if (It != FC.Thunks.end())
- BC.MIB->replaceBranchTarget(Inst, It->second->getSymbol(),
- BC.Ctx.get());
- }
+ const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
+ if (!TargetSymbol)
+ continue;
+
+ auto TargetIt = SymToCluster.find(TargetSymbol);
+ if (TargetIt == SymToCluster.end())
+ continue;
+
+ const unsigned TargetCluster = TargetIt->second;
+ if (SourceCluster == TargetCluster)
+ continue;
+
+ CrossClusterBranches.push_back(
+ {&Inst, TargetSymbol, SourceCluster, TargetCluster});
}
}
}
- // Add thunks to the function list and assign a section name matching the
- // function they follow.
- for (const FunctionCluster &FC : llvm::reverse(Clusters)) {
- std::string SectionName =
- OutputFunctions[FC.LastFunctionIndex]->getCodeSectionName().str().str();
- for (BinaryFunction *Thunk : FC.ThunkList) {
- Thunk->setCodeSectionName(SectionName);
+ size_t NumBranchThunks = 0;
+ auto createBranchThunk = [&](const MCSymbol *TargetSymbol,
+ const bool IsForward) {
+ std::string ThunkName = IsForward ? "__AArch64BranchForwardThunk_"
+ : "__AArch64BranchBackwardThunk_";
+ ThunkName += std::to_string(NumBranchThunks++);
+
+ BinaryFunction *ThunkBF = BC.createThunkBinaryFunction(ThunkName);
+ MCInst Inst;
+ BC.MIB->createUncondBranch(Inst, TargetSymbol, BC.Ctx.get());
+ ThunkBF->addBasicBlock()->addInstruction(Inst);
+ return ThunkBF;
+ };
+
+ auto getOrCreateBranchThunk =
+ [&](FragmentCluster &Cluster, const MCSymbol *TargetSymbol,
+ const MCSymbol *NextTarget, const bool IsForward) {
+ auto &Thunks = IsForward ? Cluster.ForwardBranchThunks
+ : Cluster.BackwardBranchThunks;
+ auto It = Thunks.find(TargetSymbol);
+ if (It != Thunks.end())
+ return It->second;
+
+ BinaryFunction *Thunk = createBranchThunk(NextTarget, IsForward);
+ Thunk->setCodeSectionName(Cluster.SectionName);
+ auto &ThunkList =
+ IsForward ? Cluster.ForwardThunkList : Cluster.BackwardThunkList;
+ ThunkList.push_back(Thunk);
+ Thunks[TargetSymbol] = Thunk;
+ return Thunk;
+ };
+
+ auto getOrCreateBranchThunkChain =
+ [&](const CrossClusterBranch &Branch) -> const MCSymbol * {
+ const unsigned SourceCluster = Branch.SourceCluster;
+ const unsigned TargetCluster = Branch.TargetCluster;
+ BinaryFunction *FirstThunk = nullptr;
+ const MCSymbol *NextTarget = Branch.TargetSymbol;
+
+ if (SourceCluster < TargetCluster) {
+ for (unsigned Cluster = TargetCluster; Cluster > SourceCluster;) {
+ --Cluster;
+ FirstThunk =
+ getOrCreateBranchThunk(Clusters[Cluster], Branch.TargetSymbol,
+ NextTarget, /*IsForward=*/true);
+ NextTarget = FirstThunk->getSymbol();
+ }
+ } else {
+ for (unsigned Cluster = TargetCluster + 1; Cluster <= SourceCluster;
+ ++Cluster) {
+ FirstThunk =
+ getOrCreateBranchThunk(Clusters[Cluster], Branch.TargetSymbol,
+ NextTarget, /*IsForward=*/false);
+ NextTarget = FirstThunk->getSymbol();
+ }
}
+ assert(FirstThunk && "expected branch thunk chain");
+ return FirstThunk->getSymbol();
+ };
+
+ for (const CrossClusterBranch &Branch : CrossClusterBranches) {
+ const MCSymbol *Target = getOrCreateBranchThunkChain(Branch);
+ BC.MIB->replaceBranchTarget(*Branch.Inst, Target, BC.Ctx.get());
+ }
+
+ if (!CrossClusterBranches.empty())
+ BC.outs() << "BOLT-INFO: relaxed " << CrossClusterBranches.size()
+ << " cross-cluster branches\n";
+
+ if (NumBranchThunks)
+ BC.outs() << "BOLT-INFO: " << NumBranchThunks << " branch thunks created\n";
+}
+
+void LongJmpPass::insertClusterThunks(BinaryFunctionListType &OutputFunctions,
+ FragmentClusterLayout &Layout) {
+ struct ThunkInsertion {
+ size_t Position;
+ bool IsForward;
+ BinaryFunctionListType *ThunkList;
+ };
+
+ SmallVector<ThunkInsertion> Insertions;
+ for (FragmentCluster &Cluster : Layout.Clusters) {
+ if (!Cluster.BackwardThunkList.empty())
+ Insertions.push_back({Cluster.FirstFunctionIndex, /*IsForward=*/false,
+ &Cluster.BackwardThunkList});
+
+ if (!Cluster.ForwardThunkList.empty())
+ Insertions.push_back({Cluster.LastFunctionIndex + 1,
+ /*IsForward=*/true, &Cluster.ForwardThunkList});
+ }
+
+ // Apply insertions from high to low indices so earlier insertions do not
+ // invalidate later positions. At a shared boundary, repeated insertion at the
+ // same index reverses application order, yielding forward thunks before
+ // backward thunks.
+ llvm::sort(Insertions, [](const ThunkInsertion &A, const ThunkInsertion &B) {
+ if (A.Position != B.Position)
+ return A.Position > B.Position;
+ return !A.IsForward && B.IsForward;
+ });
+
+ for (ThunkInsertion &Insertion : Insertions) {
OutputFunctions.insert(
- std::next(OutputFunctions.begin(), FC.LastFunctionIndex + 1),
- FC.ThunkList.begin(), FC.ThunkList.end());
+ std::next(OutputFunctions.begin(), Insertion.Position),
+ Insertion.ThunkList->begin(), Insertion.ThunkList->end());
}
+}
+
+void LongJmpPass::relaxWithClusters(BinaryContext &BC) {
+ BinaryFunctionListType OutputFunctions = BC.getOutputBinaryFunctions();
+ FragmentClusterLayout Layout = buildClusterLayout(BC, OutputFunctions);
+
+ if (Layout.Clusters.empty())
+ return;
+
+ relaxCalls(BC, OutputFunctions, Layout);
+ relaxUnconditionalBranches(BC, OutputFunctions, Layout);
+ insertClusterThunks(OutputFunctions, Layout);
LLVM_DEBUG(dbgs() << "\nFunction layout with thunks:\n";
for (const auto *BF : OutputFunctions) { dbgs() << *BF << '\n'; });
@@ -1375,7 +1512,7 @@ Error LongJmpPass::runOnFunctions(BinaryContext &BC) {
return Error::success();
BC.outs() << "BOLT-INFO: starting experimental relaxation pass\n";
- relaxCalls(BC);
+ relaxWithClusters(BC);
return Error::success();
}
diff --git a/bolt/lib/Rewrite/RewriteInstance.cpp b/bolt/lib/Rewrite/RewriteInstance.cpp
index 512af53f0ec59..1d1728b868bdf 100644
--- a/bolt/lib/Rewrite/RewriteInstance.cpp
+++ b/bolt/lib/Rewrite/RewriteInstance.cpp
@@ -4363,111 +4363,17 @@ void RewriteInstance::mapFileSections(BOLTLinker::SectionMapper MapSection) {
}
}
-namespace {
-
-/// Defines the strict weak ordering for BOLT-produced code sections.
-class CodeSectionOrder {
-public:
- CodeSectionOrder(StringRef ColdSectionName, StringRef HotTextMoverSectionName,
- StringRef MainSectionName, StringRef WarmSectionName,
- bool HotText, bool HotFunctionsAtEnd)
- : ColdSectionName(ColdSectionName),
- HotTextMoverSectionName(HotTextMoverSectionName),
- MainSectionName(MainSectionName), WarmSectionName(WarmSectionName),
- HotText(HotText), HotFunctionsAtEnd(HotFunctionsAtEnd) {}
-
- bool operator()(StringRef AName, StringRef BName) const {
- const SectionKind AKind = getKind(AName);
- const SectionKind BKind = getKind(BName);
- const unsigned ARank = getRank(AKind);
- const unsigned BRank = getRank(BKind);
- if (ARank != BRank)
- return ARank < BRank;
-
- if (AKind == SectionKind::Cold) {
- if (AName.size() != BName.size())
- return HotFunctionsAtEnd ? AName.size() > BName.size()
- : AName.size() < BName.size();
- if (AName != BName)
- return HotFunctionsAtEnd ? AName > BName : AName < BName;
- }
-
- return false;
- }
-
-private:
- enum class SectionKind { Mover, Main, Warm, Cold, Other };
-
- SectionKind getKind(StringRef Name) const {
- if (HotText && Name == HotTextMoverSectionName)
- return SectionKind::Mover;
- if (Name == MainSectionName)
- return SectionKind::Main;
- if (Name == WarmSectionName)
- return SectionKind::Warm;
- if (Name.starts_with(ColdSectionName))
- return SectionKind::Cold;
- return SectionKind::Other;
- }
-
- unsigned getRank(SectionKind Kind) const {
- if (Kind == SectionKind::Mover)
- return 0;
- if (HotFunctionsAtEnd) {
- switch (Kind) {
- case SectionKind::Other:
- return 1;
- case SectionKind::Cold:
- return 2;
- case SectionKind::Warm:
- return 3;
- case SectionKind::Main:
- return 4;
- case SectionKind::Mover:
- llvm_unreachable("handled above");
- }
- }
- switch (Kind) {
- case SectionKind::Main:
- return 1;
- case SectionKind::Warm:
- return 2;
- case SectionKind::Cold:
- return 3;
- case SectionKind::Other:
- return 4;
- case SectionKind::Mover:
- llvm_unreachable("handled above");
- }
- llvm_unreachable("unknown section kind");
- }
-
- StringRef ColdSectionName;
- StringRef HotTextMoverSectionName;
- StringRef MainSectionName;
- StringRef WarmSectionName;
- bool HotText;
- bool HotFunctionsAtEnd;
-};
-
-} // namespace
-
std::vector<BinarySection *> RewriteInstance::getCodeSections() {
std::vector<BinarySection *> CodeSections;
for (BinarySection &Section : BC->textSections())
if (Section.hasValidSectionID())
CodeSections.emplace_back(&Section);
- const CodeSectionOrder CompareSections(
- BC->getColdCodeSectionName(), BC->getHotTextMoverSectionName(),
- BC->getMainCodeSectionName(), BC->getWarmCodeSectionName(), opts::HotText,
- opts::HotFunctionsAtEnd);
-
// Determine the order of sections.
- llvm::stable_sort(CodeSections,
- [&](const BinarySection *A, const BinarySection *B) {
- return CompareSections(A->getName(), B->getName());
- });
+ llvm::stable_sort(
+ CodeSections, [&](const BinarySection *A, const BinarySection *B) {
+ return BC->compareSectionNames(A->getName(), B->getName());
+ });
#ifndef NDEBUG
// Verify that the order of sections and functions is consistent.
@@ -4477,7 +4383,7 @@ std::vector<BinarySection *> RewriteInstance::getCodeSections() {
uint32_t LastIndex = 0;
for (const BinaryFunction *BF : BC->getOutputBinaryFunctions()) {
- if (!BF->isEmitted() || BF->isPatch())
+ if (!BF->isEmitted() || BF->isPatch() || BF->isThunk())
continue;
ErrorOr<BinarySection &> Sec = BF->getCodeSection();
diff --git a/bolt/test/AArch64/relax-branches-with-thunk-chain.s b/bolt/test/AArch64/relax-branches-with-thunk-chain.s
new file mode 100644
index 0000000000000..6c46afdb1c40a
--- /dev/null
+++ b/bolt/test/AArch64/relax-branches-with-thunk-chain.s
@@ -0,0 +1,309 @@
+## Check branch thunk chains with adjacent fragment clusters. Split functions
+## have small duplicated constant islands, while hot-only pad functions create
+## most of the hot-section distance.
+##
+## Input layout:
+##
+## A 8MiB island, pad_hot_0 50MiB, B 8MiB island, pad_hot_1 50MiB,
+## C 8MiB island, pad_hot_2 50MiB, D 8MiB island, pad_hot_3 50MiB
+##
+## With --split-functions, BOLT places the cold blocks of A/B/C/D in
+## .text.cold. With the default 120MiB function-fragment cluster size, branch
+## relaxation sees:
+##
+## normal layout:
+## cluster 0, .text: A, pad_hot_0, B, pad_hot_1
+## cluster 1, .text: C, pad_hot_2, D, pad_hot_3
+## cluster 2, .text.cold: A.cold, B.cold, C.cold, D.cold
+##
+## --hot-functions-at-end:
+## cluster 0, .text.cold: A.cold, B.cold, C.cold, D.cold
+## cluster 1, .text: A, pad_hot_0, B, pad_hot_1
+## cluster 2, .text: C, pad_hot_2, D, pad_hot_3
+
+# REQUIRES: system-linux
+
+# RUN: %clang %cflags -Wl,-q -Wl,-e,A %s -o %t -nostdlib
+# RUN: link_fdata --no-lbr %s %t %t.fdata
+# RUN: llvm-strip --strip-unneeded %t
+# RUN: llvm-bolt %t -o %t.bolt --data %t.fdata --split-functions \
+# RUN: --compact-code-model --relax-exp \
+# RUN: | FileCheck %s --check-prefix=CHECK-BOLT
+# RUN: llvm-bolt %t -o %t.hfe.bolt --data %t.fdata --split-functions \
+# RUN: --compact-code-model --relax-exp --hot-functions-at-end \
+# RUN: | FileCheck %s --check-prefix=CHECK-BOLT-HFE
+# RUN: llvm-readelf -S %t.bolt | FileCheck %s --check-prefix=CHECK-SECTIONS
+# RUN: llvm-objdump -d \
+# RUN: --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64BranchForwardThunk_0,__AArch64BranchForwardThunk_1,__AArch64BranchForwardThunk_4,__AArch64BranchForwardThunk_5,__AArch64BranchForwardThunk_8,__AArch64BranchForwardThunk_10,__AArch64BranchBackwardThunk_2,__AArch64BranchBackwardThunk_3,__AArch64BranchBackwardThunk_6,__AArch64BranchBackwardThunk_7,__AArch64BranchBackwardThunk_9,__AArch64BranchBackwardThunk_11 \
+# RUN: %t.bolt | FileCheck %s --check-prefix=CHECK-OUTPUT
+# RUN: llvm-objdump -d \
+# RUN: --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64BranchForwardThunk_1,__AArch64BranchForwardThunk_3,__AArch64BranchForwardThunk_6,__AArch64BranchForwardThunk_7,__AArch64BranchForwardThunk_10,__AArch64BranchForwardThunk_11,__AArch64BranchBackwardThunk_0,__AArch64BranchBackwardThunk_2,__AArch64BranchBackwardThunk_4,__AArch64BranchBackwardThunk_5,__AArch64BranchBackwardThunk_8,__AArch64BranchBackwardThunk_9 \
+# RUN: %t.hfe.bolt | FileCheck %s --check-prefix=CHECK-HFE-OUTPUT
+
+# CHECK-BOLT: BOLT-INFO: built 3 function fragment cluster(s)
+# CHECK-BOLT: BOLT-INFO: relaxed 8 cross-cluster branches
+# CHECK-BOLT: BOLT-INFO: 12 branch thunks created
+
+# CHECK-BOLT-HFE: BOLT-INFO: built 3 function fragment cluster(s)
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 8 cross-cluster branches
+# CHECK-BOLT-HFE: BOLT-INFO: 12 branch thunks created
+
+# CHECK-SECTIONS: .text
+# CHECK-SECTIONS: .text.cold
+
+ .text
+ .globl A
+ .type A, %function
+A:
+.A_entry:
+# FDATA: 1 A #.A_entry# 100
+ cbz x0, .A_ret
+ b .A_cold
+.A_ret:
+# FDATA: 1 A #.A_ret# 100
+ ret
+.A_cold:
+ mov x0, #1
+ b .A_ret
+ .space 0x800000
+ .size A, .-A
+
+ .globl pad_hot_0
+ .type pad_hot_0, %function
+pad_hot_0:
+.pad_hot_0_entry:
+# FDATA: 1 pad_hot_0 #.pad_hot_0_entry# 100
+ ret
+ .space 0x3200000
+ .size pad_hot_0, .-pad_hot_0
+
+ .globl B
+ .type B, %function
+B:
+.B_entry:
+# FDATA: 1 B #.B_entry# 100
+ cbz x0, .B_ret
+ b .B_cold
+.B_ret:
+# FDATA: 1 B #.B_ret# 100
+ ret
+.B_cold:
+ mov x0, #2
+ b .B_ret
+ .space 0x800000
+ .size B, .-B
+
+ .globl pad_hot_1
+ .type pad_hot_1, %function
+pad_hot_1:
+.pad_hot_1_entry:
+# FDATA: 1 pad_hot_1 #.pad_hot_1_entry# 100
+ ret
+ .space 0x3200000
+ .size pad_hot_1, .-pad_hot_1
+
+ .globl C
+ .type C, %function
+C:
+.C_entry:
+# FDATA: 1 C #.C_entry# 100
+ cbz x0, .C_ret
+ b .C_cold
+.C_ret:
+# FDATA: 1 C #.C_ret# 100
+ ret
+.C_cold:
+ mov x0, #3
+ b .C_ret
+ .space 0x800000
+ .size C, .-C
+
+ .globl pad_hot_2
+ .type pad_hot_2, %function
+pad_hot_2:
+.pad_hot_2_entry:
+# FDATA: 1 pad_hot_2 #.pad_hot_2_entry# 100
+ ret
+ .space 0x3200000
+ .size pad_hot_2, .-pad_hot_2
+
+ .globl D
+ .type D, %function
+D:
+.D_entry:
+# FDATA: 1 D #.D_entry# 100
+ cbz x0, .D_ret
+ b .D_cold
+.D_ret:
+# FDATA: 1 D #.D_ret# 100
+ ret
+.D_cold:
+ mov x0, #4
+ b .D_ret
+ .space 0x800000
+ .size D, .-D
+
+ .globl pad_hot_3
+ .type pad_hot_3, %function
+pad_hot_3:
+.pad_hot_3_entry:
+# FDATA: 1 pad_hot_3 #.pad_hot_3_entry# 100
+ ret
+ .space 0x3200000
+ .size pad_hot_3, .-pad_hot_3
+
+## Force relocation mode.
+ .reloc 0, R_AARCH64_NONE
+
+# CHECK-OUTPUT: Disassembly of section .text:
+
+# CHECK-OUTPUT: <A>:
+# CHECK-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[A_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-OUTPUT-NEXT: [[A_RET:[0-9a-f]+]]: {{.*}} ret
+# CHECK-OUTPUT-NEXT: [[A_BR]]: {{.*}} b 0x[[A_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_1>
+
+# CHECK-OUTPUT: <B>:
+# CHECK-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[B_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-OUTPUT-NEXT: [[B_RET:[0-9a-f]+]]: {{.*}} ret
+# CHECK-OUTPUT-NEXT: [[B_BR]]: {{.*}} b 0x[[B_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_5>
+
+# CHECK-OUTPUT: <__AArch64BranchForwardThunk_1>:
+# CHECK-OUTPUT-NEXT: [[A_FW0]]: {{.*}} b 0x[[A_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_0>
+
+# CHECK-OUTPUT: <__AArch64BranchForwardThunk_5>:
+# CHECK-OUTPUT-NEXT: [[B_FW0]]: {{.*}} b 0x[[B_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_4>
+
+# CHECK-OUTPUT: <__AArch64BranchBackwardThunk_2>:
+# CHECK-OUTPUT-NEXT: [[A_BW1:[0-9a-f]+]]: {{.*}} b 0x[[A_RET]] <A+0x4>
+
+# CHECK-OUTPUT: <__AArch64BranchBackwardThunk_6>:
+# CHECK-OUTPUT-NEXT: [[B_BW1:[0-9a-f]+]]: {{.*}} b 0x[[B_RET]] <B+0x4>
+
+# CHECK-OUTPUT: <C>:
+# CHECK-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[C_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-OUTPUT-NEXT: [[C_RET:[0-9a-f]+]]: {{.*}} ret
+# CHECK-OUTPUT-NEXT: [[C_BR]]: {{.*}} b 0x[[C_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_8>
+
+# CHECK-OUTPUT: <D>:
+# CHECK-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[D_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-OUTPUT-NEXT: [[D_RET:[0-9a-f]+]]: {{.*}} ret
+# CHECK-OUTPUT-NEXT: [[D_BR]]: {{.*}} b 0x[[D_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_10>
+
+# CHECK-OUTPUT: <__AArch64BranchForwardThunk_0>:
+# CHECK-OUTPUT-NEXT: [[A_FW1]]: {{.*}} b 0x[[A_COLD:[0-9a-f]+]] <A.cold.0>
+
+# CHECK-OUTPUT: <__AArch64BranchForwardThunk_4>:
+# CHECK-OUTPUT-NEXT: [[B_FW1]]: {{.*}} b 0x[[B_COLD:[0-9a-f]+]] <B.cold.0>
+
+# CHECK-OUTPUT: <__AArch64BranchForwardThunk_8>:
+# CHECK-OUTPUT-NEXT: [[C_FW0]]: {{.*}} b 0x[[C_COLD:[0-9a-f]+]] <C.cold.0>
+
+# CHECK-OUTPUT: <__AArch64BranchForwardThunk_10>:
+# CHECK-OUTPUT-NEXT: [[D_FW0]]: {{.*}} b 0x[[D_COLD:[0-9a-f]+]] <D.cold.0>
+
+# CHECK-OUTPUT: Disassembly of section .text.cold:
+
+# CHECK-OUTPUT: <__AArch64BranchBackwardThunk_3>:
+# CHECK-OUTPUT-NEXT: [[A_BW0:[0-9a-f]+]]: {{.*}} b 0x[[A_BW1]] <__AArch64BranchBackwardThunk_2>
+
+# CHECK-OUTPUT: <__AArch64BranchBackwardThunk_7>:
+# CHECK-OUTPUT-NEXT: [[B_BW0:[0-9a-f]+]]: {{.*}} b 0x[[B_BW1]] <__AArch64BranchBackwardThunk_6>
+
+# CHECK-OUTPUT: <__AArch64BranchBackwardThunk_9>:
+# CHECK-OUTPUT-NEXT: [[C_BW0:[0-9a-f]+]]: {{.*}} b 0x[[C_RET]] <C+0x4>
+
+# CHECK-OUTPUT: <__AArch64BranchBackwardThunk_11>:
+# CHECK-OUTPUT-NEXT: [[D_BW0:[0-9a-f]+]]: {{.*}} b 0x[[D_RET]] <D+0x4>
+
+# CHECK-OUTPUT: <A.cold.0>:
+# CHECK-OUTPUT-NEXT: [[A_COLD]]: {{.*}} mov x0, #0x1
+# CHECK-OUTPUT-NEXT: {{.*}} b 0x[[A_BW0]] <__AArch64BranchBackwardThunk_3>
+
+# CHECK-OUTPUT: <B.cold.0>:
+# CHECK-OUTPUT-NEXT: [[B_COLD]]: {{.*}} mov x0, #0x2
+# CHECK-OUTPUT-NEXT: {{.*}} b 0x[[B_BW0]] <__AArch64BranchBackwardThunk_7>
+
+# CHECK-OUTPUT: <C.cold.0>:
+# CHECK-OUTPUT-NEXT: [[C_COLD]]: {{.*}} mov x0, #0x3
+# CHECK-OUTPUT-NEXT: {{.*}} b 0x[[C_BW0]] <__AArch64BranchBackwardThunk_9>
+
+# CHECK-OUTPUT: <D.cold.0>:
+# CHECK-OUTPUT-NEXT: [[D_COLD]]: {{.*}} mov x0, #0x4
+# CHECK-OUTPUT-NEXT: {{.*}} b 0x[[D_BW0]] <__AArch64BranchBackwardThunk_11>
+
+
+# CHECK-HFE-OUTPUT: Disassembly of section .text.cold:
+
+# CHECK-HFE-OUTPUT: <A.cold.0>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x1
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_A_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_1>
+
+# CHECK-HFE-OUTPUT: <B.cold.0>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x2
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_B_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_3>
+
+# CHECK-HFE-OUTPUT: <C.cold.0>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x3
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_C_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_7>
+
+# CHECK-HFE-OUTPUT: <D.cold.0>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x4
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_D_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_11>
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_1>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_FW]]: {{.*}} b 0x[[HFE_A_RET:[0-9a-f]+]] <A+0x4>
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_3>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_FW]]: {{.*}} b 0x[[HFE_B_RET:[0-9a-f]+]] <B+0x4>
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_7>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_FW0]]: {{.*}} b 0x[[HFE_C_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_6>
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_11>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_FW0]]: {{.*}} b 0x[[HFE_D_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_10>
+
+# CHECK-HFE-OUTPUT: Disassembly of section .text:
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_0>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_A_COLD]] <A.cold.0>
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_2>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_B_COLD]] <B.cold.0>
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_4>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BW1:[0-9a-f]+]]: {{.*}} b 0x[[HFE_C_COLD]] <C.cold.0>
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_8>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BW1:[0-9a-f]+]]: {{.*}} b 0x[[HFE_D_COLD]] <D.cold.0>
+
+# CHECK-HFE-OUTPUT: <A>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[HFE_A_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_RET]]: {{.*}} ret
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_BR]]: {{.*}} b 0x{{[0-9a-f]+}} <__AArch64BranchBackwardThunk_0>
+
+# CHECK-HFE-OUTPUT: <B>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[HFE_B_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_RET]]: {{.*}} ret
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_BR]]: {{.*}} b 0x{{[0-9a-f]+}} <__AArch64BranchBackwardThunk_2>
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_6>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_FW1]]: {{.*}} b 0x[[HFE_C_RET:[0-9a-f]+]] <C+0x4>
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_10>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_FW1]]: {{.*}} b 0x[[HFE_D_RET:[0-9a-f]+]] <D+0x4>
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_5>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BW0:[0-9a-f]+]]: {{.*}} b 0x[[HFE_C_BW1]] <__AArch64BranchBackwardThunk_4>
+
+# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_9>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BW0:[0-9a-f]+]]: {{.*}} b 0x[[HFE_D_BW1]] <__AArch64BranchBackwardThunk_8>
+
+# CHECK-HFE-OUTPUT: <C>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[HFE_C_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_RET]]: {{.*}} ret
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BR]]: {{.*}} b 0x[[HFE_C_BW0]] <__AArch64BranchBackwardThunk_5>
+
+# CHECK-HFE-OUTPUT: <D>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[HFE_D_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_RET]]: {{.*}} ret
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BR]]: {{.*}} b 0x[[HFE_D_BW0]] <__AArch64BranchBackwardThunk_9>
diff --git a/bolt/test/AArch64/relax-exp.s b/bolt/test/AArch64/relax-calls.s
similarity index 86%
rename from bolt/test/AArch64/relax-exp.s
rename to bolt/test/AArch64/relax-calls.s
index f6394b7a39df7..3f2b0375e304b 100644
--- a/bolt/test/AArch64/relax-exp.s
+++ b/bolt/test/AArch64/relax-calls.s
@@ -55,17 +55,17 @@ hot:
# CHECK-BOLT-LITE: BOLT-INFO: 3 long thunks created
## Check the number of thunks created in other modes.
-# CHECK-BOLT: BOLT-INFO: 4 short thunks created
-# CHECK-BOLT: BOLT-INFO: 3 long thunks created
+# CHECK-BOLT: BOLT-INFO: 5 short thunks created
+# CHECK-BOLT: BOLT-INFO: 5 long thunks created
-# CHECK-BOLT-HOT-END: BOLT-INFO: 4 short thunks created
-# CHECK-BOLT-HOT-END: BOLT-INFO: 2 long thunks created
+# CHECK-BOLT-HOT-END: BOLT-INFO: 6 short thunks created
+# CHECK-BOLT-HOT-END: BOLT-INFO: 4 long thunks created
## Check that correct veneers are used depending on the target proximity.
# CHECK-OUTPUT-LABEL: <hot>:
# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64ADRPThunk_foo>
-# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64Thunk_bar>
-# CHECK-OUTPUT-NEXT: bl {{.*}} <_start>
+# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64ADRPThunk_bar>
+# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64Thunk__start>
.global _start
.type _start, %function
diff --git a/bolt/test/AArch64/relax-cross-fragment-calls.s b/bolt/test/AArch64/relax-cross-fragment-calls.s
new file mode 100644
index 0000000000000..f99d30ee94f8d
--- /dev/null
+++ b/bolt/test/AArch64/relax-cross-fragment-calls.s
@@ -0,0 +1,216 @@
+## Check call relaxation with function fragment clusters. This uses split
+## functions with calls from hot and cold fragments so call thunks are placed
+## around the clusters.
+##
+## Input layout:
+##
+## A 8MiB island, pad_hot_0 50MiB, B 8MiB island, pad_hot_1 50MiB,
+## C 8MiB island, pad_hot_2 50MiB, D 8MiB island, pad_hot_3 50MiB
+##
+## With --split-functions, BOLT places the cold blocks of A/B/C/D in
+## .text.cold. With the default 120MiB function-fragment cluster size, call
+## relaxation sees:
+##
+## normal layout:
+## cluster 0, .text: A, pad_hot_0, B, pad_hot_1
+## cluster 1, .text: C, pad_hot_2, D, pad_hot_3
+## cluster 2, .text.cold: A.cold, B.cold, C.cold, D.cold
+##
+## --hot-functions-at-end:
+## cluster 0, .text.cold: A.cold, B.cold, C.cold, D.cold
+## cluster 1, .text: A, pad_hot_0, B, pad_hot_1
+## cluster 2, .text: C, pad_hot_2, D, pad_hot_3
+
+# REQUIRES: system-linux
+
+# RUN: %clang %cflags -Wl,-q -Wl,-e,A %s -o %t -nostdlib
+# RUN: link_fdata --no-lbr %s %t %t.fdata
+# RUN: llvm-strip --strip-unneeded %t
+# RUN: llvm-bolt %t -o %t.bolt --data %t.fdata --split-functions \
+# RUN: --compact-code-model --relax-exp \
+# RUN: | FileCheck %s --check-prefix=CHECK-BOLT
+# RUN: llvm-bolt %t -o %t.hfe.bolt --data %t.fdata --split-functions \
+# RUN: --compact-code-model --relax-exp --hot-functions-at-end \
+# RUN: | FileCheck %s --check-prefix=CHECK-BOLT-HFE
+# RUN: llvm-readelf -S %t.bolt | FileCheck %s --check-prefix=CHECK-SECTIONS
+# RUN: llvm-objdump -d \
+# RUN: --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64Thunk_A,__AArch64Thunk_C,__AArch64ADRPThunk_A \
+# RUN: %t.bolt | FileCheck %s --check-prefix=CHECK-OUTPUT
+# RUN: llvm-objdump -d \
+# RUN: --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64Thunk_A,__AArch64Thunk_C,__AArch64ADRPThunk_C \
+# RUN: %t.hfe.bolt | FileCheck %s --check-prefix=CHECK-HFE-OUTPUT
+
+# CHECK-BOLT: BOLT-INFO: built 3 function fragment cluster(s)
+# CHECK-BOLT: BOLT-INFO: relaxed 4 calls with thunks
+# CHECK-BOLT: BOLT-INFO: 3 short thunks created
+# CHECK-BOLT: BOLT-INFO: 1 long thunks created
+# CHECK-BOLT: BOLT-INFO: relaxed 8 cross-cluster branches
+# CHECK-BOLT: BOLT-INFO: 12 branch thunks created
+
+# CHECK-BOLT-HFE: BOLT-INFO: built 3 function fragment cluster(s)
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 4 calls with thunks
+# CHECK-BOLT-HFE: BOLT-INFO: 3 short thunks created
+# CHECK-BOLT-HFE: BOLT-INFO: 1 long thunks created
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 8 cross-cluster branches
+# CHECK-BOLT-HFE: BOLT-INFO: 12 branch thunks created
+
+# CHECK-SECTIONS: .text
+# CHECK-SECTIONS: .text.cold
+
+ .text
+ .globl A
+ .type A, %function
+A:
+.A_entry:
+# FDATA: 1 A #.A_entry# 100
+ bl B
+ bl C
+ cbz x0, .A_ret
+ b .A_cold
+.A_ret:
+# FDATA: 1 A #.A_ret# 100
+ ret
+.A_cold:
+ mov x0, #1
+ bl C
+ bl A
+ b .A_ret
+ .space 0x800000
+ .size A, .-A
+
+ .globl pad_hot_0
+ .type pad_hot_0, %function
+pad_hot_0:
+.pad_hot_0_entry:
+# FDATA: 1 pad_hot_0 #.pad_hot_0_entry# 100
+ ret
+ .space 0x3200000
+ .size pad_hot_0, .-pad_hot_0
+
+ .globl B
+ .type B, %function
+B:
+.B_entry:
+# FDATA: 1 B #.B_entry# 100
+ cbz x0, .B_ret
+ b .B_cold
+.B_ret:
+# FDATA: 1 B #.B_ret# 100
+ ret
+.B_cold:
+ mov x0, #2
+ b .B_ret
+ .space 0x800000
+ .size B, .-B
+
+ .globl pad_hot_1
+ .type pad_hot_1, %function
+pad_hot_1:
+.pad_hot_1_entry:
+# FDATA: 1 pad_hot_1 #.pad_hot_1_entry# 100
+ ret
+ .space 0x3200000
+ .size pad_hot_1, .-pad_hot_1
+
+ .globl C
+ .type C, %function
+C:
+.C_entry:
+# FDATA: 1 C #.C_entry# 100
+ bl A
+ cbz x0, .C_ret
+ b .C_cold
+.C_ret:
+# FDATA: 1 C #.C_ret# 100
+ ret
+.C_cold:
+ mov x0, #3
+ b .C_ret
+ .space 0x800000
+ .size C, .-C
+
+ .globl pad_hot_2
+ .type pad_hot_2, %function
+pad_hot_2:
+.pad_hot_2_entry:
+# FDATA: 1 pad_hot_2 #.pad_hot_2_entry# 100
+ ret
+ .space 0x3200000
+ .size pad_hot_2, .-pad_hot_2
+
+ .globl D
+ .type D, %function
+D:
+.D_entry:
+# FDATA: 1 D #.D_entry# 100
+ cbz x0, .D_ret
+ b .D_cold
+.D_ret:
+# FDATA: 1 D #.D_ret# 100
+ ret
+.D_cold:
+ mov x0, #4
+ b .D_ret
+ .space 0x800000
+ .size D, .-D
+
+ .globl pad_hot_3
+ .type pad_hot_3, %function
+pad_hot_3:
+.pad_hot_3_entry:
+# FDATA: 1 pad_hot_3 #.pad_hot_3_entry# 100
+ ret
+ .space 0x3200000
+ .size pad_hot_3, .-pad_hot_3
+
+## Force relocation mode.
+ .reloc 0, R_AARCH64_NONE
+
+# CHECK-OUTPUT: <A>:
+# CHECK-OUTPUT-NEXT: {{.*}} bl {{.*}} <B>
+# CHECK-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_C>
+
+# CHECK-OUTPUT: <__AArch64Thunk_C>:
+# CHECK-OUTPUT-NEXT: {{.*}} b {{.*}} <C>
+
+# CHECK-OUTPUT: <__AArch64Thunk_A>:
+# CHECK-OUTPUT-NEXT: {{.*}} b {{.*}} <A>
+
+# CHECK-OUTPUT: <C>:
+# CHECK-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_A>
+
+# CHECK-OUTPUT: <__AArch64ADRPThunk_A>:
+# CHECK-OUTPUT-NEXT: {{.*}} adrp x16, {{.*}} <A>
+# CHECK-OUTPUT-NEXT: {{.*}} add x16, x16, #0x0
+# CHECK-OUTPUT-NEXT: {{.*}} br x16
+
+# CHECK-OUTPUT: <A.cold.0>:
+# CHECK-OUTPUT-NEXT: {{.*}} mov x0, #0x1
+# CHECK-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_C>
+# CHECK-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64ADRPThunk_A>
+
+# CHECK-HFE-OUTPUT: <A.cold.0>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} mov x0, #0x1
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64ADRPThunk_C>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_A>
+
+# CHECK-HFE-OUTPUT: <__AArch64ADRPThunk_C>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} adrp x16, {{.*}}
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} add x16, x16, #0x{{[0-9a-f]+}}
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} br x16
+
+# CHECK-HFE-OUTPUT: <__AArch64Thunk_A>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b {{.*}} <A>
+
+# CHECK-HFE-OUTPUT: <A>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <B>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_C>
+
+# CHECK-HFE-OUTPUT: <__AArch64Thunk_C>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b {{.*}} <C>
+
+# CHECK-HFE-OUTPUT: <__AArch64Thunk_A>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b {{.*}} <A>
+
+# CHECK-HFE-OUTPUT: <C>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_A>
>From 6514e5993db0a696630c50fdc53432f088e3d356 Mon Sep 17 00:00:00 2001
From: Alexandros Lamprineas <alexandros.lamprineas at arm.com>
Date: Thu, 27 Aug 2026 09:00:27 +0000
Subject: [PATCH 2/2] allow clusters to span across sections
---
bolt/include/bolt/Passes/LongJmp.h | 18 +++--
bolt/lib/Passes/LongJmp.cpp | 15 ++--
.../AArch64/relax-branches-with-thunk-chain.s | 73 +++++++------------
bolt/test/AArch64/relax-calls.s | 14 ++--
.../test/AArch64/relax-cross-fragment-calls.s | 34 ++++-----
5 files changed, 65 insertions(+), 89 deletions(-)
diff --git a/bolt/include/bolt/Passes/LongJmp.h b/bolt/include/bolt/Passes/LongJmp.h
index 37c7e1452d5bf..e021e8fad503b 100644
--- a/bolt/include/bolt/Passes/LongJmp.h
+++ b/bolt/include/bolt/Passes/LongJmp.h
@@ -84,10 +84,13 @@ class LongJmpPass : public BinaryFunctionPass {
/// A group of function fragments that are located within the longest direct
/// branch/call instruction distance. Jumps within the cluster do not require
- /// a thunk. The cluster may include thunks for jumps to targets outside.
+ /// a thunk. The cluster may span output sections and include thunks for jumps
+ /// to targets outside. Backward thunks are inserted before the cluster, while
+ /// forward thunks are inserted after it.
struct FragmentCluster {
- /// Output code section containing fragments in this cluster.
- SmallString<32> SectionName;
+ /// Output code sections containing the first and last cluster fragments.
+ SmallString<32> StartSectionName;
+ SmallString<32> EndSectionName;
/// Estimated size of the cluster in bytes.
uint64_t Size{0};
@@ -115,12 +118,11 @@ class LongJmpPass : public BinaryFunctionPass {
/// <Destination Symbol> -> <Thunk Function>.
DenseMap<const MCSymbol *, BinaryFunction *> ForwardBranchThunks;
DenseMap<const MCSymbol *, BinaryFunction *> BackwardBranchThunks;
- };
- /// Maximum size of combined regular functions in the cluster. Note that it's
- /// less than 128MB, because the size of the cluster plus its thunks should be
- /// less than 128MB.
- static constexpr uint64_t MaxClusterSize = 120 * 1024 * 1024;
+ StringRef getThunkSectionName(bool IsForward) const {
+ return IsForward ? EndSectionName : StartSectionName;
+ }
+ };
struct FragmentClusterLayout {
SmallVector<FragmentCluster, 4> Clusters;
diff --git a/bolt/lib/Passes/LongJmp.cpp b/bolt/lib/Passes/LongJmp.cpp
index ccdb2ef37f77f..3a0dda76e0b08 100644
--- a/bolt/lib/Passes/LongJmp.cpp
+++ b/bolt/lib/Passes/LongJmp.cpp
@@ -40,6 +40,11 @@ static cl::opt<bool>
static cl::opt<bool> RelaxPLT("relax-plt",
cl::desc("indicate PLT proximity to hot text"),
cl::init(true), cl::cat(BoltOptCategory));
+
+static cl::opt<unsigned long long> MaxClusterSize(
+ "max-cluster-size",
+ cl::desc("maximum estimated size of a function fragment cluster in bytes"),
+ cl::init(124 * 1024 * 1024), cl::cat(BoltOptCategory));
}
namespace llvm {
@@ -1079,15 +1084,15 @@ LongJmpPass::buildClusterLayout(BinaryContext &BC,
const uint64_t FFSize = estimateFragmentSize(BF, FF);
if (Layout.Clusters.empty() ||
- Layout.Clusters.back().SectionName != Fragment.SectionName ||
- Layout.Clusters.back().Size + FFSize > MaxClusterSize) {
+ Layout.Clusters.back().Size + FFSize > opts::MaxClusterSize) {
Layout.Clusters.emplace_back(FragmentCluster());
FragmentCluster &FC = Layout.Clusters.back();
- FC.SectionName = Fragment.SectionName;
+ FC.StartSectionName = Fragment.SectionName;
FC.FirstFunctionIndex = Fragment.FunctionIndex;
}
FragmentCluster &FC = Layout.Clusters.back();
+ FC.EndSectionName = Fragment.SectionName;
FC.LastFunctionIndex = Fragment.FunctionIndex;
++FC.NumFragments;
EstimatedSize += FFSize;
@@ -1249,7 +1254,7 @@ void LongJmpPass::relaxCalls(BinaryContext &BC,
BinaryFunction *Thunk = createCallThunk(TargetSymbol, IsShort);
FragmentCluster &ThunkCluster = Clusters[SourceCluster];
- Thunk->setCodeSectionName(ThunkCluster.SectionName);
+ Thunk->setCodeSectionName(ThunkCluster.getThunkSectionName(IsForward));
BinaryFunctionListType &ThunkList = IsForward
? ThunkCluster.ForwardThunkList
: ThunkCluster.BackwardThunkList;
@@ -1355,7 +1360,7 @@ void LongJmpPass::relaxUnconditionalBranches(
return It->second;
BinaryFunction *Thunk = createBranchThunk(NextTarget, IsForward);
- Thunk->setCodeSectionName(Cluster.SectionName);
+ Thunk->setCodeSectionName(Cluster.getThunkSectionName(IsForward));
auto &ThunkList =
IsForward ? Cluster.ForwardThunkList : Cluster.BackwardThunkList;
ThunkList.push_back(Thunk);
diff --git a/bolt/test/AArch64/relax-branches-with-thunk-chain.s b/bolt/test/AArch64/relax-branches-with-thunk-chain.s
index 6c46afdb1c40a..1eddf8160fca9 100644
--- a/bolt/test/AArch64/relax-branches-with-thunk-chain.s
+++ b/bolt/test/AArch64/relax-branches-with-thunk-chain.s
@@ -8,8 +8,8 @@
## C 8MiB island, pad_hot_2 50MiB, D 8MiB island, pad_hot_3 50MiB
##
## With --split-functions, BOLT places the cold blocks of A/B/C/D in
-## .text.cold. With the default 120MiB function-fragment cluster size, branch
-## relaxation sees:
+## .text.cold. With the default 124MiB function-fragment cluster size,
+## branch relaxation sees:
##
## normal layout:
## cluster 0, .text: A, pad_hot_0, B, pad_hot_1
@@ -17,9 +17,10 @@
## cluster 2, .text.cold: A.cold, B.cold, C.cold, D.cold
##
## --hot-functions-at-end:
-## cluster 0, .text.cold: A.cold, B.cold, C.cold, D.cold
-## cluster 1, .text: A, pad_hot_0, B, pad_hot_1
-## cluster 2, .text: C, pad_hot_2, D, pad_hot_3
+## cluster 0: .text.cold A.cold, B.cold, C.cold, D.cold
+## .text A, pad_hot_0, B
+## cluster 1: .text pad_hot_1, C, pad_hot_2, D
+## cluster 2: .text pad_hot_3
# REQUIRES: system-linux
@@ -37,7 +38,7 @@
# RUN: --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64BranchForwardThunk_0,__AArch64BranchForwardThunk_1,__AArch64BranchForwardThunk_4,__AArch64BranchForwardThunk_5,__AArch64BranchForwardThunk_8,__AArch64BranchForwardThunk_10,__AArch64BranchBackwardThunk_2,__AArch64BranchBackwardThunk_3,__AArch64BranchBackwardThunk_6,__AArch64BranchBackwardThunk_7,__AArch64BranchBackwardThunk_9,__AArch64BranchBackwardThunk_11 \
# RUN: %t.bolt | FileCheck %s --check-prefix=CHECK-OUTPUT
# RUN: llvm-objdump -d \
-# RUN: --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64BranchForwardThunk_1,__AArch64BranchForwardThunk_3,__AArch64BranchForwardThunk_6,__AArch64BranchForwardThunk_7,__AArch64BranchForwardThunk_10,__AArch64BranchForwardThunk_11,__AArch64BranchBackwardThunk_0,__AArch64BranchBackwardThunk_2,__AArch64BranchBackwardThunk_4,__AArch64BranchBackwardThunk_5,__AArch64BranchBackwardThunk_8,__AArch64BranchBackwardThunk_9 \
+# RUN: --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64BranchForwardThunk_1,__AArch64BranchForwardThunk_3,__AArch64BranchBackwardThunk_0,__AArch64BranchBackwardThunk_2 \
# RUN: %t.hfe.bolt | FileCheck %s --check-prefix=CHECK-HFE-OUTPUT
# CHECK-BOLT: BOLT-INFO: built 3 function fragment cluster(s)
@@ -45,8 +46,8 @@
# CHECK-BOLT: BOLT-INFO: 12 branch thunks created
# CHECK-BOLT-HFE: BOLT-INFO: built 3 function fragment cluster(s)
-# CHECK-BOLT-HFE: BOLT-INFO: relaxed 8 cross-cluster branches
-# CHECK-BOLT-HFE: BOLT-INFO: 12 branch thunks created
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 4 cross-cluster branches
+# CHECK-BOLT-HFE: BOLT-INFO: 4 branch thunks created
# CHECK-SECTIONS: .text
# CHECK-SECTIONS: .text.cold
@@ -236,74 +237,50 @@ pad_hot_3:
# CHECK-HFE-OUTPUT: <A.cold.0>:
# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x1
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_A_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_1>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_A_RET:[0-9a-f]+]] <A+0x4>
# CHECK-HFE-OUTPUT: <B.cold.0>:
# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x2
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_B_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_3>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_B_RET:[0-9a-f]+]] <B+0x4>
# CHECK-HFE-OUTPUT: <C.cold.0>:
# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x3
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_C_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_7>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_C_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_1>
# CHECK-HFE-OUTPUT: <D.cold.0>:
# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x4
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_D_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_11>
-
-# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_1>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_FW]]: {{.*}} b 0x[[HFE_A_RET:[0-9a-f]+]] <A+0x4>
-
-# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_3>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_FW]]: {{.*}} b 0x[[HFE_B_RET:[0-9a-f]+]] <B+0x4>
-
-# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_7>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_FW0]]: {{.*}} b 0x[[HFE_C_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_6>
-
-# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_11>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_FW0]]: {{.*}} b 0x[[HFE_D_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_10>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_D_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_3>
# CHECK-HFE-OUTPUT: Disassembly of section .text:
-# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_0>:
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_A_COLD]] <A.cold.0>
-
-# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_2>:
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} b 0x[[HFE_B_COLD]] <B.cold.0>
-
-# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_4>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BW1:[0-9a-f]+]]: {{.*}} b 0x[[HFE_C_COLD]] <C.cold.0>
-
-# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_8>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BW1:[0-9a-f]+]]: {{.*}} b 0x[[HFE_D_COLD]] <D.cold.0>
-
# CHECK-HFE-OUTPUT: <A>:
# CHECK-HFE-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[HFE_A_BR:[0-9a-f]+]] <{{.*}}>
# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_RET]]: {{.*}} ret
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_BR]]: {{.*}} b 0x{{[0-9a-f]+}} <__AArch64BranchBackwardThunk_0>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_BR]]: {{.*}} b 0x[[HFE_A_COLD]] <A.cold.0>
# CHECK-HFE-OUTPUT: <B>:
# CHECK-HFE-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[HFE_B_BR:[0-9a-f]+]] <{{.*}}>
# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_RET]]: {{.*}} ret
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_BR]]: {{.*}} b 0x{{[0-9a-f]+}} <__AArch64BranchBackwardThunk_2>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_BR]]: {{.*}} b 0x[[HFE_B_COLD]] <B.cold.0>
-# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_6>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_FW1]]: {{.*}} b 0x[[HFE_C_RET:[0-9a-f]+]] <C+0x4>
+# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_1>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_FW]]: {{.*}} b 0x[[HFE_C_RET:[0-9a-f]+]] <C+0x4>
-# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_10>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_FW1]]: {{.*}} b 0x[[HFE_D_RET:[0-9a-f]+]] <D+0x4>
+# CHECK-HFE-OUTPUT: <__AArch64BranchForwardThunk_3>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_FW]]: {{.*}} b 0x[[HFE_D_RET:[0-9a-f]+]] <D+0x4>
-# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_5>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BW0:[0-9a-f]+]]: {{.*}} b 0x[[HFE_C_BW1]] <__AArch64BranchBackwardThunk_4>
+# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_0>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BW:[0-9a-f]+]]: {{.*}} b 0x[[HFE_C_COLD]] <C.cold.0>
-# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_9>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BW0:[0-9a-f]+]]: {{.*}} b 0x[[HFE_D_BW1]] <__AArch64BranchBackwardThunk_8>
+# CHECK-HFE-OUTPUT: <__AArch64BranchBackwardThunk_2>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BW:[0-9a-f]+]]: {{.*}} b 0x[[HFE_D_COLD]] <D.cold.0>
# CHECK-HFE-OUTPUT: <C>:
# CHECK-HFE-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[HFE_C_BR:[0-9a-f]+]] <{{.*}}>
# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_RET]]: {{.*}} ret
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BR]]: {{.*}} b 0x[[HFE_C_BW0]] <__AArch64BranchBackwardThunk_5>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BR]]: {{.*}} b 0x[[HFE_C_BW]] <__AArch64BranchBackwardThunk_0>
# CHECK-HFE-OUTPUT: <D>:
# CHECK-HFE-OUTPUT-NEXT: {{.*}} cbnz x0, 0x[[HFE_D_BR:[0-9a-f]+]] <{{.*}}>
# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_RET]]: {{.*}} ret
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BR]]: {{.*}} b 0x[[HFE_D_BW0]] <__AArch64BranchBackwardThunk_9>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BR]]: {{.*}} b 0x[[HFE_D_BW]] <__AArch64BranchBackwardThunk_2>
diff --git a/bolt/test/AArch64/relax-calls.s b/bolt/test/AArch64/relax-calls.s
index 3f2b0375e304b..767195f648c4a 100644
--- a/bolt/test/AArch64/relax-calls.s
+++ b/bolt/test/AArch64/relax-calls.s
@@ -55,17 +55,17 @@ hot:
# CHECK-BOLT-LITE: BOLT-INFO: 3 long thunks created
## Check the number of thunks created in other modes.
-# CHECK-BOLT: BOLT-INFO: 5 short thunks created
-# CHECK-BOLT: BOLT-INFO: 5 long thunks created
+# CHECK-BOLT: BOLT-INFO: 4 short thunks created
+# CHECK-BOLT: BOLT-INFO: 3 long thunks created
-# CHECK-BOLT-HOT-END: BOLT-INFO: 6 short thunks created
-# CHECK-BOLT-HOT-END: BOLT-INFO: 4 long thunks created
+# CHECK-BOLT-HOT-END: BOLT-INFO: 4 short thunks created
+# CHECK-BOLT-HOT-END: BOLT-INFO: 2 long thunks created
## Check that correct veneers are used depending on the target proximity.
# CHECK-OUTPUT-LABEL: <hot>:
-# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64ADRPThunk_foo>
-# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64ADRPThunk_bar>
-# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64Thunk__start>
+# CHECK-OUTPUT-NEXT: bl {{.*}} <foo>
+# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64Thunk_bar>
+# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64ADRPThunk__start>
.global _start
.type _start, %function
diff --git a/bolt/test/AArch64/relax-cross-fragment-calls.s b/bolt/test/AArch64/relax-cross-fragment-calls.s
index f99d30ee94f8d..b1522e2db38bc 100644
--- a/bolt/test/AArch64/relax-cross-fragment-calls.s
+++ b/bolt/test/AArch64/relax-cross-fragment-calls.s
@@ -8,8 +8,8 @@
## C 8MiB island, pad_hot_2 50MiB, D 8MiB island, pad_hot_3 50MiB
##
## With --split-functions, BOLT places the cold blocks of A/B/C/D in
-## .text.cold. With the default 120MiB function-fragment cluster size, call
-## relaxation sees:
+## .text.cold. With the default 124MiB function-fragment cluster size,
+## call relaxation sees:
##
## normal layout:
## cluster 0, .text: A, pad_hot_0, B, pad_hot_1
@@ -17,9 +17,10 @@
## cluster 2, .text.cold: A.cold, B.cold, C.cold, D.cold
##
## --hot-functions-at-end:
-## cluster 0, .text.cold: A.cold, B.cold, C.cold, D.cold
-## cluster 1, .text: A, pad_hot_0, B, pad_hot_1
-## cluster 2, .text: C, pad_hot_2, D, pad_hot_3
+## cluster 0: .text.cold A.cold, B.cold, C.cold, D.cold
+## .text A, pad_hot_0, B
+## cluster 1: .text pad_hot_1, C, pad_hot_2, D
+## cluster 2: .text pad_hot_3
# REQUIRES: system-linux
@@ -37,7 +38,7 @@
# RUN: --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64Thunk_A,__AArch64Thunk_C,__AArch64ADRPThunk_A \
# RUN: %t.bolt | FileCheck %s --check-prefix=CHECK-OUTPUT
# RUN: llvm-objdump -d \
-# RUN: --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64Thunk_A,__AArch64Thunk_C,__AArch64ADRPThunk_C \
+# RUN: --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64Thunk_A,__AArch64Thunk_C \
# RUN: %t.hfe.bolt | FileCheck %s --check-prefix=CHECK-HFE-OUTPUT
# CHECK-BOLT: BOLT-INFO: built 3 function fragment cluster(s)
@@ -48,11 +49,10 @@
# CHECK-BOLT: BOLT-INFO: 12 branch thunks created
# CHECK-BOLT-HFE: BOLT-INFO: built 3 function fragment cluster(s)
-# CHECK-BOLT-HFE: BOLT-INFO: relaxed 4 calls with thunks
-# CHECK-BOLT-HFE: BOLT-INFO: 3 short thunks created
-# CHECK-BOLT-HFE: BOLT-INFO: 1 long thunks created
-# CHECK-BOLT-HFE: BOLT-INFO: relaxed 8 cross-cluster branches
-# CHECK-BOLT-HFE: BOLT-INFO: 12 branch thunks created
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 3 calls with thunks
+# CHECK-BOLT-HFE: BOLT-INFO: 2 short thunks created
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 4 cross-cluster branches
+# CHECK-BOLT-HFE: BOLT-INFO: 4 branch thunks created
# CHECK-SECTIONS: .text
# CHECK-SECTIONS: .text.cold
@@ -191,16 +191,8 @@ pad_hot_3:
# CHECK-HFE-OUTPUT: <A.cold.0>:
# CHECK-HFE-OUTPUT-NEXT: {{.*}} mov x0, #0x1
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64ADRPThunk_C>
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_A>
-
-# CHECK-HFE-OUTPUT: <__AArch64ADRPThunk_C>:
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} adrp x16, {{.*}}
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} add x16, x16, #0x{{[0-9a-f]+}}
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} br x16
-
-# CHECK-HFE-OUTPUT: <__AArch64Thunk_A>:
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} b {{.*}} <A>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_C>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <A>
# CHECK-HFE-OUTPUT: <A>:
# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <B>
More information about the llvm-commits
mailing list