[llvm] [BOLT][AArch64] Relax calls and branches with fragment clusters (PR #215825)
Alexandros Lamprineas via llvm-commits
llvm-commits at lists.llvm.org
Tue Aug 25 02:21:40 PDT 2026
================
@@ -1023,297 +1023,434 @@ bool LongJmpPass::relaxLocalBranches(BinaryFunction &BF,
return true;
}
-void LongJmpPass::relaxCalls(BinaryContext &BC) {
- // Operate on a copy of binary functions. We are going to manually insert new
- // thunks and update the list.
- BinaryFunctionListType OutputFunctions = BC.getOutputBinaryFunctions();
+static uint64_t estimateFragmentSize(const BinaryFunction &BF,
+ const FunctionFragment &FF) {
+ uint64_t Size = 0;
+ for (const BinaryBasicBlock *BB : FF)
+ Size += BB->estimateSize();
+
+ if (BF.hasIslandsInfo()) {
+ Size += BF.estimateConstantIslandSize();
+ if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
+ Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
+ }
- // Conservatively estimate emitted function size. Assume the worst case
- // alignment.
- auto estimateFunctionSize = [&](const BinaryFunction &BF) -> uint64_t {
- if (!BC.shouldEmit(BF))
- return 0;
- uint64_t Size = BF.estimateSize() + BF.getMaxAlignmentBytes();
-
- // Each additional fragment can attribute extra bytes due to its alignment
- // requirements.
- for ([[maybe_unused]] const FunctionFragment &FF :
- BF.getLayout().getSplitFragments())
- Size += BF.getMaxColdAlignmentBytes();
-
- if (BF.hasIslandsInfo()) {
- Size += BF.estimateConstantIslandSize();
- if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
- Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
- }
+ Size += FF.isSplitFragment() ? BF.getMaxColdAlignmentBytes()
+ : BF.getMaxAlignmentBytes();
+ return Size;
+}
- return Size;
+LongJmpPass::FragmentClusterLayout
+LongJmpPass::buildClusterLayout(BinaryContext &BC,
+ const BinaryFunctionListType &OutputFunctions) {
+ struct OutputFragment {
+ const FunctionFragment *FF;
+ size_t FunctionIndex;
+ SmallString<32> SectionName;
};
- // Map every function to its direct callees. Note that this is different from
- // the regular call graph as here we completely ignore indirect calls.
+ FragmentClusterLayout Layout;
+ SmallVector<OutputFragment> OrderedFragments;
uint64_t EstimatedSize = 0;
- DenseMap<BinaryFunction *, std::set<const MCSymbol *>> CallMap;
- for (BinaryFunction *BF : OutputFunctions) {
+ for (size_t I = 0; I < OutputFunctions.size(); ++I) {
+ BinaryFunction *BF = OutputFunctions[I];
if (!BC.shouldEmit(*BF) || BF->isPatch())
continue;
- EstimatedSize += estimateFunctionSize(*BF);
-
- for (const BinaryBasicBlock &BB : *BF) {
- for (const MCInst &Inst : BB) {
- if (!BC.MIB->isCall(Inst) || BC.MIB->isIndirectCall(Inst) ||
- BC.MIB->isIndirectBranch(Inst))
- continue;
- const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
- assert(TargetSymbol);
-
- // Ignore internal calls that use basic block labels as a destination.
- if (!BC.getFunctionForSymbol(TargetSymbol))
- continue;
+ for (const FunctionFragment &FF : BF->getLayout().fragments()) {
+ if (FF.empty() && !BF->hasConstantIsland())
+ continue;
- CallMap[BF].insert(TargetSymbol);
- }
+ OrderedFragments.push_back(
+ {&FF, I, BF->getCodeSectionName(FF.getFragmentNum())});
}
}
- LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << EstimatedSize
- << '\n');
-
- // Build clusters in the order the functions will appear in the output.
- std::vector<FunctionCluster> Clusters;
- for (size_t Index = 0, NumFuncs = OutputFunctions.size(); Index < NumFuncs;
- ++Index) {
- const size_t BFIndex =
- opts::HotFunctionsAtEnd ? NumFuncs - Index - 1 : Index;
- BinaryFunction *BF = OutputFunctions[BFIndex];
- if (!BC.shouldEmit(*BF) || BF->isPatch())
- continue;
+ // Model final output layout by grouping function fragments in output section
+ // order. Within each section, fragments remain in OutputFunctions order.
+ llvm::stable_sort(
+ OrderedFragments, [&](const OutputFragment &A, const OutputFragment &B) {
+ return BC.compareSectionNames(A.SectionName, B.SectionName);
+ });
- const uint64_t BFSize = estimateFunctionSize(*BF);
- if (Clusters.empty() || Clusters.back().Size + BFSize > MaxClusterSize) {
- Clusters.emplace_back(FunctionCluster());
- Clusters.back().FirstFunctionIndex = BFIndex;
+ auto addFragmentToCluster = [&](const OutputFragment &Fragment) {
+ BinaryFunction &BF = *OutputFunctions[Fragment.FunctionIndex];
+ const FunctionFragment &FF = *Fragment.FF;
+ const uint64_t FFSize = estimateFragmentSize(BF, FF);
+
+ if (Layout.Clusters.empty() ||
+ Layout.Clusters.back().SectionName != Fragment.SectionName ||
----------------
labrinea wrote:
We could let a cluster span across multiple sections and store the first and last. Backward thunks would be emitted in the first section, forward thunks in the last. Effect on Chromium:
| Fragment clusters | 3 | 2 |
| Cluster 0 | 38,360 fragments, 15,075,564 bytes | 279,314 fragments, 125,826,948 bytes |
| Cluster 1 | 276,625 fragments, 125,828,392 bytes | 212,395 fragments, 98,052,928 bytes |
| Cluster 2 | 176,705 fragments, 82,974,616 bytes | none |
| Calls relaxed with thunks | 2,290,090 | 1,426,434 |
| Short call thunks | 37,151 | 32,215 |
| Long call thunks | 11,325 | 4,743 |
| Cross-cluster branches | 298,823 | none reported |
| Branch thunks/chains | 286,027 | none reported |
https://github.com/llvm/llvm-project/pull/215825
More information about the llvm-commits
mailing list