[llvm] [BOLT][AArch64] Relax calls and branches with fragment clusters (PR #215825)
Rafael Auler via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 19:06:53 PDT 2026
================
@@ -1023,301 +1034,761 @@ bool LongJmpPass::relaxLocalBranches(BinaryFunction &BF,
return true;
}
-void LongJmpPass::relaxCalls(BinaryContext &BC) {
- // Operate on a copy of binary functions. We are going to manually insert new
- // thunks and update the list.
- BinaryFunctionListType OutputFunctions = BC.getOutputBinaryFunctions();
+namespace {
+class ClusteredRelaxation {
+public:
+ ClusteredRelaxation(BinaryContext &BC,
+ BinaryFunctionListType &OutputFunctions)
+ : BC(BC), OutputFunctions(OutputFunctions) {}
+
+ bool run();
+
+private:
+ static constexpr uint64_t ShortThunkSize = 4;
+ static constexpr uint64_t LongThunkSize = 12;
+
+ /// A group of function fragments that are located within the longest direct
+ /// branch/call instruction distance. Jumps within the cluster do not require
+ /// a thunk. The cluster may span output sections and include thunks for jumps
+ /// to targets outside. Backward thunks are inserted before the cluster, while
+ /// forward thunks are inserted after it.
+ struct FragmentCluster {
+ /// Output code sections containing the first and last cluster fragments.
+ SmallString<32> StartSectionName;
+ SmallString<32> EndSectionName;
+
+ /// Estimated size of the cluster in bytes.
+ uint64_t Size{0};
+
+ /// Estimated output offset of the cluster.
+ uint64_t StartOffset{0};
+
+ /// Estimated bytes for thunks that may be inserted around this cluster.
+ uint64_t EstimatedThunkBytes{0};
+
+ /// Actual bytes for thunks emitted by this cluster.
+ uint64_t ActualThunkBytes{0};
+
+ /// Number of function fragments in the cluster.
+ size_t NumFragments{0};
+
+ /// The indices of the first and last functions contributing fragments to
+ /// this cluster. Used as insertion points for adding thunks to the output
+ /// function list.
+ size_t FirstFunctionIndex = -1;
+ size_t LastFunctionIndex = -1;
+
+ /// Thunks located after this cluster.
+ BinaryFunctionListType ForwardThunkList;
- // Conservatively estimate emitted function size. Assume the worst case
- // alignment.
- auto estimateFunctionSize = [&](const BinaryFunction &BF) -> uint64_t {
- if (!BC.shouldEmit(BF))
- return 0;
- uint64_t Size = BF.estimateSize() + BF.getMaxAlignmentBytes();
-
- // Each additional fragment can attribute extra bytes due to its alignment
- // requirements.
- for ([[maybe_unused]] const FunctionFragment &FF :
- BF.getLayout().getSplitFragments())
- Size += BF.getMaxColdAlignmentBytes();
-
- if (BF.hasIslandsInfo()) {
- Size += BF.estimateConstantIslandSize();
- if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
- Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
+ /// Thunks located before this cluster.
+ BinaryFunctionListType BackwardThunkList;
+
+ /// Long call thunks emitted by this cluster.
+ ///
+ /// <Target Symbol> -> <Thunk Function>.
+ DenseMap<const MCSymbol *, BinaryFunction *> LongThunks;
+
+ /// B-only thunks emitted by this cluster.
+ ///
+ /// <Target Symbol> -> <Thunk Function>.
+ DenseMap<const MCSymbol *, BinaryFunction *> BranchThunks;
+
+ StringRef getThunkSectionName(bool IsForward) const {
+ return IsForward ? EndSectionName : StartSectionName;
}
- return Size;
+ uint64_t getEndOffset() const { return StartOffset + Size; }
};
- // Map every function to its direct callees. Note that this is different from
- // the regular call graph as here we completely ignore indirect calls.
- uint64_t EstimatedSize = 0;
- DenseMap<BinaryFunction *, std::set<const MCSymbol *>> CallMap;
- for (BinaryFunction *BF : OutputFunctions) {
- if (!BC.shouldEmit(*BF) || BF->isPatch())
- continue;
+ struct Position {
+ unsigned Cluster;
+ uint64_t Offset;
+ };
- EstimatedSize += estimateFunctionSize(*BF);
+ /// Relaxation info for out-of-range cross-cluster references.
+ struct OutOfRangeRef {
+ MCInst *Inst;
+ const MCSymbol *TargetSymbol;
+ uint64_t SourceOffset;
+ uint64_t TargetOffset;
+ unsigned SourceCluster;
+ unsigned TargetCluster;
+ };
- for (const BinaryBasicBlock &BB : *BF) {
- for (const MCInst &Inst : BB) {
- if (!BC.MIB->isCall(Inst) || BC.MIB->isIndirectCall(Inst) ||
- BC.MIB->isIndirectBranch(Inst))
- continue;
- const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
- assert(TargetSymbol);
+ static unsigned getClusterDistance(unsigned A, unsigned B);
+ static uint64_t estimateFragmentSize(const BinaryFunction &BF,
+ const FunctionFragment &FF);
+ void buildLayout();
+ void printStats() const;
+ void collectOutOfRangeReferences();
+ void estimateThunkBytes();
+ const MCSymbol *getOrCreateBranchThunkChain(const OutOfRangeRef &Ref,
+ unsigned MaxThunks);
+ void relaxCalls();
+ void relaxUnconditionalBranches();
+ void insertThunks();
+
+ BinaryContext &BC;
+ BinaryFunctionListType &OutputFunctions;
+
+ /// Estimated cluster layout and source/target position maps.
+ SmallVector<FragmentCluster, 4> Clusters;
+ DenseMap<const BinaryBasicBlock *, Position> BBLayout;
+ DenseMap<const MCSymbol *, Position> SymLayout;
+
+ /// Out-of-range calls and branches grouped for relaxation.
+ SmallVector<OutOfRangeRef> OutOfLayoutCalls;
+ SmallVector<SmallVector<OutOfRangeRef>, 4> CallsByDistance;
+ SmallVector<OutOfRangeRef> Branches;
+
+ /// Relaxation counters reported in BOLT-INFO.
+ size_t NumShortThunkCalls = 0;
+ size_t NumLongThunkCalls = 0;
+ size_t NumShortThunks = 0;
+ size_t NumShortThunksReused = 0;
+ size_t NumLongThunks = 0;
+ size_t NumLongThunksReused = 0;
+ size_t NumRelaxedBranches = 0;
+ size_t NumBranchThunks = 0;
+ size_t NumBranchThunksReused = 0;
+};
+} // namespace
+
+unsigned ClusteredRelaxation::getClusterDistance(unsigned A, unsigned B) {
+ return A > B ? A - B : B - A;
+}
- // Ignore internal calls that use basic block labels as a destination.
- if (!BC.getFunctionForSymbol(TargetSymbol))
- continue;
+uint64_t ClusteredRelaxation::estimateFragmentSize(const BinaryFunction &BF,
+ const FunctionFragment &FF) {
+ uint64_t Size = 0;
+ for (const BinaryBasicBlock *BB : FF)
+ Size += BB->estimateSize();
- CallMap[BF].insert(TargetSymbol);
- }
- }
+ if (BF.hasIslandsInfo()) {
+ Size += BF.estimateConstantIslandSize();
+ if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
+ Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
}
- LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << EstimatedSize
- << '\n');
+ Size += FF.isSplitFragment() ? BF.getMaxColdAlignmentBytes()
+ : BF.getMaxAlignmentBytes();
+ return Size;
+}
+
+void ClusteredRelaxation::buildLayout() {
+ struct OutputFragment {
+ const FunctionFragment *FF;
+ size_t FunctionIndex;
+ SmallString<32> SectionName;
+ uint64_t Size;
+ };
- // Build clusters in the order the functions will appear in the output.
- std::vector<FunctionCluster> Clusters;
- for (size_t Index = 0, NumFuncs = OutputFunctions.size(); Index < NumFuncs;
- ++Index) {
- const size_t BFIndex =
- opts::HotFunctionsAtEnd ? NumFuncs - Index - 1 : Index;
- BinaryFunction *BF = OutputFunctions[BFIndex];
+ struct FragmentRange {
+ size_t Begin;
+ size_t End;
+ };
+
+ SmallVector<SmallVector<OutputFragment>, 4> FragmentsBySection;
+ StringMap<size_t> SectionToBucket;
+ auto addOrderedFragment = [&](OutputFragment &&Fragment) {
+ auto [It, Inserted] = SectionToBucket.try_emplace(
+ Fragment.SectionName, FragmentsBySection.size());
+ if (Inserted)
+ FragmentsBySection.emplace_back();
+ FragmentsBySection[It->second].push_back(std::move(Fragment));
+ };
+
+ for (size_t I = 0; I < OutputFunctions.size(); ++I) {
+ BinaryFunction *BF = OutputFunctions[I];
if (!BC.shouldEmit(*BF) || BF->isPatch())
continue;
- const uint64_t BFSize = estimateFunctionSize(*BF);
- if (Clusters.empty() || Clusters.back().Size + BFSize > MaxClusterSize) {
- Clusters.emplace_back(FunctionCluster());
- Clusters.back().FirstFunctionIndex = BFIndex;
+ for (const FunctionFragment &FF : BF->getLayout().fragments()) {
+ if (FF.empty() && !BF->hasConstantIsland())
+ continue;
+
+ addOrderedFragment({&FF, I, BF->getCodeSectionName(FF.getFragmentNum()),
+ estimateFragmentSize(*BF, FF)});
}
+ }
+
+ // Model final output layout by grouping function fragments in output section
+ // order. Within each section, fragments remain in OutputFunctions order.
+ SmallVector<size_t, 4> SectionOrder;
+ for (size_t I = 0; I < FragmentsBySection.size(); ++I)
+ SectionOrder.push_back(I);
- FunctionCluster &FC = Clusters.back();
- FC.Functions.insert(BF);
+ llvm::sort(SectionOrder, [&](size_t A, size_t B) {
+ return BC.compareSectionNames(FragmentsBySection[A].front().SectionName,
+ FragmentsBySection[B].front().SectionName);
+ });
- // When a function is added to the cluster, we have to remove all of its
- // symbols from the cluster callee list. These include alternative symbols
- // (e.g. after ICF) and secondary entry point symbols.
- for (const MCSymbol *Symbol : BF->getSymbols()) {
- auto It = FC.Callees.find(Symbol);
- if (It != FC.Callees.end())
- FC.Callees.erase(It);
+ SmallVector<const OutputFragment *> OrderedFragments;
+ for (size_t SectionIndex : SectionOrder)
+ for (const OutputFragment &Fragment : FragmentsBySection[SectionIndex])
+ OrderedFragments.push_back(&Fragment);
+
+ // Fragment clusters are built starting from hot code.
+ auto buildClusterRanges = [&]() {
+ SmallVector<FragmentRange> ClusterRanges;
+ if (OrderedFragments.empty())
+ return ClusterRanges;
+
+ // Hot fragments appear first, so perform forward walk.
+ if (!opts::HotFunctionsAtEnd) {
+ size_t Begin = 0;
+ uint64_t Size = OrderedFragments[0]->Size;
+ for (size_t I = 1; I < OrderedFragments.size(); ++I) {
+ if (Size + OrderedFragments[I]->Size > opts::MaxClusterSize) {
+ ClusterRanges.push_back({Begin, I});
+ Begin = I;
+ Size = 0;
+ }
+ Size += OrderedFragments[I]->Size;
+ }
+ ClusterRanges.push_back({Begin, OrderedFragments.size()});
+ return ClusterRanges;
}
- BF->forEachEntryPoint(
- [&FC](uint64_t Offset, const MCSymbol *EntrySymbol) -> bool {
- auto It = FC.Callees.find(EntrySymbol);
- if (It != FC.Callees.end())
- FC.Callees.erase(It);
- return true;
- });
-
- // Update cluster callee list with added function callees.
- for (const MCSymbol *CalleeSymbol : CallMap[BF]) {
- BinaryFunction *Callee = BC.getFunctionForSymbol(CalleeSymbol);
- if (!FC.Functions.count(Callee)) {
- FC.Callees.insert(CalleeSymbol);
+
+ // Hot fragments appear last, so perform reverse walk.
+ size_t End = OrderedFragments.size();
+ uint64_t Size = OrderedFragments.back()->Size;
+ for (size_t I = End - 1; I > 0;) {
+ --I;
+ if (Size + OrderedFragments[I]->Size > opts::MaxClusterSize) {
+ ClusterRanges.push_back({I + 1, End});
+ End = I + 1;
+ Size = 0;
}
+ Size += OrderedFragments[I]->Size;
}
+ ClusterRanges.push_back({0, End});
+ std::reverse(ClusterRanges.begin(), ClusterRanges.end());
+ return ClusterRanges;
+ };
- FC.Size += BFSize;
- FC.LastFunctionIndex = BFIndex;
- }
+ uint64_t LayoutOffset = 0;
+ auto addFragmentToCluster = [&](const OutputFragment &Fragment) {
+ FragmentCluster &FC = Clusters.back();
+ const unsigned ClusterNum = Clusters.size() - 1;
+ BinaryFunction &BF = *OutputFunctions[Fragment.FunctionIndex];
+ const FunctionFragment &FF = *Fragment.FF;
+ const uint64_t FragmentOffset = LayoutOffset;
+
+ if (FC.NumFragments == 0) {
+ FC.StartSectionName = Fragment.SectionName;
+ FC.StartOffset = FragmentOffset;
+ FC.FirstFunctionIndex = Fragment.FunctionIndex;
+ }
- if (opts::HotFunctionsAtEnd) {
- std::reverse(Clusters.begin(), Clusters.end());
- llvm::for_each(Clusters, [](FunctionCluster &FC) {
- std::swap(FC.LastFunctionIndex, FC.FirstFunctionIndex);
- });
+ FC.EndSectionName = Fragment.SectionName;
+ FC.LastFunctionIndex = Fragment.FunctionIndex;
+ ++FC.NumFragments;
+
+ // Map primary entry points.
+ if (FF.isMainFragment())
+ for (const MCSymbol *Symbol : BF.getSymbols())
+ SymLayout[Symbol] = {ClusterNum, FragmentOffset};
+
+ uint64_t BBOffset = FragmentOffset;
+ for (const BinaryBasicBlock *BB : FF) {
+ BBLayout[BB] = {ClusterNum, BBOffset};
+ // Map the local BB label.
+ if (const MCSymbol *Label = BB->getLabel())
+ SymLayout[Label] = {ClusterNum, BBOffset};
+ // Map the secondary entry point, which can differ from the BB label.
+ if (MCSymbol *Label = BF.getLabelAtOffset(BB->getOffset()))
+ if (MCSymbol *EntrySymbol = BF.getSecondaryEntryPointSymbol(Label))
+ SymLayout[EntrySymbol] = {ClusterNum, BBOffset};
+
+ BBOffset += BB->estimateSize();
+ }
+
+ FC.Size += Fragment.Size;
+ LayoutOffset += Fragment.Size;
+ };
+
+ for (const FragmentRange &Range : buildClusterRanges()) {
+ Clusters.emplace_back();
+ for (size_t I = Range.Begin; I < Range.End; ++I)
+ addFragmentToCluster(*OrderedFragments[I]);
}
if (Clusters.empty())
return;
- // Print cluster stats.
+ LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << LayoutOffset
+ << '\n');
+}
+
+void ClusteredRelaxation::printStats() const {
BC.outs() << "BOLT-INFO: built " << Clusters.size()
- << " function cluster(s)\n";
- uint64_t ClusterIndex = 0;
- for (const FunctionCluster &FC : Clusters) {
- BC.outs() << "BOLT-INFO: cluster: " << ClusterIndex++ << '\n'
- << "BOLT-INFO: " << FC.Functions.size() << " function(s)\n"
- << "BOLT-INFO: " << FC.Callees.size() << " callee(s)\n"
- << "BOLT-INFO: " << FC.Size << " estimated bytes\n";
+ << " function fragment cluster(s)\n";
+ for (size_t I = 0; I < Clusters.size(); ++I) {
+ const FragmentCluster &FC = Clusters[I];
----------------
rafaelauler wrote:
```
assert(FC.ActualThunkBytes <= FC.EstimatedThunkBytes &&
"thunk estimate exceeded; range checks used a stale safety margin");
```
https://github.com/llvm/llvm-project/pull/215825
More information about the llvm-commits
mailing list