[llvm] [BOLT][AArch64] Relax calls and branches with fragment clusters (PR #215825)
Rafael Auler via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 19:06:52 PDT 2026
================
@@ -1023,301 +1034,761 @@ bool LongJmpPass::relaxLocalBranches(BinaryFunction &BF,
return true;
}
-void LongJmpPass::relaxCalls(BinaryContext &BC) {
- // Operate on a copy of binary functions. We are going to manually insert new
- // thunks and update the list.
- BinaryFunctionListType OutputFunctions = BC.getOutputBinaryFunctions();
+namespace {
+class ClusteredRelaxation {
+public:
+ ClusteredRelaxation(BinaryContext &BC,
+ BinaryFunctionListType &OutputFunctions)
+ : BC(BC), OutputFunctions(OutputFunctions) {}
+
+ bool run();
+
+private:
+ static constexpr uint64_t ShortThunkSize = 4;
+ static constexpr uint64_t LongThunkSize = 12;
+
+ /// A group of function fragments that are located within the longest direct
+ /// branch/call instruction distance. Jumps within the cluster do not require
+ /// a thunk. The cluster may span output sections and include thunks for jumps
+ /// to targets outside. Backward thunks are inserted before the cluster, while
+ /// forward thunks are inserted after it.
+ struct FragmentCluster {
+ /// Output code sections containing the first and last cluster fragments.
+ SmallString<32> StartSectionName;
+ SmallString<32> EndSectionName;
+
+ /// Estimated size of the cluster in bytes.
+ uint64_t Size{0};
+
+ /// Estimated output offset of the cluster.
+ uint64_t StartOffset{0};
+
+ /// Estimated bytes for thunks that may be inserted around this cluster.
+ uint64_t EstimatedThunkBytes{0};
+
+ /// Actual bytes for thunks emitted by this cluster.
+ uint64_t ActualThunkBytes{0};
+
+ /// Number of function fragments in the cluster.
+ size_t NumFragments{0};
+
+ /// The indices of the first and last functions contributing fragments to
+ /// this cluster. Used as insertion points for adding thunks to the output
+ /// function list.
+ size_t FirstFunctionIndex = -1;
+ size_t LastFunctionIndex = -1;
+
+ /// Thunks located after this cluster.
+ BinaryFunctionListType ForwardThunkList;
- // Conservatively estimate emitted function size. Assume the worst case
- // alignment.
- auto estimateFunctionSize = [&](const BinaryFunction &BF) -> uint64_t {
- if (!BC.shouldEmit(BF))
- return 0;
- uint64_t Size = BF.estimateSize() + BF.getMaxAlignmentBytes();
-
- // Each additional fragment can attribute extra bytes due to its alignment
- // requirements.
- for ([[maybe_unused]] const FunctionFragment &FF :
- BF.getLayout().getSplitFragments())
- Size += BF.getMaxColdAlignmentBytes();
-
- if (BF.hasIslandsInfo()) {
- Size += BF.estimateConstantIslandSize();
- if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
- Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
+ /// Thunks located before this cluster.
+ BinaryFunctionListType BackwardThunkList;
+
+ /// Long call thunks emitted by this cluster.
+ ///
+ /// <Target Symbol> -> <Thunk Function>.
+ DenseMap<const MCSymbol *, BinaryFunction *> LongThunks;
+
+ /// B-only thunks emitted by this cluster.
+ ///
+ /// <Target Symbol> -> <Thunk Function>.
+ DenseMap<const MCSymbol *, BinaryFunction *> BranchThunks;
+
+ StringRef getThunkSectionName(bool IsForward) const {
+ return IsForward ? EndSectionName : StartSectionName;
}
- return Size;
+ uint64_t getEndOffset() const { return StartOffset + Size; }
};
- // Map every function to its direct callees. Note that this is different from
- // the regular call graph as here we completely ignore indirect calls.
- uint64_t EstimatedSize = 0;
- DenseMap<BinaryFunction *, std::set<const MCSymbol *>> CallMap;
- for (BinaryFunction *BF : OutputFunctions) {
- if (!BC.shouldEmit(*BF) || BF->isPatch())
- continue;
+ struct Position {
+ unsigned Cluster;
+ uint64_t Offset;
+ };
- EstimatedSize += estimateFunctionSize(*BF);
+ /// Relaxation info for out-of-range cross-cluster references.
+ struct OutOfRangeRef {
+ MCInst *Inst;
+ const MCSymbol *TargetSymbol;
+ uint64_t SourceOffset;
+ uint64_t TargetOffset;
+ unsigned SourceCluster;
+ unsigned TargetCluster;
+ };
- for (const BinaryBasicBlock &BB : *BF) {
- for (const MCInst &Inst : BB) {
- if (!BC.MIB->isCall(Inst) || BC.MIB->isIndirectCall(Inst) ||
- BC.MIB->isIndirectBranch(Inst))
- continue;
- const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
- assert(TargetSymbol);
+ static unsigned getClusterDistance(unsigned A, unsigned B);
+ static uint64_t estimateFragmentSize(const BinaryFunction &BF,
+ const FunctionFragment &FF);
+ void buildLayout();
+ void printStats() const;
+ void collectOutOfRangeReferences();
+ void estimateThunkBytes();
+ const MCSymbol *getOrCreateBranchThunkChain(const OutOfRangeRef &Ref,
+ unsigned MaxThunks);
+ void relaxCalls();
+ void relaxUnconditionalBranches();
+ void insertThunks();
+
+ BinaryContext &BC;
+ BinaryFunctionListType &OutputFunctions;
+
+ /// Estimated cluster layout and source/target position maps.
+ SmallVector<FragmentCluster, 4> Clusters;
+ DenseMap<const BinaryBasicBlock *, Position> BBLayout;
+ DenseMap<const MCSymbol *, Position> SymLayout;
+
+ /// Out-of-range calls and branches grouped for relaxation.
+ SmallVector<OutOfRangeRef> OutOfLayoutCalls;
+ SmallVector<SmallVector<OutOfRangeRef>, 4> CallsByDistance;
+ SmallVector<OutOfRangeRef> Branches;
+
+ /// Relaxation counters reported in BOLT-INFO.
+ size_t NumShortThunkCalls = 0;
+ size_t NumLongThunkCalls = 0;
+ size_t NumShortThunks = 0;
+ size_t NumShortThunksReused = 0;
+ size_t NumLongThunks = 0;
+ size_t NumLongThunksReused = 0;
+ size_t NumRelaxedBranches = 0;
+ size_t NumBranchThunks = 0;
+ size_t NumBranchThunksReused = 0;
+};
+} // namespace
+
+unsigned ClusteredRelaxation::getClusterDistance(unsigned A, unsigned B) {
+ return A > B ? A - B : B - A;
+}
- // Ignore internal calls that use basic block labels as a destination.
- if (!BC.getFunctionForSymbol(TargetSymbol))
- continue;
+uint64_t ClusteredRelaxation::estimateFragmentSize(const BinaryFunction &BF,
+ const FunctionFragment &FF) {
+ uint64_t Size = 0;
+ for (const BinaryBasicBlock *BB : FF)
+ Size += BB->estimateSize();
- CallMap[BF].insert(TargetSymbol);
- }
- }
+ if (BF.hasIslandsInfo()) {
+ Size += BF.estimateConstantIslandSize();
+ if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
+ Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
}
- LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << EstimatedSize
- << '\n');
+ Size += FF.isSplitFragment() ? BF.getMaxColdAlignmentBytes()
+ : BF.getMaxAlignmentBytes();
+ return Size;
+}
+
+void ClusteredRelaxation::buildLayout() {
+ struct OutputFragment {
+ const FunctionFragment *FF;
+ size_t FunctionIndex;
+ SmallString<32> SectionName;
+ uint64_t Size;
+ };
- // Build clusters in the order the functions will appear in the output.
- std::vector<FunctionCluster> Clusters;
- for (size_t Index = 0, NumFuncs = OutputFunctions.size(); Index < NumFuncs;
- ++Index) {
- const size_t BFIndex =
- opts::HotFunctionsAtEnd ? NumFuncs - Index - 1 : Index;
- BinaryFunction *BF = OutputFunctions[BFIndex];
+ struct FragmentRange {
+ size_t Begin;
+ size_t End;
+ };
+
+ SmallVector<SmallVector<OutputFragment>, 4> FragmentsBySection;
+ StringMap<size_t> SectionToBucket;
+ auto addOrderedFragment = [&](OutputFragment &&Fragment) {
+ auto [It, Inserted] = SectionToBucket.try_emplace(
+ Fragment.SectionName, FragmentsBySection.size());
+ if (Inserted)
+ FragmentsBySection.emplace_back();
+ FragmentsBySection[It->second].push_back(std::move(Fragment));
+ };
+
+ for (size_t I = 0; I < OutputFunctions.size(); ++I) {
+ BinaryFunction *BF = OutputFunctions[I];
if (!BC.shouldEmit(*BF) || BF->isPatch())
continue;
- const uint64_t BFSize = estimateFunctionSize(*BF);
- if (Clusters.empty() || Clusters.back().Size + BFSize > MaxClusterSize) {
- Clusters.emplace_back(FunctionCluster());
- Clusters.back().FirstFunctionIndex = BFIndex;
+ for (const FunctionFragment &FF : BF->getLayout().fragments()) {
+ if (FF.empty() && !BF->hasConstantIsland())
+ continue;
+
+ addOrderedFragment({&FF, I, BF->getCodeSectionName(FF.getFragmentNum()),
+ estimateFragmentSize(*BF, FF)});
}
+ }
+
+ // Model final output layout by grouping function fragments in output section
+ // order. Within each section, fragments remain in OutputFunctions order.
+ SmallVector<size_t, 4> SectionOrder;
+ for (size_t I = 0; I < FragmentsBySection.size(); ++I)
+ SectionOrder.push_back(I);
- FunctionCluster &FC = Clusters.back();
- FC.Functions.insert(BF);
+ llvm::sort(SectionOrder, [&](size_t A, size_t B) {
+ return BC.compareSectionNames(FragmentsBySection[A].front().SectionName,
+ FragmentsBySection[B].front().SectionName);
+ });
- // When a function is added to the cluster, we have to remove all of its
- // symbols from the cluster callee list. These include alternative symbols
- // (e.g. after ICF) and secondary entry point symbols.
- for (const MCSymbol *Symbol : BF->getSymbols()) {
- auto It = FC.Callees.find(Symbol);
- if (It != FC.Callees.end())
- FC.Callees.erase(It);
+ SmallVector<const OutputFragment *> OrderedFragments;
+ for (size_t SectionIndex : SectionOrder)
+ for (const OutputFragment &Fragment : FragmentsBySection[SectionIndex])
+ OrderedFragments.push_back(&Fragment);
+
+ // Fragment clusters are built starting from hot code.
+ auto buildClusterRanges = [&]() {
+ SmallVector<FragmentRange> ClusterRanges;
+ if (OrderedFragments.empty())
+ return ClusterRanges;
+
+ // Hot fragments appear first, so perform forward walk.
+ if (!opts::HotFunctionsAtEnd) {
+ size_t Begin = 0;
+ uint64_t Size = OrderedFragments[0]->Size;
+ for (size_t I = 1; I < OrderedFragments.size(); ++I) {
+ if (Size + OrderedFragments[I]->Size > opts::MaxClusterSize) {
+ ClusterRanges.push_back({Begin, I});
+ Begin = I;
+ Size = 0;
+ }
+ Size += OrderedFragments[I]->Size;
+ }
+ ClusterRanges.push_back({Begin, OrderedFragments.size()});
+ return ClusterRanges;
}
- BF->forEachEntryPoint(
- [&FC](uint64_t Offset, const MCSymbol *EntrySymbol) -> bool {
- auto It = FC.Callees.find(EntrySymbol);
- if (It != FC.Callees.end())
- FC.Callees.erase(It);
- return true;
- });
-
- // Update cluster callee list with added function callees.
- for (const MCSymbol *CalleeSymbol : CallMap[BF]) {
- BinaryFunction *Callee = BC.getFunctionForSymbol(CalleeSymbol);
- if (!FC.Functions.count(Callee)) {
- FC.Callees.insert(CalleeSymbol);
+
+ // Hot fragments appear last, so perform reverse walk.
+ size_t End = OrderedFragments.size();
+ uint64_t Size = OrderedFragments.back()->Size;
+ for (size_t I = End - 1; I > 0;) {
+ --I;
+ if (Size + OrderedFragments[I]->Size > opts::MaxClusterSize) {
+ ClusterRanges.push_back({I + 1, End});
+ End = I + 1;
+ Size = 0;
}
+ Size += OrderedFragments[I]->Size;
}
+ ClusterRanges.push_back({0, End});
+ std::reverse(ClusterRanges.begin(), ClusterRanges.end());
+ return ClusterRanges;
+ };
- FC.Size += BFSize;
- FC.LastFunctionIndex = BFIndex;
- }
+ uint64_t LayoutOffset = 0;
+ auto addFragmentToCluster = [&](const OutputFragment &Fragment) {
+ FragmentCluster &FC = Clusters.back();
+ const unsigned ClusterNum = Clusters.size() - 1;
+ BinaryFunction &BF = *OutputFunctions[Fragment.FunctionIndex];
+ const FunctionFragment &FF = *Fragment.FF;
+ const uint64_t FragmentOffset = LayoutOffset;
+
+ if (FC.NumFragments == 0) {
+ FC.StartSectionName = Fragment.SectionName;
+ FC.StartOffset = FragmentOffset;
+ FC.FirstFunctionIndex = Fragment.FunctionIndex;
+ }
- if (opts::HotFunctionsAtEnd) {
- std::reverse(Clusters.begin(), Clusters.end());
- llvm::for_each(Clusters, [](FunctionCluster &FC) {
- std::swap(FC.LastFunctionIndex, FC.FirstFunctionIndex);
- });
+ FC.EndSectionName = Fragment.SectionName;
+ FC.LastFunctionIndex = Fragment.FunctionIndex;
+ ++FC.NumFragments;
+
+ // Map primary entry points.
+ if (FF.isMainFragment())
+ for (const MCSymbol *Symbol : BF.getSymbols())
+ SymLayout[Symbol] = {ClusterNum, FragmentOffset};
+
+ uint64_t BBOffset = FragmentOffset;
+ for (const BinaryBasicBlock *BB : FF) {
+ BBLayout[BB] = {ClusterNum, BBOffset};
+ // Map the local BB label.
+ if (const MCSymbol *Label = BB->getLabel())
+ SymLayout[Label] = {ClusterNum, BBOffset};
+ // Map the secondary entry point, which can differ from the BB label.
+ if (MCSymbol *Label = BF.getLabelAtOffset(BB->getOffset()))
+ if (MCSymbol *EntrySymbol = BF.getSecondaryEntryPointSymbol(Label))
+ SymLayout[EntrySymbol] = {ClusterNum, BBOffset};
+
+ BBOffset += BB->estimateSize();
+ }
+
+ FC.Size += Fragment.Size;
+ LayoutOffset += Fragment.Size;
+ };
+
+ for (const FragmentRange &Range : buildClusterRanges()) {
+ Clusters.emplace_back();
+ for (size_t I = Range.Begin; I < Range.End; ++I)
+ addFragmentToCluster(*OrderedFragments[I]);
}
if (Clusters.empty())
return;
- // Print cluster stats.
+ LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << LayoutOffset
+ << '\n');
+}
+
+void ClusteredRelaxation::printStats() const {
BC.outs() << "BOLT-INFO: built " << Clusters.size()
- << " function cluster(s)\n";
- uint64_t ClusterIndex = 0;
- for (const FunctionCluster &FC : Clusters) {
- BC.outs() << "BOLT-INFO: cluster: " << ClusterIndex++ << '\n'
- << "BOLT-INFO: " << FC.Functions.size() << " function(s)\n"
- << "BOLT-INFO: " << FC.Callees.size() << " callee(s)\n"
- << "BOLT-INFO: " << FC.Size << " estimated bytes\n";
+ << " function fragment cluster(s)\n";
+ for (size_t I = 0; I < Clusters.size(); ++I) {
+ const FragmentCluster &FC = Clusters[I];
+ BC.outs() << "BOLT-INFO: cluster: " << I << '\n'
+ << "BOLT-INFO: " << FC.NumFragments << " fragment(s)\n"
+ << "BOLT-INFO: " << FC.Size
+ << " estimated bytes without thunks\n"
+ << "BOLT-INFO: " << FC.EstimatedThunkBytes
+ << " estimated thunk bytes\n"
+ << "BOLT-INFO: " << FC.ActualThunkBytes
+ << " actual thunk bytes\n";
}
- if (opts::RelaxPLT) {
- // Populate one of the clusters with PLT functions based on the proximity of
- // the PLT section to avoid unneeded thunk redirection.
- const size_t PLTClusterNum = opts::UseOldText ? Clusters.size() - 1 : 0;
- auto &PLTCluster = Clusters[PLTClusterNum];
- for (BinaryFunction &BF :
- llvm::make_second_range(BC.getBinaryFunctions())) {
- if (BF.isPLTFunction()) {
- PLTCluster.Functions.insert(&BF);
- auto It = PLTCluster.Callees.find(BF.getSymbol());
- if (It != PLTCluster.Callees.end())
- PLTCluster.Callees.erase(It);
+ if (NumShortThunkCalls)
+ BC.outs() << "BOLT-INFO: relaxed " << NumShortThunkCalls
+ << " calls with short thunks\n";
+
+ if (NumLongThunkCalls)
+ BC.outs() << "BOLT-INFO: relaxed " << NumLongThunkCalls
+ << " calls with long thunks\n";
+
+ if (NumShortThunks)
+ BC.outs() << "BOLT-INFO: " << NumShortThunks << " short thunks created\n";
+
+ if (NumLongThunks)
+ BC.outs() << "BOLT-INFO: " << NumLongThunks << " long thunks created\n";
+
+ if (NumShortThunksReused)
+ BC.outs() << "BOLT-INFO: " << NumShortThunksReused
+ << " short thunks reused\n";
+
+ if (NumLongThunksReused)
+ BC.outs() << "BOLT-INFO: " << NumLongThunksReused
+ << " long thunks reused\n";
+
+ if (NumRelaxedBranches)
+ BC.outs() << "BOLT-INFO: relaxed " << NumRelaxedBranches
+ << " unconditional branches\n";
+
+ if (NumBranchThunks)
+ BC.outs() << "BOLT-INFO: " << NumBranchThunks << " branch thunks created\n";
+
+ if (NumBranchThunksReused)
+ BC.outs() << "BOLT-INFO: " << NumBranchThunksReused
+ << " branch thunks reused\n";
+}
+
+void ClusteredRelaxation::collectOutOfRangeReferences() {
+ CallsByDistance.resize(Clusters.size());
+ auto isPrimaryEntryTarget = [&](const MCSymbol *TargetSymbol) {
+ uint64_t EntryID = 0;
+ const BinaryFunction *BF = BC.getFunctionForSymbol(TargetSymbol, &EntryID);
+ return BF && EntryID == 0;
+ };
+
+ // Walk all instructions once and collect both branches and calls.
+ for (BinaryFunction *BF : OutputFunctions) {
+ if (!BC.shouldEmit(*BF) || BF->isPatch())
+ continue;
+
+ for (BinaryBasicBlock &BB : *BF) {
+ auto SourceIt = BBLayout.find(&BB);
+ if (SourceIt == BBLayout.end())
+ continue;
+ const Position Source = SourceIt->second;
+ uint64_t InstOffset = Source.Offset;
+
+ for (MCInst &Inst : BB) {
+ const uint64_t SourceOffset = InstOffset;
+ if (!BC.MIB->isPseudo(Inst))
+ InstOffset += 4;
+
+ const bool IsCall = BC.MIB->isCall(Inst);
+ const bool IsTailCall = BC.MIB->isTailCall(Inst);
+ const bool IsUncondBranch = BC.MIB->isUnconditionalBranch(Inst);
+ if (!IsCall && !IsUncondBranch)
+ continue;
+
+ const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
+ if (!TargetSymbol)
+ continue;
+
+ auto TargetIt = SymLayout.find(TargetSymbol);
+ const bool Found = TargetIt != SymLayout.end();
+ // Unmapped targets use maximum offset and are treated as out of range.
+ const Position Target = Found ? TargetIt->second : Position{-1u, -1ULL};
+
+ // References within the same cluster do not need relaxation.
+ if (Source.Cluster == Target.Cluster)
+ continue;
+
+ const OutOfRangeRef Reference{&Inst, TargetSymbol,
+ SourceOffset, Target.Offset,
+ Source.Cluster, Target.Cluster};
+ const bool UseBranchChain =
+ Found && (IsUncondBranch ||
+ (IsTailCall && !isPrimaryEntryTarget(TargetSymbol)));
+ if (UseBranchChain) {
+ // A direct B to a body entry may be annotated as a tail call. It is
+ // not an ABI call boundary, so use branch chains instead of call
+ // thunks that may clobber x16/x17.
+ if (IsTailCall)
+ BC.MIB->convertTailCallToJmp(Inst);
+
+ Branches.push_back(Reference);
+ continue;
+ }
+
+ assert(IsCall && "expected call after branch-chain handling");
+
+ if (!Found) {
+ OutOfLayoutCalls.push_back(Reference);
+ } else {
+ const unsigned Distance = getClusterDistance(Reference.SourceCluster,
+ Reference.TargetCluster);
+ CallsByDistance[Distance].push_back(Reference);
+ }
}
}
}
+}
+
+void ClusteredRelaxation::estimateThunkBytes() {
+ // Estimate in the same order as relaxation so reuse approximates the
+ // thunks that will actually be created below. This models the worst case
+ // by assuming selected thunks are necessary without performing range checks.
+ SmallVector<DenseSet<const MCSymbol *>, 4> BranchThunkTargets;
+ SmallVector<DenseSet<const MCSymbol *>, 4> LongThunkTargets;
+ BranchThunkTargets.resize(Clusters.size());
+ LongThunkTargets.resize(Clusters.size());
+
+ auto estimateBranchThunks = [&](const OutOfRangeRef &Ref) {
+ const bool IsForward = Ref.SourceCluster < Ref.TargetCluster;
+ auto getClusterAtHop = [&](unsigned Hop) {
+ return IsForward ? Ref.SourceCluster + Hop : Ref.SourceCluster - Hop;
+ };
+
+ const unsigned NumHops =
+ getClusterDistance(Ref.SourceCluster, Ref.TargetCluster);
+ for (unsigned Hop = 0; Hop < NumHops; ++Hop) {
+ const unsigned Cluster = getClusterAtHop(Hop);
+ if (BranchThunkTargets[Cluster].insert(Ref.TargetSymbol).second)
+ Clusters[Cluster].EstimatedThunkBytes += ShortThunkSize;
+ }
+ };
+
+ auto estimateLongThunk = [&](const OutOfRangeRef &Ref) {
+ const unsigned SourceCluster = Ref.SourceCluster;
+ const bool IsForward = SourceCluster < Ref.TargetCluster;
+ if (LongThunkTargets[SourceCluster].contains(Ref.TargetSymbol))
+ return;
+
+ unsigned ReuseCluster = -1u;
+ if (IsForward && SourceCluster > 0)
+ ReuseCluster = SourceCluster - 1;
+ else if (!IsForward && SourceCluster + 1 < Clusters.size())
+ ReuseCluster = SourceCluster + 1;
+
+ if (ReuseCluster != -1u &&
+ LongThunkTargets[ReuseCluster].contains(Ref.TargetSymbol))
+ return;
+
+ LongThunkTargets[SourceCluster].insert(Ref.TargetSymbol);
+ Clusters[SourceCluster].EstimatedThunkBytes += LongThunkSize;
----------------
rafaelauler wrote:
We can't really model the borrow behavior here, this is not known until relaxation runs. We might underestimate here, just be conservative and assume no borrowing exists, otherwise we will need to keep syncing this estimator to the real relaxation behavior:
```
if (LongThunkTargets[SourceCluster].insert(Ref.TargetSymbol).second)
Clusters[SourceCluster].EstimatedThunkBytes += LongThunkSize;
```
https://github.com/llvm/llvm-project/pull/215825
More information about the llvm-commits
mailing list