[llvm] [BOLT][AArch64] Relax calls and branches with fragment clusters (PR #215825)

Alexandros Lamprineas via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 25 02:21:40 PDT 2026


================
@@ -1023,297 +1023,434 @@ bool LongJmpPass::relaxLocalBranches(BinaryFunction &BF,
   return true;
 }
 
-void LongJmpPass::relaxCalls(BinaryContext &BC) {
-  // Operate on a copy of binary functions. We are going to manually insert new
-  // thunks and update the list.
-  BinaryFunctionListType OutputFunctions = BC.getOutputBinaryFunctions();
+static uint64_t estimateFragmentSize(const BinaryFunction &BF,
+                                     const FunctionFragment &FF) {
+  uint64_t Size = 0;
+  for (const BinaryBasicBlock *BB : FF)
+    Size += BB->estimateSize();
+
+  if (BF.hasIslandsInfo()) {
+    Size += BF.estimateConstantIslandSize();
+    if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
+      Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
+  }
 
-  // Conservatively estimate emitted function size. Assume the worst case
-  // alignment.
-  auto estimateFunctionSize = [&](const BinaryFunction &BF) -> uint64_t {
-    if (!BC.shouldEmit(BF))
-      return 0;
-    uint64_t Size = BF.estimateSize() + BF.getMaxAlignmentBytes();
-
-    // Each additional fragment can attribute extra bytes due to its alignment
-    // requirements.
-    for ([[maybe_unused]] const FunctionFragment &FF :
-         BF.getLayout().getSplitFragments())
-      Size += BF.getMaxColdAlignmentBytes();
-
-    if (BF.hasIslandsInfo()) {
-      Size += BF.estimateConstantIslandSize();
-      if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
-        Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
-    }
+  Size += FF.isSplitFragment() ? BF.getMaxColdAlignmentBytes()
+                               : BF.getMaxAlignmentBytes();
+  return Size;
+}
 
-    return Size;
+LongJmpPass::FragmentClusterLayout
+LongJmpPass::buildClusterLayout(BinaryContext &BC,
+                                const BinaryFunctionListType &OutputFunctions) {
+  struct OutputFragment {
+    const FunctionFragment *FF;
+    size_t FunctionIndex;
+    SmallString<32> SectionName;
   };
 
-  // Map every function to its direct callees. Note that this is different from
-  // the regular call graph as here we completely ignore indirect calls.
+  FragmentClusterLayout Layout;
+  SmallVector<OutputFragment> OrderedFragments;
   uint64_t EstimatedSize = 0;
-  DenseMap<BinaryFunction *, std::set<const MCSymbol *>> CallMap;
-  for (BinaryFunction *BF : OutputFunctions) {
+  for (size_t I = 0; I < OutputFunctions.size(); ++I) {
+    BinaryFunction *BF = OutputFunctions[I];
     if (!BC.shouldEmit(*BF) || BF->isPatch())
       continue;
 
-    EstimatedSize += estimateFunctionSize(*BF);
-
-    for (const BinaryBasicBlock &BB : *BF) {
-      for (const MCInst &Inst : BB) {
-        if (!BC.MIB->isCall(Inst) || BC.MIB->isIndirectCall(Inst) ||
-            BC.MIB->isIndirectBranch(Inst))
-          continue;
-        const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
-        assert(TargetSymbol);
-
-        // Ignore internal calls that use basic block labels as a destination.
-        if (!BC.getFunctionForSymbol(TargetSymbol))
-          continue;
+    for (const FunctionFragment &FF : BF->getLayout().fragments()) {
+      if (FF.empty() && !BF->hasConstantIsland())
+        continue;
 
-        CallMap[BF].insert(TargetSymbol);
-      }
+      OrderedFragments.push_back(
+          {&FF, I, BF->getCodeSectionName(FF.getFragmentNum())});
     }
   }
 
-  LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << EstimatedSize
-                    << '\n');
-
-  // Build clusters in the order the functions will appear in the output.
-  std::vector<FunctionCluster> Clusters;
-  for (size_t Index = 0, NumFuncs = OutputFunctions.size(); Index < NumFuncs;
-       ++Index) {
-    const size_t BFIndex =
-        opts::HotFunctionsAtEnd ? NumFuncs - Index - 1 : Index;
-    BinaryFunction *BF = OutputFunctions[BFIndex];
-    if (!BC.shouldEmit(*BF) || BF->isPatch())
-      continue;
+  // Model final output layout by grouping function fragments in output section
+  // order. Within each section, fragments remain in OutputFunctions order.
+  llvm::stable_sort(
+      OrderedFragments, [&](const OutputFragment &A, const OutputFragment &B) {
+        return BC.compareSectionNames(A.SectionName, B.SectionName);
+      });
 
-    const uint64_t BFSize = estimateFunctionSize(*BF);
-    if (Clusters.empty() || Clusters.back().Size + BFSize > MaxClusterSize) {
-      Clusters.emplace_back(FunctionCluster());
-      Clusters.back().FirstFunctionIndex = BFIndex;
+  auto addFragmentToCluster = [&](const OutputFragment &Fragment) {
+    BinaryFunction &BF = *OutputFunctions[Fragment.FunctionIndex];
+    const FunctionFragment &FF = *Fragment.FF;
+    const uint64_t FFSize = estimateFragmentSize(BF, FF);
+
+    if (Layout.Clusters.empty() ||
+        Layout.Clusters.back().SectionName != Fragment.SectionName ||
----------------
labrinea wrote:

We could let a cluster span across multiple sections and store the first and last. Backward thunks would be emitted in the first section, forward thunks in the last. Effect on Chromium:

  | Fragment clusters | 3 | 2 |
  | Cluster 0 | 38,360 fragments, 15,075,564 bytes | 279,314 fragments, 125,826,948 bytes |
  | Cluster 1 | 276,625 fragments, 125,828,392 bytes | 212,395 fragments, 98,052,928 bytes |
  | Cluster 2 | 176,705 fragments, 82,974,616 bytes | none |
  | Calls relaxed with thunks | 2,290,090 | 1,426,434 |
  | Short call thunks | 37,151 | 32,215 |
  | Long call thunks | 11,325 | 4,743 |
  | Cross-cluster branches | 298,823 | none reported |
  | Branch thunks/chains | 286,027 | none reported |


https://github.com/llvm/llvm-project/pull/215825


More information about the llvm-commits mailing list