[llvm] [BOLT][AArch64] Relax calls and branches with fragment clusters (PR #215825)

Rafael Auler via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 19:06:52 PDT 2026


================
@@ -1023,301 +1034,761 @@ bool LongJmpPass::relaxLocalBranches(BinaryFunction &BF,
   return true;
 }
 
-void LongJmpPass::relaxCalls(BinaryContext &BC) {
-  // Operate on a copy of binary functions. We are going to manually insert new
-  // thunks and update the list.
-  BinaryFunctionListType OutputFunctions = BC.getOutputBinaryFunctions();
+namespace {
+class ClusteredRelaxation {
+public:
+  ClusteredRelaxation(BinaryContext &BC,
+                      BinaryFunctionListType &OutputFunctions)
+      : BC(BC), OutputFunctions(OutputFunctions) {}
+
+  bool run();
+
+private:
+  static constexpr uint64_t ShortThunkSize = 4;
+  static constexpr uint64_t LongThunkSize = 12;
+
+  /// A group of function fragments that are located within the longest direct
+  /// branch/call instruction distance. Jumps within the cluster do not require
+  /// a thunk. The cluster may span output sections and include thunks for jumps
+  /// to targets outside. Backward thunks are inserted before the cluster, while
+  /// forward thunks are inserted after it.
+  struct FragmentCluster {
+    /// Output code sections containing the first and last cluster fragments.
+    SmallString<32> StartSectionName;
+    SmallString<32> EndSectionName;
+
+    /// Estimated size of the cluster in bytes.
+    uint64_t Size{0};
+
+    /// Estimated output offset of the cluster.
+    uint64_t StartOffset{0};
+
+    /// Estimated bytes for thunks that may be inserted around this cluster.
+    uint64_t EstimatedThunkBytes{0};
+
+    /// Actual bytes for thunks emitted by this cluster.
+    uint64_t ActualThunkBytes{0};
+
+    /// Number of function fragments in the cluster.
+    size_t NumFragments{0};
+
+    /// The indices of the first and last functions contributing fragments to
+    /// this cluster. Used as insertion points for adding thunks to the output
+    /// function list.
+    size_t FirstFunctionIndex = -1;
+    size_t LastFunctionIndex = -1;
+
+    /// Thunks located after this cluster.
+    BinaryFunctionListType ForwardThunkList;
 
-  // Conservatively estimate emitted function size. Assume the worst case
-  // alignment.
-  auto estimateFunctionSize = [&](const BinaryFunction &BF) -> uint64_t {
-    if (!BC.shouldEmit(BF))
-      return 0;
-    uint64_t Size = BF.estimateSize() + BF.getMaxAlignmentBytes();
-
-    // Each additional fragment can attribute extra bytes due to its alignment
-    // requirements.
-    for ([[maybe_unused]] const FunctionFragment &FF :
-         BF.getLayout().getSplitFragments())
-      Size += BF.getMaxColdAlignmentBytes();
-
-    if (BF.hasIslandsInfo()) {
-      Size += BF.estimateConstantIslandSize();
-      if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
-        Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
+    /// Thunks located before this cluster.
+    BinaryFunctionListType BackwardThunkList;
+
+    /// Long call thunks emitted by this cluster.
+    ///
+    /// <Target Symbol> -> <Thunk Function>.
+    DenseMap<const MCSymbol *, BinaryFunction *> LongThunks;
+
+    /// B-only thunks emitted by this cluster.
+    ///
+    /// <Target Symbol> -> <Thunk Function>.
+    DenseMap<const MCSymbol *, BinaryFunction *> BranchThunks;
+
+    StringRef getThunkSectionName(bool IsForward) const {
+      return IsForward ? EndSectionName : StartSectionName;
     }
 
-    return Size;
+    uint64_t getEndOffset() const { return StartOffset + Size; }
   };
 
-  // Map every function to its direct callees. Note that this is different from
-  // the regular call graph as here we completely ignore indirect calls.
-  uint64_t EstimatedSize = 0;
-  DenseMap<BinaryFunction *, std::set<const MCSymbol *>> CallMap;
-  for (BinaryFunction *BF : OutputFunctions) {
-    if (!BC.shouldEmit(*BF) || BF->isPatch())
-      continue;
+  struct Position {
+    unsigned Cluster;
+    uint64_t Offset;
+  };
 
-    EstimatedSize += estimateFunctionSize(*BF);
+  /// Relaxation info for out-of-range cross-cluster references.
+  struct OutOfRangeRef {
+    MCInst *Inst;
+    const MCSymbol *TargetSymbol;
+    uint64_t SourceOffset;
+    uint64_t TargetOffset;
+    unsigned SourceCluster;
+    unsigned TargetCluster;
+  };
 
-    for (const BinaryBasicBlock &BB : *BF) {
-      for (const MCInst &Inst : BB) {
-        if (!BC.MIB->isCall(Inst) || BC.MIB->isIndirectCall(Inst) ||
-            BC.MIB->isIndirectBranch(Inst))
-          continue;
-        const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
-        assert(TargetSymbol);
+  static unsigned getClusterDistance(unsigned A, unsigned B);
+  static uint64_t estimateFragmentSize(const BinaryFunction &BF,
+                                       const FunctionFragment &FF);
+  void buildLayout();
+  void printStats() const;
+  void collectOutOfRangeReferences();
+  void estimateThunkBytes();
+  const MCSymbol *getOrCreateBranchThunkChain(const OutOfRangeRef &Ref,
+                                              unsigned MaxThunks);
+  void relaxCalls();
+  void relaxUnconditionalBranches();
+  void insertThunks();
+
+  BinaryContext &BC;
+  BinaryFunctionListType &OutputFunctions;
+
+  /// Estimated cluster layout and source/target position maps.
+  SmallVector<FragmentCluster, 4> Clusters;
+  DenseMap<const BinaryBasicBlock *, Position> BBLayout;
+  DenseMap<const MCSymbol *, Position> SymLayout;
+
+  /// Out-of-range calls and branches grouped for relaxation.
+  SmallVector<OutOfRangeRef> OutOfLayoutCalls;
+  SmallVector<SmallVector<OutOfRangeRef>, 4> CallsByDistance;
+  SmallVector<OutOfRangeRef> Branches;
+
+  /// Relaxation counters reported in BOLT-INFO.
+  size_t NumShortThunkCalls = 0;
+  size_t NumLongThunkCalls = 0;
+  size_t NumShortThunks = 0;
+  size_t NumShortThunksReused = 0;
+  size_t NumLongThunks = 0;
+  size_t NumLongThunksReused = 0;
+  size_t NumRelaxedBranches = 0;
+  size_t NumBranchThunks = 0;
+  size_t NumBranchThunksReused = 0;
+};
+} // namespace
+
+unsigned ClusteredRelaxation::getClusterDistance(unsigned A, unsigned B) {
+  return A > B ? A - B : B - A;
+}
 
-        // Ignore internal calls that use basic block labels as a destination.
-        if (!BC.getFunctionForSymbol(TargetSymbol))
-          continue;
+uint64_t ClusteredRelaxation::estimateFragmentSize(const BinaryFunction &BF,
+                                                   const FunctionFragment &FF) {
+  uint64_t Size = 0;
+  for (const BinaryBasicBlock *BB : FF)
+    Size += BB->estimateSize();
 
-        CallMap[BF].insert(TargetSymbol);
-      }
-    }
+  if (BF.hasIslandsInfo()) {
+    Size += BF.estimateConstantIslandSize();
+    if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
+      Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
   }
 
-  LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << EstimatedSize
-                    << '\n');
+  Size += FF.isSplitFragment() ? BF.getMaxColdAlignmentBytes()
+                               : BF.getMaxAlignmentBytes();
+  return Size;
+}
+
+void ClusteredRelaxation::buildLayout() {
+  struct OutputFragment {
+    const FunctionFragment *FF;
+    size_t FunctionIndex;
+    SmallString<32> SectionName;
+    uint64_t Size;
+  };
 
-  // Build clusters in the order the functions will appear in the output.
-  std::vector<FunctionCluster> Clusters;
-  for (size_t Index = 0, NumFuncs = OutputFunctions.size(); Index < NumFuncs;
-       ++Index) {
-    const size_t BFIndex =
-        opts::HotFunctionsAtEnd ? NumFuncs - Index - 1 : Index;
-    BinaryFunction *BF = OutputFunctions[BFIndex];
+  struct FragmentRange {
+    size_t Begin;
+    size_t End;
+  };
+
+  SmallVector<SmallVector<OutputFragment>, 4> FragmentsBySection;
+  StringMap<size_t> SectionToBucket;
+  auto addOrderedFragment = [&](OutputFragment &&Fragment) {
+    auto [It, Inserted] = SectionToBucket.try_emplace(
+        Fragment.SectionName, FragmentsBySection.size());
+    if (Inserted)
+      FragmentsBySection.emplace_back();
+    FragmentsBySection[It->second].push_back(std::move(Fragment));
+  };
+
+  for (size_t I = 0; I < OutputFunctions.size(); ++I) {
+    BinaryFunction *BF = OutputFunctions[I];
     if (!BC.shouldEmit(*BF) || BF->isPatch())
       continue;
 
-    const uint64_t BFSize = estimateFunctionSize(*BF);
-    if (Clusters.empty() || Clusters.back().Size + BFSize > MaxClusterSize) {
-      Clusters.emplace_back(FunctionCluster());
-      Clusters.back().FirstFunctionIndex = BFIndex;
+    for (const FunctionFragment &FF : BF->getLayout().fragments()) {
+      if (FF.empty() && !BF->hasConstantIsland())
+        continue;
+
+      addOrderedFragment({&FF, I, BF->getCodeSectionName(FF.getFragmentNum()),
+                          estimateFragmentSize(*BF, FF)});
     }
+  }
+
+  // Model final output layout by grouping function fragments in output section
+  // order. Within each section, fragments remain in OutputFunctions order.
+  SmallVector<size_t, 4> SectionOrder;
+  for (size_t I = 0; I < FragmentsBySection.size(); ++I)
+    SectionOrder.push_back(I);
 
-    FunctionCluster &FC = Clusters.back();
-    FC.Functions.insert(BF);
+  llvm::sort(SectionOrder, [&](size_t A, size_t B) {
+    return BC.compareSectionNames(FragmentsBySection[A].front().SectionName,
+                                  FragmentsBySection[B].front().SectionName);
+  });
 
-    // When a function is added to the cluster, we have to remove all of its
-    // symbols from the cluster callee list. These include alternative symbols
-    // (e.g. after ICF) and secondary entry point symbols.
-    for (const MCSymbol *Symbol : BF->getSymbols()) {
-      auto It = FC.Callees.find(Symbol);
-      if (It != FC.Callees.end())
-        FC.Callees.erase(It);
+  SmallVector<const OutputFragment *> OrderedFragments;
+  for (size_t SectionIndex : SectionOrder)
+    for (const OutputFragment &Fragment : FragmentsBySection[SectionIndex])
+      OrderedFragments.push_back(&Fragment);
+
+  // Fragment clusters are built starting from hot code.
+  auto buildClusterRanges = [&]() {
+    SmallVector<FragmentRange> ClusterRanges;
+    if (OrderedFragments.empty())
+      return ClusterRanges;
+
+    // Hot fragments appear first, so perform forward walk.
+    if (!opts::HotFunctionsAtEnd) {
+      size_t Begin = 0;
+      uint64_t Size = OrderedFragments[0]->Size;
+      for (size_t I = 1; I < OrderedFragments.size(); ++I) {
+        if (Size + OrderedFragments[I]->Size > opts::MaxClusterSize) {
+          ClusterRanges.push_back({Begin, I});
+          Begin = I;
+          Size = 0;
+        }
+        Size += OrderedFragments[I]->Size;
+      }
+      ClusterRanges.push_back({Begin, OrderedFragments.size()});
+      return ClusterRanges;
     }
-    BF->forEachEntryPoint(
-        [&FC](uint64_t Offset, const MCSymbol *EntrySymbol) -> bool {
-          auto It = FC.Callees.find(EntrySymbol);
-          if (It != FC.Callees.end())
-            FC.Callees.erase(It);
-          return true;
-        });
-
-    // Update cluster callee list with added function callees.
-    for (const MCSymbol *CalleeSymbol : CallMap[BF]) {
-      BinaryFunction *Callee = BC.getFunctionForSymbol(CalleeSymbol);
-      if (!FC.Functions.count(Callee)) {
-        FC.Callees.insert(CalleeSymbol);
+
+    // Hot fragments appear last, so perform reverse walk.
+    size_t End = OrderedFragments.size();
+    uint64_t Size = OrderedFragments.back()->Size;
+    for (size_t I = End - 1; I > 0;) {
+      --I;
+      if (Size + OrderedFragments[I]->Size > opts::MaxClusterSize) {
+        ClusterRanges.push_back({I + 1, End});
+        End = I + 1;
+        Size = 0;
       }
+      Size += OrderedFragments[I]->Size;
     }
+    ClusterRanges.push_back({0, End});
+    std::reverse(ClusterRanges.begin(), ClusterRanges.end());
+    return ClusterRanges;
+  };
 
-    FC.Size += BFSize;
-    FC.LastFunctionIndex = BFIndex;
-  }
+  uint64_t LayoutOffset = 0;
+  auto addFragmentToCluster = [&](const OutputFragment &Fragment) {
+    FragmentCluster &FC = Clusters.back();
+    const unsigned ClusterNum = Clusters.size() - 1;
+    BinaryFunction &BF = *OutputFunctions[Fragment.FunctionIndex];
+    const FunctionFragment &FF = *Fragment.FF;
+    const uint64_t FragmentOffset = LayoutOffset;
+
+    if (FC.NumFragments == 0) {
+      FC.StartSectionName = Fragment.SectionName;
+      FC.StartOffset = FragmentOffset;
+      FC.FirstFunctionIndex = Fragment.FunctionIndex;
+    }
 
-  if (opts::HotFunctionsAtEnd) {
-    std::reverse(Clusters.begin(), Clusters.end());
-    llvm::for_each(Clusters, [](FunctionCluster &FC) {
-      std::swap(FC.LastFunctionIndex, FC.FirstFunctionIndex);
-    });
+    FC.EndSectionName = Fragment.SectionName;
+    FC.LastFunctionIndex = Fragment.FunctionIndex;
+    ++FC.NumFragments;
+
+    // Map primary entry points.
+    if (FF.isMainFragment())
+      for (const MCSymbol *Symbol : BF.getSymbols())
+        SymLayout[Symbol] = {ClusterNum, FragmentOffset};
+
+    uint64_t BBOffset = FragmentOffset;
+    for (const BinaryBasicBlock *BB : FF) {
+      BBLayout[BB] = {ClusterNum, BBOffset};
+      // Map the local BB label.
+      if (const MCSymbol *Label = BB->getLabel())
+        SymLayout[Label] = {ClusterNum, BBOffset};
+      // Map the secondary entry point, which can differ from the BB label.
+      if (MCSymbol *Label = BF.getLabelAtOffset(BB->getOffset()))
+        if (MCSymbol *EntrySymbol = BF.getSecondaryEntryPointSymbol(Label))
+          SymLayout[EntrySymbol] = {ClusterNum, BBOffset};
+
+      BBOffset += BB->estimateSize();
+    }
+
+    FC.Size += Fragment.Size;
+    LayoutOffset += Fragment.Size;
+  };
+
+  for (const FragmentRange &Range : buildClusterRanges()) {
+    Clusters.emplace_back();
+    for (size_t I = Range.Begin; I < Range.End; ++I)
+      addFragmentToCluster(*OrderedFragments[I]);
   }
 
   if (Clusters.empty())
     return;
 
-  // Print cluster stats.
+  LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << LayoutOffset
+                    << '\n');
+}
+
+void ClusteredRelaxation::printStats() const {
   BC.outs() << "BOLT-INFO: built " << Clusters.size()
-            << " function cluster(s)\n";
-  uint64_t ClusterIndex = 0;
-  for (const FunctionCluster &FC : Clusters) {
-    BC.outs() << "BOLT-INFO: cluster: " << ClusterIndex++ << '\n'
-              << "BOLT-INFO:   " << FC.Functions.size() << " function(s)\n"
-              << "BOLT-INFO:   " << FC.Callees.size() << " callee(s)\n"
-              << "BOLT-INFO:   " << FC.Size << " estimated bytes\n";
+            << " function fragment cluster(s)\n";
+  for (size_t I = 0; I < Clusters.size(); ++I) {
+    const FragmentCluster &FC = Clusters[I];
+    BC.outs() << "BOLT-INFO: cluster: " << I << '\n'
+              << "BOLT-INFO:   " << FC.NumFragments << " fragment(s)\n"
+              << "BOLT-INFO:   " << FC.Size
+              << " estimated bytes without thunks\n"
+              << "BOLT-INFO:   " << FC.EstimatedThunkBytes
+              << " estimated thunk bytes\n"
+              << "BOLT-INFO:   " << FC.ActualThunkBytes
+              << " actual thunk bytes\n";
   }
 
-  if (opts::RelaxPLT) {
-    // Populate one of the clusters with PLT functions based on the proximity of
-    // the PLT section to avoid unneeded thunk redirection.
-    const size_t PLTClusterNum = opts::UseOldText ? Clusters.size() - 1 : 0;
-    auto &PLTCluster = Clusters[PLTClusterNum];
-    for (BinaryFunction &BF :
-         llvm::make_second_range(BC.getBinaryFunctions())) {
-      if (BF.isPLTFunction()) {
-        PLTCluster.Functions.insert(&BF);
-        auto It = PLTCluster.Callees.find(BF.getSymbol());
-        if (It != PLTCluster.Callees.end())
-          PLTCluster.Callees.erase(It);
+  if (NumShortThunkCalls)
+    BC.outs() << "BOLT-INFO: relaxed " << NumShortThunkCalls
+              << " calls with short thunks\n";
+
+  if (NumLongThunkCalls)
+    BC.outs() << "BOLT-INFO: relaxed " << NumLongThunkCalls
+              << " calls with long thunks\n";
+
+  if (NumShortThunks)
+    BC.outs() << "BOLT-INFO: " << NumShortThunks << " short thunks created\n";
+
+  if (NumLongThunks)
+    BC.outs() << "BOLT-INFO: " << NumLongThunks << " long thunks created\n";
+
+  if (NumShortThunksReused)
+    BC.outs() << "BOLT-INFO: " << NumShortThunksReused
+              << " short thunks reused\n";
+
+  if (NumLongThunksReused)
+    BC.outs() << "BOLT-INFO: " << NumLongThunksReused
+              << " long thunks reused\n";
+
+  if (NumRelaxedBranches)
+    BC.outs() << "BOLT-INFO: relaxed " << NumRelaxedBranches
+              << " unconditional branches\n";
+
+  if (NumBranchThunks)
+    BC.outs() << "BOLT-INFO: " << NumBranchThunks << " branch thunks created\n";
+
+  if (NumBranchThunksReused)
+    BC.outs() << "BOLT-INFO: " << NumBranchThunksReused
+              << " branch thunks reused\n";
+}
+
+void ClusteredRelaxation::collectOutOfRangeReferences() {
+  CallsByDistance.resize(Clusters.size());
+  auto isPrimaryEntryTarget = [&](const MCSymbol *TargetSymbol) {
+    uint64_t EntryID = 0;
+    const BinaryFunction *BF = BC.getFunctionForSymbol(TargetSymbol, &EntryID);
+    return BF && EntryID == 0;
+  };
+
+  // Walk all instructions once and collect both branches and calls.
+  for (BinaryFunction *BF : OutputFunctions) {
+    if (!BC.shouldEmit(*BF) || BF->isPatch())
+      continue;
+
+    for (BinaryBasicBlock &BB : *BF) {
+      auto SourceIt = BBLayout.find(&BB);
+      if (SourceIt == BBLayout.end())
+        continue;
+      const Position Source = SourceIt->second;
+      uint64_t InstOffset = Source.Offset;
+
+      for (MCInst &Inst : BB) {
+        const uint64_t SourceOffset = InstOffset;
+        if (!BC.MIB->isPseudo(Inst))
+          InstOffset += 4;
+
+        const bool IsCall = BC.MIB->isCall(Inst);
+        const bool IsTailCall = BC.MIB->isTailCall(Inst);
+        const bool IsUncondBranch = BC.MIB->isUnconditionalBranch(Inst);
+        if (!IsCall && !IsUncondBranch)
+          continue;
+
+        const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
+        if (!TargetSymbol)
+          continue;
+
+        auto TargetIt = SymLayout.find(TargetSymbol);
+        const bool Found = TargetIt != SymLayout.end();
+        // Unmapped targets use maximum offset and are treated as out of range.
+        const Position Target = Found ? TargetIt->second : Position{-1u, -1ULL};
+
+        // References within the same cluster do not need relaxation.
+        if (Source.Cluster == Target.Cluster)
+          continue;
+
+        const OutOfRangeRef Reference{&Inst,          TargetSymbol,
+                                      SourceOffset,   Target.Offset,
+                                      Source.Cluster, Target.Cluster};
+        const bool UseBranchChain =
+            Found && (IsUncondBranch ||
+                      (IsTailCall && !isPrimaryEntryTarget(TargetSymbol)));
+        if (UseBranchChain) {
+          // A direct B to a body entry may be annotated as a tail call. It is
+          // not an ABI call boundary, so use branch chains instead of call
+          // thunks that may clobber x16/x17.
+          if (IsTailCall)
+            BC.MIB->convertTailCallToJmp(Inst);
+
+          Branches.push_back(Reference);
+          continue;
+        }
+
+        assert(IsCall && "expected call after branch-chain handling");
+
+        if (!Found) {
+          OutOfLayoutCalls.push_back(Reference);
+        } else {
+          const unsigned Distance = getClusterDistance(Reference.SourceCluster,
+                                                       Reference.TargetCluster);
+          CallsByDistance[Distance].push_back(Reference);
+        }
       }
     }
   }
+}
+
+void ClusteredRelaxation::estimateThunkBytes() {
+  // Estimate in the same order as relaxation so reuse approximates the
+  // thunks that will actually be created below. This models the worst case
+  // by assuming selected thunks are necessary without performing range checks.
+  SmallVector<DenseSet<const MCSymbol *>, 4> BranchThunkTargets;
+  SmallVector<DenseSet<const MCSymbol *>, 4> LongThunkTargets;
+  BranchThunkTargets.resize(Clusters.size());
+  LongThunkTargets.resize(Clusters.size());
+
+  auto estimateBranchThunks = [&](const OutOfRangeRef &Ref) {
+    const bool IsForward = Ref.SourceCluster < Ref.TargetCluster;
+    auto getClusterAtHop = [&](unsigned Hop) {
+      return IsForward ? Ref.SourceCluster + Hop : Ref.SourceCluster - Hop;
+    };
+
+    const unsigned NumHops =
+        getClusterDistance(Ref.SourceCluster, Ref.TargetCluster);
+    for (unsigned Hop = 0; Hop < NumHops; ++Hop) {
+      const unsigned Cluster = getClusterAtHop(Hop);
+      if (BranchThunkTargets[Cluster].insert(Ref.TargetSymbol).second)
+        Clusters[Cluster].EstimatedThunkBytes += ShortThunkSize;
+    }
+  };
+
+  auto estimateLongThunk = [&](const OutOfRangeRef &Ref) {
+    const unsigned SourceCluster = Ref.SourceCluster;
+    const bool IsForward = SourceCluster < Ref.TargetCluster;
+    if (LongThunkTargets[SourceCluster].contains(Ref.TargetSymbol))
+      return;
+
+    unsigned ReuseCluster = -1u;
+    if (IsForward && SourceCluster > 0)
+      ReuseCluster = SourceCluster - 1;
+    else if (!IsForward && SourceCluster + 1 < Clusters.size())
+      ReuseCluster = SourceCluster + 1;
+
+    if (ReuseCluster != -1u &&
+        LongThunkTargets[ReuseCluster].contains(Ref.TargetSymbol))
+      return;
+
+    LongThunkTargets[SourceCluster].insert(Ref.TargetSymbol);
+    Clusters[SourceCluster].EstimatedThunkBytes += LongThunkSize;
+  };
+
+  for (const OutOfRangeRef &Call : OutOfLayoutCalls)
+    estimateLongThunk(Call);
+
+  for (auto &Calls : llvm::reverse(CallsByDistance)) {
+    for (const OutOfRangeRef &Call : Calls) {
+      const unsigned Distance =
+          getClusterDistance(Call.SourceCluster, Call.TargetCluster);
+      if (Distance <= opts::MaxThunkChainLength)
+        estimateBranchThunks(Call);
+      else
+        estimateLongThunk(Call);
+    }
+  }
----------------
rafaelauler wrote:

Here you need to account for both paths, using long or short branch thunks.
```
  for (const auto &Calls : CallsByDistance)
    for (const OutOfRangeRef &Call : Calls) {
      estimateBranchThunks(Call);
      estimateLongThunk(Call);
    }
```

This estimator needs to be simple and overestimate, we can't model the usage of short thunks because that logic in itself depends on the estimated thunk size. Just overestimate.

https://github.com/llvm/llvm-project/pull/215825


More information about the llvm-commits mailing list