[llvm] [BOLT][AArch64] Relax calls and branches with fragment clusters (PR #215825)

Alexandros Lamprineas via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 02:16:41 PDT 2026


https://github.com/labrinea updated https://github.com/llvm/llvm-project/pull/215825

>From e989c9aea5e047c1eaf8db273a3033e8ffce6d8d Mon Sep 17 00:00:00 2001
From: Alexandros Lamprineas <alexandros.lamprineas at arm.com>
Date: Sun, 23 Aug 2026 19:20:43 +0000
Subject: [PATCH 1/2] [BOLT][AArch64] Relax calls and branches with fragment
 clusters

Build relaxation clusters from emitted function fragments instead of
whole functions, so split hot/cold code is modeled in the same order
it will be emitted. Use those clusters to place short and long call
thunks, and to build forward/backward branch thunk chains for
cross-cluster unconditional branches. This lets the experimental
AArch64 relaxation path handle large split binaries where direct
call/branch relocations can cross the +/-128MiB branch range after
BOLT layout changes.

Tested with stage2-clang-bolt and Chromium using Speedometer-generated
profile data, with lite mode disabled and without assuming PLT entries
are nearby.
---
 bolt/include/bolt/Core/BinaryContext.h        |   3 +
 bolt/include/bolt/Core/BinaryFunction.h       |  12 +
 bolt/include/bolt/Passes/LongJmp.h            |  75 ++-
 bolt/lib/Core/BinaryContext.cpp               | 102 ++-
 bolt/lib/Passes/LongJmp.cpp                   | 597 +++++++++++-------
 bolt/lib/Rewrite/RewriteInstance.cpp          | 104 +--
 .../AArch64/relax-branches-with-thunk-chain.s | 309 +++++++++
 .../AArch64/{relax-exp.s => relax-calls.s}    |  12 +-
 .../test/AArch64/relax-cross-fragment-calls.s | 216 +++++++
 9 files changed, 1070 insertions(+), 360 deletions(-)
 create mode 100644 bolt/test/AArch64/relax-branches-with-thunk-chain.s
 rename bolt/test/AArch64/{relax-exp.s => relax-calls.s} (86%)
 create mode 100644 bolt/test/AArch64/relax-cross-fragment-calls.s

diff --git a/bolt/include/bolt/Core/BinaryContext.h b/bolt/include/bolt/Core/BinaryContext.h
index 92cd853870cae..f0bd997effa3a 100644
--- a/bolt/include/bolt/Core/BinaryContext.h
+++ b/bolt/include/bolt/Core/BinaryContext.h
@@ -1199,6 +1199,9 @@ class BinaryContext {
     return ".text.injected.cold";
   }
 
+  /// Return true if \p A should be emitted before \p B in code section order.
+  bool compareSectionNames(StringRef A, StringRef B) const;
+
   ErrorOr<BinarySection &> getGdbIndexSection() const {
     return getUniqueSectionByName(".gdb_index");
   }
diff --git a/bolt/include/bolt/Core/BinaryFunction.h b/bolt/include/bolt/Core/BinaryFunction.h
index 14d7f9b5b5359..3865e87dc25ef 100644
--- a/bolt/include/bolt/Core/BinaryFunction.h
+++ b/bolt/include/bolt/Core/BinaryFunction.h
@@ -394,6 +394,9 @@ class BinaryFunction {
   /// True if the function is used for patching code at a fixed address.
   bool IsPatch{false};
 
+  /// True if the function is a synthetic branch/call thunk.
+  bool IsThunk{false};
+
   /// True if the original entry point of the function may get called, but the
   /// original body cannot be executed and needs to be patched with code that
   /// redirects execution to the new function body.
@@ -1504,6 +1507,9 @@ class BinaryFunction {
   /// Return true if this function is used for patching existing code.
   bool isPatch() const { return IsPatch; }
 
+  /// Return true if this function is a synthetic branch/call thunk.
+  bool isThunk() const { return IsThunk; }
+
   /// Return true if the function requires a patch.
   bool needsPatch() const { return NeedsPatch; }
 
@@ -1941,6 +1947,12 @@ class BinaryFunction {
     IsPatch = V;
   }
 
+  /// Indicate that this function is a synthetic branch/call thunk.
+  void setIsThunk(bool V) {
+    assert(isInjected() && "Only injected functions can be used as thunks");
+    IsThunk = V;
+  }
+
   /// Mark the function for patching.
   void setNeedsPatch(bool V) { NeedsPatch = V; }
 
diff --git a/bolt/include/bolt/Passes/LongJmp.h b/bolt/include/bolt/Passes/LongJmp.h
index fa8cf67a95d73..37c7e1452d5bf 100644
--- a/bolt/include/bolt/Passes/LongJmp.h
+++ b/bolt/include/bolt/Passes/LongJmp.h
@@ -10,6 +10,8 @@
 #define BOLT_PASSES_LONGJMP_H
 
 #include "bolt/Passes/BinaryPasses.h"
+#include "llvm/ADT/SmallString.h"
+#include "llvm/ADT/SmallVector.h"
 
 namespace llvm {
 namespace bolt {
@@ -80,46 +82,71 @@ class LongJmpPass : public BinaryFunctionPass {
   bool relaxLocalBranches(BinaryFunction &BF,
                           const BranchLivenessInfo *BLI = nullptr);
 
-  /// A group of functions that are located within the longest direct
-  /// branch/call instruction distance. Functions within the cluster do not
-  /// require a thunk for calls in the same cluster. The cluster may include
-  /// a set of thunks for covering calls to functions outside.
-  struct FunctionCluster {
-    /// All functions in this cluster.
-    DenseSet<BinaryFunction *> Functions;
-
-    /// Symbols corresponding to entry points of functions that this cluster
-    /// calls. Note that it excludes all functions in the cluster itself.
-    DenseSet<const MCSymbol *> Callees;
+  /// A group of function fragments that are located within the longest direct
+  /// branch/call instruction distance. Jumps within the cluster do not require
+  /// a thunk. The cluster may include thunks for jumps to targets outside.
+  struct FragmentCluster {
+    /// Output code section containing fragments in this cluster.
+    SmallString<32> SectionName;
 
     /// Estimated size of the cluster in bytes.
     uint64_t Size{0};
 
-    /// The index of the last function in the cluster. Used as an insertion
-    /// point for adding thunks to the output function list.
-    size_t LastFunctionIndex = -1;
+    /// Number of function fragments in the cluster.
+    size_t NumFragments{0};
 
-    /// When placing hot code at the end of the binary, track the first function
-    /// for insertion purposes.
+    /// The indices of the first and last functions contributing fragments to
+    /// this cluster. Used as insertion points for adding thunks to the output
+    /// function list.
     size_t FirstFunctionIndex = -1;
+    size_t LastFunctionIndex = -1;
 
-    /// Thunks located at the end of this cluster.
-    BinaryFunctionListType ThunkList;
+    /// Thunks located after this cluster.
+    BinaryFunctionListType ForwardThunkList;
 
-    /// Thunks used by this cluster. Some could be in a ThunkList of the
-    /// preceding cluster.
+    /// Thunks located before this cluster.
+    BinaryFunctionListType BackwardThunkList;
+
+    /// Call thunks used by this cluster.
     ///
     /// <Function Symbol> -> <Thunk Function>.
-    DenseMap<const MCSymbol *, BinaryFunction *> Thunks;
+    DenseMap<const MCSymbol *, BinaryFunction *> CallThunks;
+
+    /// <Destination Symbol> -> <Thunk Function>.
+    DenseMap<const MCSymbol *, BinaryFunction *> ForwardBranchThunks;
+    DenseMap<const MCSymbol *, BinaryFunction *> BackwardBranchThunks;
   };
 
   /// Maximum size of combined regular functions in the cluster. Note that it's
   /// less than 128MB, because the size of the cluster plus its thunks should be
   /// less than 128MB.
-  static constexpr uint64_t MaxClusterSize = 125 * 1024 * 1024;
+  static constexpr uint64_t MaxClusterSize = 120 * 1024 * 1024;
+
+  struct FragmentClusterLayout {
+    SmallVector<FragmentCluster, 4> Clusters;
+    DenseMap<const BinaryBasicBlock *, unsigned> BBToCluster;
+    DenseMap<const MCSymbol *, unsigned> SymToCluster;
+  };
+
+  FragmentClusterLayout
+  buildClusterLayout(BinaryContext &BC,
+                     const BinaryFunctionListType &OutputFunctions);
+
+  /// Relax calls using function fragment clusters.
+  void relaxCalls(BinaryContext &BC, BinaryFunctionListType &OutputFunctions,
+                  FragmentClusterLayout &Layout);
+
+  /// Relax direct unconditional branches using function fragment clusters.
+  void relaxUnconditionalBranches(BinaryContext &BC,
+                                  BinaryFunctionListType &OutputFunctions,
+                                  FragmentClusterLayout &Layout);
+
+  /// Insert all thunks owned by the fragment cluster layout.
+  void insertClusterThunks(BinaryFunctionListType &OutputFunctions,
+                           FragmentClusterLayout &Layout);
 
-  /// Relax calls using function cluster approach.
-  void relaxCalls(BinaryContext &BC);
+  /// Relax calls and direct unconditional branches using one cluster layout.
+  void relaxWithClusters(BinaryContext &BC);
 
   ///                 -- Layout estimation methods --
   /// Try to do layout before running the emitter, by looking at BinaryFunctions
diff --git a/bolt/lib/Core/BinaryContext.cpp b/bolt/lib/Core/BinaryContext.cpp
index d911f8191f791..94d8770dae493 100644
--- a/bolt/lib/Core/BinaryContext.cpp
+++ b/bolt/lib/Core/BinaryContext.cpp
@@ -54,6 +54,7 @@ namespace opts {
 
 extern cl::opt<bool> LargeCodeModel;
 extern cl::opt<bool> UpdateDebugSections;
+extern cl::opt<bool> HotFunctionsAtEnd;
 
 static cl::opt<bool>
     NoHugePages("no-huge-pages",
@@ -353,6 +354,103 @@ bool BinaryContext::forceSymbolRelocations(StringRef SymbolName) const {
   return false;
 }
 
+namespace {
+
+/// Defines the strict weak ordering for BOLT-produced code sections.
+class CodeSectionOrder {
+public:
+  CodeSectionOrder(StringRef ColdSectionName, StringRef HotTextMoverSectionName,
+                   StringRef MainSectionName, StringRef WarmSectionName,
+                   bool HotText, bool HotFunctionsAtEnd)
+      : ColdSectionName(ColdSectionName),
+        HotTextMoverSectionName(HotTextMoverSectionName),
+        MainSectionName(MainSectionName), WarmSectionName(WarmSectionName),
+        HotText(HotText), HotFunctionsAtEnd(HotFunctionsAtEnd) {}
+
+  bool operator()(StringRef AName, StringRef BName) const {
+    const SectionKind AKind = getKind(AName);
+    const SectionKind BKind = getKind(BName);
+    const unsigned ARank = getRank(AKind);
+    const unsigned BRank = getRank(BKind);
+    if (ARank != BRank)
+      return ARank < BRank;
+
+    if (AKind == SectionKind::Cold) {
+      if (AName.size() != BName.size())
+        return HotFunctionsAtEnd ? AName.size() > BName.size()
+                                 : AName.size() < BName.size();
+      if (AName != BName)
+        return HotFunctionsAtEnd ? AName > BName : AName < BName;
+    }
+
+    return false;
+  }
+
+private:
+  enum class SectionKind { Mover, Main, Warm, Cold, Other };
+
+  SectionKind getKind(StringRef Name) const {
+    if (HotText && Name == HotTextMoverSectionName)
+      return SectionKind::Mover;
+    if (Name == MainSectionName)
+      return SectionKind::Main;
+    if (Name == WarmSectionName)
+      return SectionKind::Warm;
+    if (Name.starts_with(ColdSectionName))
+      return SectionKind::Cold;
+    return SectionKind::Other;
+  }
+
+  unsigned getRank(SectionKind Kind) const {
+    if (Kind == SectionKind::Mover)
+      return 0;
+    if (HotFunctionsAtEnd) {
+      switch (Kind) {
+      case SectionKind::Other:
+        return 1;
+      case SectionKind::Cold:
+        return 2;
+      case SectionKind::Warm:
+        return 3;
+      case SectionKind::Main:
+        return 4;
+      case SectionKind::Mover:
+        llvm_unreachable("handled above");
+      }
+    }
+    switch (Kind) {
+    case SectionKind::Main:
+      return 1;
+    case SectionKind::Warm:
+      return 2;
+    case SectionKind::Cold:
+      return 3;
+    case SectionKind::Other:
+      return 4;
+    case SectionKind::Mover:
+      llvm_unreachable("handled above");
+    }
+    llvm_unreachable("unknown section kind");
+  }
+
+  StringRef ColdSectionName;
+  StringRef HotTextMoverSectionName;
+  StringRef MainSectionName;
+  StringRef WarmSectionName;
+  bool HotText;
+  bool HotFunctionsAtEnd;
+};
+
+} // namespace
+
+bool BinaryContext::compareSectionNames(StringRef A, StringRef B) const {
+  const CodeSectionOrder CompareSections(
+      getColdCodeSectionName(), getHotTextMoverSectionName(),
+      getMainCodeSectionName(), getWarmCodeSectionName(), opts::HotText,
+      opts::HotFunctionsAtEnd);
+  return CompareSections(A, B);
+}
+
 std::unique_ptr<MCObjectWriter>
 BinaryContext::createObjectWriter(raw_pwrite_stream &OS) {
   return MAB->createObjectWriter(OS);
@@ -2816,7 +2914,9 @@ BinaryContext::createInstructionPatch(uint64_t Address,
 BinaryFunction *
 BinaryContext::createThunkBinaryFunction(const std::string &Name) {
   static NameResolver NR;
-  return createInjectedBinaryFunction(NR.uniquify(Name));
+  BinaryFunction *BF = createInjectedBinaryFunction(NR.uniquify(Name));
+  BF->setIsThunk(true);
+  return BF;
 }
 
 std::pair<size_t, size_t>
diff --git a/bolt/lib/Passes/LongJmp.cpp b/bolt/lib/Passes/LongJmp.cpp
index 30ff9dc4ddc71..ccdb2ef37f77f 100644
--- a/bolt/lib/Passes/LongJmp.cpp
+++ b/bolt/lib/Passes/LongJmp.cpp
@@ -1023,297 +1023,434 @@ bool LongJmpPass::relaxLocalBranches(BinaryFunction &BF,
   return true;
 }
 
-void LongJmpPass::relaxCalls(BinaryContext &BC) {
-  // Operate on a copy of binary functions. We are going to manually insert new
-  // thunks and update the list.
-  BinaryFunctionListType OutputFunctions = BC.getOutputBinaryFunctions();
+static uint64_t estimateFragmentSize(const BinaryFunction &BF,
+                                     const FunctionFragment &FF) {
+  uint64_t Size = 0;
+  for (const BinaryBasicBlock *BB : FF)
+    Size += BB->estimateSize();
+
+  if (BF.hasIslandsInfo()) {
+    Size += BF.estimateConstantIslandSize();
+    if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
+      Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
+  }
 
-  // Conservatively estimate emitted function size. Assume the worst case
-  // alignment.
-  auto estimateFunctionSize = [&](const BinaryFunction &BF) -> uint64_t {
-    if (!BC.shouldEmit(BF))
-      return 0;
-    uint64_t Size = BF.estimateSize() + BF.getMaxAlignmentBytes();
-
-    // Each additional fragment can attribute extra bytes due to its alignment
-    // requirements.
-    for ([[maybe_unused]] const FunctionFragment &FF :
-         BF.getLayout().getSplitFragments())
-      Size += BF.getMaxColdAlignmentBytes();
-
-    if (BF.hasIslandsInfo()) {
-      Size += BF.estimateConstantIslandSize();
-      if (BF.getConstantIslandAlignment() > BF.getMinAlignment())
-        Size += BF.getConstantIslandAlignment() - BF.getMinAlignment();
-    }
+  Size += FF.isSplitFragment() ? BF.getMaxColdAlignmentBytes()
+                               : BF.getMaxAlignmentBytes();
+  return Size;
+}
 
-    return Size;
+LongJmpPass::FragmentClusterLayout
+LongJmpPass::buildClusterLayout(BinaryContext &BC,
+                                const BinaryFunctionListType &OutputFunctions) {
+  struct OutputFragment {
+    const FunctionFragment *FF;
+    size_t FunctionIndex;
+    SmallString<32> SectionName;
   };
 
-  // Map every function to its direct callees. Note that this is different from
-  // the regular call graph as here we completely ignore indirect calls.
+  FragmentClusterLayout Layout;
+  SmallVector<OutputFragment> OrderedFragments;
   uint64_t EstimatedSize = 0;
-  DenseMap<BinaryFunction *, std::set<const MCSymbol *>> CallMap;
-  for (BinaryFunction *BF : OutputFunctions) {
+  for (size_t I = 0; I < OutputFunctions.size(); ++I) {
+    BinaryFunction *BF = OutputFunctions[I];
     if (!BC.shouldEmit(*BF) || BF->isPatch())
       continue;
 
-    EstimatedSize += estimateFunctionSize(*BF);
-
-    for (const BinaryBasicBlock &BB : *BF) {
-      for (const MCInst &Inst : BB) {
-        if (!BC.MIB->isCall(Inst) || BC.MIB->isIndirectCall(Inst) ||
-            BC.MIB->isIndirectBranch(Inst))
-          continue;
-        const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
-        assert(TargetSymbol);
-
-        // Ignore internal calls that use basic block labels as a destination.
-        if (!BC.getFunctionForSymbol(TargetSymbol))
-          continue;
+    for (const FunctionFragment &FF : BF->getLayout().fragments()) {
+      if (FF.empty() && !BF->hasConstantIsland())
+        continue;
 
-        CallMap[BF].insert(TargetSymbol);
-      }
+      OrderedFragments.push_back(
+          {&FF, I, BF->getCodeSectionName(FF.getFragmentNum())});
     }
   }
 
-  LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << EstimatedSize
-                    << '\n');
-
-  // Build clusters in the order the functions will appear in the output.
-  std::vector<FunctionCluster> Clusters;
-  for (size_t Index = 0, NumFuncs = OutputFunctions.size(); Index < NumFuncs;
-       ++Index) {
-    const size_t BFIndex =
-        opts::HotFunctionsAtEnd ? NumFuncs - Index - 1 : Index;
-    BinaryFunction *BF = OutputFunctions[BFIndex];
-    if (!BC.shouldEmit(*BF) || BF->isPatch())
-      continue;
+  // Model final output layout by grouping function fragments in output section
+  // order. Within each section, fragments remain in OutputFunctions order.
+  llvm::stable_sort(
+      OrderedFragments, [&](const OutputFragment &A, const OutputFragment &B) {
+        return BC.compareSectionNames(A.SectionName, B.SectionName);
+      });
 
-    const uint64_t BFSize = estimateFunctionSize(*BF);
-    if (Clusters.empty() || Clusters.back().Size + BFSize > MaxClusterSize) {
-      Clusters.emplace_back(FunctionCluster());
-      Clusters.back().FirstFunctionIndex = BFIndex;
+  auto addFragmentToCluster = [&](const OutputFragment &Fragment) {
+    BinaryFunction &BF = *OutputFunctions[Fragment.FunctionIndex];
+    const FunctionFragment &FF = *Fragment.FF;
+    const uint64_t FFSize = estimateFragmentSize(BF, FF);
+
+    if (Layout.Clusters.empty() ||
+        Layout.Clusters.back().SectionName != Fragment.SectionName ||
+        Layout.Clusters.back().Size + FFSize > MaxClusterSize) {
+      Layout.Clusters.emplace_back(FragmentCluster());
+      FragmentCluster &FC = Layout.Clusters.back();
+      FC.SectionName = Fragment.SectionName;
+      FC.FirstFunctionIndex = Fragment.FunctionIndex;
     }
 
-    FunctionCluster &FC = Clusters.back();
-    FC.Functions.insert(BF);
-
-    // When a function is added to the cluster, we have to remove all of its
-    // symbols from the cluster callee list. These include alternative symbols
-    // (e.g. after ICF) and secondary entry point symbols.
-    for (const MCSymbol *Symbol : BF->getSymbols()) {
-      auto It = FC.Callees.find(Symbol);
-      if (It != FC.Callees.end())
-        FC.Callees.erase(It);
-    }
-    BF->forEachEntryPoint(
-        [&FC](uint64_t Offset, const MCSymbol *EntrySymbol) -> bool {
-          auto It = FC.Callees.find(EntrySymbol);
-          if (It != FC.Callees.end())
-            FC.Callees.erase(It);
-          return true;
-        });
-
-    // Update cluster callee list with added function callees.
-    for (const MCSymbol *CalleeSymbol : CallMap[BF]) {
-      BinaryFunction *Callee = BC.getFunctionForSymbol(CalleeSymbol);
-      if (!FC.Functions.count(Callee)) {
-        FC.Callees.insert(CalleeSymbol);
-      }
+    FragmentCluster &FC = Layout.Clusters.back();
+    FC.LastFunctionIndex = Fragment.FunctionIndex;
+    ++FC.NumFragments;
+    EstimatedSize += FFSize;
+    const unsigned ClusterNum = Layout.Clusters.size() - 1;
+
+    // Map primary entry points.
+    if (FF.isMainFragment())
+      for (const MCSymbol *Symbol : BF.getSymbols())
+        Layout.SymToCluster[Symbol] = ClusterNum;
+
+    for (const BinaryBasicBlock *BB : FF) {
+      Layout.BBToCluster[BB] = ClusterNum;
+      if (const MCSymbol *Label = BB->getLabel())
+        Layout.SymToCluster[Label] = ClusterNum;
+
+      // Map secondary entry points.
+      if (MCSymbol *EntrySymbol = BF.getSecondaryEntryPointSymbol(*BB))
+        Layout.SymToCluster[EntrySymbol] = ClusterNum;
     }
 
-    FC.Size += BFSize;
-    FC.LastFunctionIndex = BFIndex;
-  }
+    FC.Size += FFSize;
+  };
 
-  if (opts::HotFunctionsAtEnd) {
-    std::reverse(Clusters.begin(), Clusters.end());
-    llvm::for_each(Clusters, [](FunctionCluster &FC) {
-      std::swap(FC.LastFunctionIndex, FC.FirstFunctionIndex);
-    });
-  }
+  for (const OutputFragment &Fragment : OrderedFragments)
+    addFragmentToCluster(Fragment);
 
-  if (Clusters.empty())
-    return;
+  if (Layout.Clusters.empty())
+    return Layout;
+
+  LLVM_DEBUG(dbgs() << "LongJmp: estimated code size : " << EstimatedSize
+                    << '\n');
 
   // Print cluster stats.
-  BC.outs() << "BOLT-INFO: built " << Clusters.size()
-            << " function cluster(s)\n";
-  uint64_t ClusterIndex = 0;
-  for (const FunctionCluster &FC : Clusters) {
-    BC.outs() << "BOLT-INFO: cluster: " << ClusterIndex++ << '\n'
-              << "BOLT-INFO:   " << FC.Functions.size() << " function(s)\n"
-              << "BOLT-INFO:   " << FC.Callees.size() << " callee(s)\n"
+  BC.outs() << "BOLT-INFO: built " << Layout.Clusters.size()
+            << " function fragment cluster(s)\n";
+  for (size_t I = 0; I < Layout.Clusters.size(); ++I) {
+    const FragmentCluster &FC = Layout.Clusters[I];
+    BC.outs() << "BOLT-INFO: cluster: " << I << '\n'
+              << "BOLT-INFO:   " << FC.NumFragments << " fragment(s)\n"
               << "BOLT-INFO:   " << FC.Size << " estimated bytes\n";
   }
 
   if (opts::RelaxPLT) {
     // Populate one of the clusters with PLT functions based on the proximity of
     // the PLT section to avoid unneeded thunk redirection.
-    const size_t PLTClusterNum = opts::UseOldText ? Clusters.size() - 1 : 0;
-    auto &PLTCluster = Clusters[PLTClusterNum];
-    for (BinaryFunction &BF :
-         llvm::make_second_range(BC.getBinaryFunctions())) {
+    const unsigned PLTCluster =
+        opts::UseOldText ? Layout.Clusters.size() - 1 : 0;
+    for (BinaryFunction &BF : llvm::make_second_range(BC.getBinaryFunctions()))
       if (BF.isPLTFunction()) {
-        PLTCluster.Functions.insert(&BF);
-        auto It = PLTCluster.Callees.find(BF.getSymbol());
-        if (It != PLTCluster.Callees.end())
-          PLTCluster.Callees.erase(It);
+        for (const MCSymbol *Symbol : BF.getSymbols())
+          Layout.SymToCluster[Symbol] = PLTCluster;
+        BF.forEachEntryPoint(
+            [&](uint64_t, const MCSymbol *EntrySymbol) -> bool {
+              Layout.SymToCluster[EntrySymbol] = PLTCluster;
+              return true;
+            });
       }
-    }
   }
 
-  // Create a thunk with +-128MB span.
-  size_t NumShortThunks = 0;
-  auto createShortThunk = [&](const MCSymbol *TargetSymbol) {
-    ++NumShortThunks;
-    BinaryFunction *ThunkBF = BC.createThunkBinaryFunction(
-        "__AArch64Thunk_" + TargetSymbol->getName().str());
-    MCInst Inst;
-    BC.MIB->createTailCall(Inst, TargetSymbol, BC.Ctx.get());
-    ThunkBF->addBasicBlock()->addInstruction(Inst);
+  return Layout;
+}
 
-    return ThunkBF;
+void LongJmpPass::relaxCalls(BinaryContext &BC,
+                             BinaryFunctionListType &OutputFunctions,
+                             FragmentClusterLayout &Layout) {
+  auto &Clusters = Layout.Clusters;
+  auto &BBToCluster = Layout.BBToCluster;
+  auto &SymToCluster = Layout.SymToCluster;
+
+  struct CrossClusterCall {
+    MCInst *Inst;
+    const MCSymbol *TargetSymbol;
+    unsigned SourceCluster;
+    unsigned TargetCluster;
   };
 
-  // Create a thunk with +-4GB span.
-  size_t NumLongThunks = 0;
-  auto createLongThunk = [&](const MCSymbol *TargetSymbol) {
-    ++NumLongThunks;
-    BinaryFunction *ThunkBF = BC.createThunkBinaryFunction(
-        "__AArch64ADRPThunk_" + TargetSymbol->getName().str());
-    InstructionListType Instructions;
-    BC.MIB->createLongTailCall(Instructions, TargetSymbol, BC.Ctx.get());
-    ThunkBF->addBasicBlock()->addInstructions(Instructions);
+  SmallVector<CrossClusterCall> CrossClusterCalls;
+  for (BinaryFunction *BF : OutputFunctions) {
+    if (!BC.shouldEmit(*BF) || BF->isPatch())
+      continue;
 
-    return ThunkBF;
-  };
+    for (BinaryBasicBlock &BB : *BF) {
+      auto SourceIt = BBToCluster.find(&BB);
+      if (SourceIt == BBToCluster.end())
+        continue;
+      const unsigned SourceCluster = SourceIt->second;
 
-  for (unsigned ClusterNum = 0; ClusterNum < Clusters.size(); ++ClusterNum) {
-    FunctionCluster &FC = Clusters[ClusterNum];
-    SmallVector<const MCSymbol *, 16> Callees(FC.Callees.begin(),
-                                              FC.Callees.end());
-
-    // Generate thunks in deterministic order.
-    llvm::sort(Callees, [&BC](const MCSymbol *A, const MCSymbol *B) {
-      uint64_t EntryA;
-      uint64_t EntryB;
-      BinaryFunction *BFA = BC.getFunctionForSymbol(A, &EntryA);
-      BinaryFunction *BFB = BC.getFunctionForSymbol(B, &EntryB);
-      if (BFA == BFB) {
-        if (EntryA != EntryB)
-          return EntryA < EntryB;
-
-        // Use lexicographical order for ICF'ed symbols.
-        return A->getName() < B->getName();
-      }
-      return compareBinaryFunctionByIndex(BFA, BFB);
-    });
+      for (MCInst &Inst : BB) {
+        if (!BC.MIB->isCall(Inst) && !BC.MIB->isUnconditionalBranch(Inst))
+          continue;
 
-    // Return index of adjacent cluster containing the function.
-    auto getAdjClusterWithFunction =
-        [&](const BinaryFunction *BF) -> std::optional<unsigned> {
-      if (ClusterNum > 0 && Clusters[ClusterNum - 1].Functions.count(BF))
-        return ClusterNum - 1;
-      if (ClusterNum + 1 < Clusters.size() &&
-          Clusters[ClusterNum + 1].Functions.count(BF))
-        return ClusterNum + 1;
-      return std::nullopt;
-    };
+        const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
+        if (!TargetSymbol)
+          continue;
 
-    const FunctionCluster *PrevCluster =
-        ClusterNum ? &Clusters[ClusterNum - 1] : nullptr;
+        auto It = SymToCluster.find(TargetSymbol);
+        const bool Found = It != SymToCluster.end();
+        // Known plain branches use B-only thunk chains.
+        // Unknown address-only branches may use tail-call thunks.
+        if (BC.MIB->isUnconditionalBranch(Inst) && Found)
+          continue;
 
-    // Create short thunks for callees in adjacent clusters and long thunks
-    // for callees outside.
-    for (const MCSymbol *Callee : Callees) {
-      if (FC.Thunks.count(Callee))
-        continue;
+        if (!BC.getFunctionForSymbol(TargetSymbol) &&
+            !BC.getSymbolValue(*TargetSymbol))
+          continue;
 
-      BinaryFunction *Thunk = 0;
-      std::optional<unsigned> AdjCluster =
-          getAdjClusterWithFunction(BC.getFunctionForSymbol(Callee));
-      if (AdjCluster) {
-        Thunk = createShortThunk(Callee);
-      } else {
-        // Previous cluster may already have a long thunk that can be reused.
-        if (PrevCluster) {
-          auto It = PrevCluster->Thunks.find(Callee);
-          // Reuse only if previous cluster hosts this thunk.
-          if (It != PrevCluster->Thunks.end() &&
-              llvm::is_contained(PrevCluster->ThunkList, It->second)) {
-            FC.Thunks[Callee] = It->second;
-            continue;
-          }
-        }
-        Thunk = createLongThunk(Callee);
-      }
+        // If not found TargetCluster becomes UINT_MAX.
+        unsigned TargetCluster = Found ? It->second : -1;
+        if (TargetCluster == SourceCluster)
+          continue;
 
-      // The cluster that will host this thunk. If the current cluster is the
-      // last one, try to use the previous one. Matters when we want to have hot
-      // functions at higher addresses under HotFunctionsAtEnd.
-      FunctionCluster *ThunkCluster = &Clusters[ClusterNum];
-      if ((AdjCluster && *AdjCluster == ClusterNum - 1) ||
-          (ClusterNum && ClusterNum == Clusters.size() - 1))
-        ThunkCluster = &Clusters[ClusterNum - 1];
-      ThunkCluster->ThunkList.push_back(Thunk);
-
-      // Register thunks for all symbols associated with the function.
-      uint64_t EntryID = 0;
-      const BinaryFunction *BF = BC.getFunctionForSymbol(Callee, &EntryID);
-      if (EntryID != 0) {
-        FC.Thunks[Callee] = Thunk;
-      } else {
-        for (const MCSymbol *Symbol : BF->getSymbols()) {
-          FC.Thunks[Symbol] = Thunk;
-        }
+        CrossClusterCalls.push_back(
+            {&Inst, TargetSymbol, SourceCluster, TargetCluster});
       }
     }
   }
 
+  size_t NumShortThunks = 0;
+  size_t NumLongThunks = 0;
+  auto createCallThunk = [&](const MCSymbol *TargetSymbol, bool IsShort) {
+    BinaryFunction *Thunk = nullptr;
+    if (IsShort) {
+      ++NumShortThunks;
+      Thunk = BC.createThunkBinaryFunction("__AArch64Thunk_" +
+                                           TargetSymbol->getName().str());
+      MCInst Inst;
+      BC.MIB->createTailCall(Inst, TargetSymbol, BC.Ctx.get());
+      Thunk->addBasicBlock()->addInstruction(Inst);
+    } else {
+      ++NumLongThunks;
+      Thunk = BC.createThunkBinaryFunction("__AArch64ADRPThunk_" +
+                                           TargetSymbol->getName().str());
+      InstructionListType Instructions;
+      BC.MIB->createLongTailCall(Instructions, TargetSymbol, BC.Ctx.get());
+      Thunk->addBasicBlock()->addInstructions(Instructions);
+    }
+    return Thunk;
+  };
+
+  auto registerCallThunk = [&](FragmentCluster &FC, const MCSymbol *Callee,
+                               BinaryFunction *Thunk) {
+    uint64_t EntryID = 0;
+    const BinaryFunction *BF = BC.getFunctionForSymbol(Callee, &EntryID);
+    const bool IsPrimaryEntry = EntryID == 0;
+    if (BF && IsPrimaryEntry)
+      for (const MCSymbol *Symbol : BF->getSymbols())
+        FC.CallThunks[Symbol] = Thunk;
+    else
+      FC.CallThunks[Callee] = Thunk;
+  };
+
+  auto getOrCreateCallThunk = [&](const MCSymbol *TargetSymbol,
+                                  unsigned SourceCluster, bool IsShort,
+                                  bool IsForward) {
+    FragmentCluster &FC = Clusters[SourceCluster];
+    if (auto It = FC.CallThunks.find(TargetSymbol); It != FC.CallThunks.end())
+      return It->second;
+
+    BinaryFunction *Thunk = createCallThunk(TargetSymbol, IsShort);
+
+    FragmentCluster &ThunkCluster = Clusters[SourceCluster];
+    Thunk->setCodeSectionName(ThunkCluster.SectionName);
+    BinaryFunctionListType &ThunkList = IsForward
+                                            ? ThunkCluster.ForwardThunkList
+                                            : ThunkCluster.BackwardThunkList;
+    ThunkList.push_back(Thunk);
+
+    // Register thunks for all symbols associated with the function.
+    registerCallThunk(FC, TargetSymbol, Thunk);
+    return Thunk;
+  };
+
+  auto areAdjacent = [](unsigned A, unsigned B) {
+    return A > B ? A - B == 1 : B - A == 1;
+  };
+
+  for (CrossClusterCall &Call : CrossClusterCalls) {
+    const bool IsForward = Call.SourceCluster < Call.TargetCluster;
+    const bool Adjacent = areAdjacent(Call.SourceCluster, Call.TargetCluster);
+
+    BinaryFunction *Thunk = getOrCreateCallThunk(
+        Call.TargetSymbol, Call.SourceCluster, Adjacent, IsForward);
+    BC.MIB->replaceBranchTarget(*Call.Inst, Thunk->getSymbol(), BC.Ctx.get());
+  }
+
+  if (!CrossClusterCalls.empty())
+    BC.outs() << "BOLT-INFO: relaxed " << CrossClusterCalls.size()
+              << " calls with thunks\n";
+
   if (NumShortThunks)
     BC.outs() << "BOLT-INFO: " << NumShortThunks << " short thunks created\n";
 
   if (NumLongThunks)
     BC.outs() << "BOLT-INFO: " << NumLongThunks << " long thunks created\n";
+}
 
-  // Replace callees with thunks.
-  for (FunctionCluster &FC : Clusters) {
-    for (BinaryFunction *BF : FC.Functions) {
-      if (!CallMap.count(BF))
+void LongJmpPass::relaxUnconditionalBranches(
+    BinaryContext &BC, BinaryFunctionListType &OutputFunctions,
+    FragmentClusterLayout &Layout) {
+  auto &Clusters = Layout.Clusters;
+  auto &BBToCluster = Layout.BBToCluster;
+  auto &SymToCluster = Layout.SymToCluster;
+
+  struct CrossClusterBranch {
+    MCInst *Inst;
+    const MCSymbol *TargetSymbol;
+    unsigned SourceCluster;
+    unsigned TargetCluster;
+  };
+
+  SmallVector<CrossClusterBranch> CrossClusterBranches;
+  for (BinaryFunction *BF : OutputFunctions) {
+    if (!BC.shouldEmit(*BF) || BF->isPatch())
+      continue;
+
+    for (BinaryBasicBlock &BB : *BF) {
+      auto SourceIt = BBToCluster.find(&BB);
+      if (SourceIt == BBToCluster.end())
         continue;
+      const unsigned SourceCluster = SourceIt->second;
 
-      for (BinaryBasicBlock &BB : *BF) {
-        for (MCInst &Inst : BB) {
-          if (!BC.MIB->isCall(Inst) || BC.MIB->isIndirectCall(Inst) ||
-              BC.MIB->isIndirectBranch(Inst))
-            continue;
-          const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
-          assert(TargetSymbol);
+      for (MCInst &Inst : BB) {
+        if (!BC.MIB->isUnconditionalBranch(Inst))
+          continue;
 
-          auto It = FC.Thunks.find(TargetSymbol);
-          if (It != FC.Thunks.end())
-            BC.MIB->replaceBranchTarget(Inst, It->second->getSymbol(),
-                                        BC.Ctx.get());
-        }
+        const MCSymbol *TargetSymbol = BC.MIB->getTargetSymbol(Inst);
+        if (!TargetSymbol)
+          continue;
+
+        auto TargetIt = SymToCluster.find(TargetSymbol);
+        if (TargetIt == SymToCluster.end())
+          continue;
+
+        const unsigned TargetCluster = TargetIt->second;
+        if (SourceCluster == TargetCluster)
+          continue;
+
+        CrossClusterBranches.push_back(
+            {&Inst, TargetSymbol, SourceCluster, TargetCluster});
       }
     }
   }
 
-  // Add thunks to the function list and assign a section name matching the
-  // function they follow.
-  for (const FunctionCluster &FC : llvm::reverse(Clusters)) {
-    std::string SectionName =
-        OutputFunctions[FC.LastFunctionIndex]->getCodeSectionName().str().str();
-    for (BinaryFunction *Thunk : FC.ThunkList) {
-      Thunk->setCodeSectionName(SectionName);
+  size_t NumBranchThunks = 0;
+  auto createBranchThunk = [&](const MCSymbol *TargetSymbol,
+                               const bool IsForward) {
+    std::string ThunkName = IsForward ? "__AArch64BranchForwardThunk_"
+                                      : "__AArch64BranchBackwardThunk_";
+    ThunkName += std::to_string(NumBranchThunks++);
+
+    BinaryFunction *ThunkBF = BC.createThunkBinaryFunction(ThunkName);
+    MCInst Inst;
+    BC.MIB->createUncondBranch(Inst, TargetSymbol, BC.Ctx.get());
+    ThunkBF->addBasicBlock()->addInstruction(Inst);
+    return ThunkBF;
+  };
+
+  auto getOrCreateBranchThunk =
+      [&](FragmentCluster &Cluster, const MCSymbol *TargetSymbol,
+          const MCSymbol *NextTarget, const bool IsForward) {
+        auto &Thunks = IsForward ? Cluster.ForwardBranchThunks
+                                 : Cluster.BackwardBranchThunks;
+        auto It = Thunks.find(TargetSymbol);
+        if (It != Thunks.end())
+          return It->second;
+
+        BinaryFunction *Thunk = createBranchThunk(NextTarget, IsForward);
+        Thunk->setCodeSectionName(Cluster.SectionName);
+        auto &ThunkList =
+            IsForward ? Cluster.ForwardThunkList : Cluster.BackwardThunkList;
+        ThunkList.push_back(Thunk);
+        Thunks[TargetSymbol] = Thunk;
+        return Thunk;
+      };
+
+  auto getOrCreateBranchThunkChain =
+      [&](const CrossClusterBranch &Branch) -> const MCSymbol * {
+    const unsigned SourceCluster = Branch.SourceCluster;
+    const unsigned TargetCluster = Branch.TargetCluster;
+    BinaryFunction *FirstThunk = nullptr;
+    const MCSymbol *NextTarget = Branch.TargetSymbol;
+
+    if (SourceCluster < TargetCluster) {
+      for (unsigned Cluster = TargetCluster; Cluster > SourceCluster;) {
+        --Cluster;
+        FirstThunk =
+            getOrCreateBranchThunk(Clusters[Cluster], Branch.TargetSymbol,
+                                   NextTarget, /*IsForward=*/true);
+        NextTarget = FirstThunk->getSymbol();
+      }
+    } else {
+      for (unsigned Cluster = TargetCluster + 1; Cluster <= SourceCluster;
+           ++Cluster) {
+        FirstThunk =
+            getOrCreateBranchThunk(Clusters[Cluster], Branch.TargetSymbol,
+                                   NextTarget, /*IsForward=*/false);
+        NextTarget = FirstThunk->getSymbol();
+      }
     }
 
+    assert(FirstThunk && "expected branch thunk chain");
+    return FirstThunk->getSymbol();
+  };
+
+  for (const CrossClusterBranch &Branch : CrossClusterBranches) {
+    const MCSymbol *Target = getOrCreateBranchThunkChain(Branch);
+    BC.MIB->replaceBranchTarget(*Branch.Inst, Target, BC.Ctx.get());
+  }
+
+  if (!CrossClusterBranches.empty())
+    BC.outs() << "BOLT-INFO: relaxed " << CrossClusterBranches.size()
+              << " cross-cluster branches\n";
+
+  if (NumBranchThunks)
+    BC.outs() << "BOLT-INFO: " << NumBranchThunks << " branch thunks created\n";
+}
+
+void LongJmpPass::insertClusterThunks(BinaryFunctionListType &OutputFunctions,
+                                      FragmentClusterLayout &Layout) {
+  struct ThunkInsertion {
+    size_t Position;
+    bool IsForward;
+    BinaryFunctionListType *ThunkList;
+  };
+
+  SmallVector<ThunkInsertion> Insertions;
+  for (FragmentCluster &Cluster : Layout.Clusters) {
+    if (!Cluster.BackwardThunkList.empty())
+      Insertions.push_back({Cluster.FirstFunctionIndex, /*IsForward=*/false,
+                            &Cluster.BackwardThunkList});
+
+    if (!Cluster.ForwardThunkList.empty())
+      Insertions.push_back({Cluster.LastFunctionIndex + 1,
+                            /*IsForward=*/true, &Cluster.ForwardThunkList});
+  }
+
+  // Apply insertions from high to low indices so earlier insertions do not
+  // invalidate later positions. At a shared boundary, repeated insertion at the
+  // same index reverses application order, yielding forward thunks before
+  // backward thunks.
+  llvm::sort(Insertions, [](const ThunkInsertion &A, const ThunkInsertion &B) {
+    if (A.Position != B.Position)
+      return A.Position > B.Position;
+    return !A.IsForward && B.IsForward;
+  });
+
+  for (ThunkInsertion &Insertion : Insertions) {
     OutputFunctions.insert(
-        std::next(OutputFunctions.begin(), FC.LastFunctionIndex + 1),
-        FC.ThunkList.begin(), FC.ThunkList.end());
+        std::next(OutputFunctions.begin(), Insertion.Position),
+        Insertion.ThunkList->begin(), Insertion.ThunkList->end());
   }
+}
+
+void LongJmpPass::relaxWithClusters(BinaryContext &BC) {
+  BinaryFunctionListType OutputFunctions = BC.getOutputBinaryFunctions();
+  FragmentClusterLayout Layout = buildClusterLayout(BC, OutputFunctions);
+
+  if (Layout.Clusters.empty())
+    return;
+
+  relaxCalls(BC, OutputFunctions, Layout);
+  relaxUnconditionalBranches(BC, OutputFunctions, Layout);
+  insertClusterThunks(OutputFunctions, Layout);
 
   LLVM_DEBUG(dbgs() << "\nFunction layout with thunks:\n";
              for (const auto *BF : OutputFunctions) { dbgs() << *BF << '\n'; });
@@ -1375,7 +1512,7 @@ Error LongJmpPass::runOnFunctions(BinaryContext &BC) {
       return Error::success();
 
     BC.outs() << "BOLT-INFO: starting experimental relaxation pass\n";
-    relaxCalls(BC);
+    relaxWithClusters(BC);
     return Error::success();
   }
 
diff --git a/bolt/lib/Rewrite/RewriteInstance.cpp b/bolt/lib/Rewrite/RewriteInstance.cpp
index 512af53f0ec59..1d1728b868bdf 100644
--- a/bolt/lib/Rewrite/RewriteInstance.cpp
+++ b/bolt/lib/Rewrite/RewriteInstance.cpp
@@ -4363,111 +4363,17 @@ void RewriteInstance::mapFileSections(BOLTLinker::SectionMapper MapSection) {
   }
 }
 
-namespace {
-
-/// Defines the strict weak ordering for BOLT-produced code sections.
-class CodeSectionOrder {
-public:
-  CodeSectionOrder(StringRef ColdSectionName, StringRef HotTextMoverSectionName,
-                   StringRef MainSectionName, StringRef WarmSectionName,
-                   bool HotText, bool HotFunctionsAtEnd)
-      : ColdSectionName(ColdSectionName),
-        HotTextMoverSectionName(HotTextMoverSectionName),
-        MainSectionName(MainSectionName), WarmSectionName(WarmSectionName),
-        HotText(HotText), HotFunctionsAtEnd(HotFunctionsAtEnd) {}
-
-  bool operator()(StringRef AName, StringRef BName) const {
-    const SectionKind AKind = getKind(AName);
-    const SectionKind BKind = getKind(BName);
-    const unsigned ARank = getRank(AKind);
-    const unsigned BRank = getRank(BKind);
-    if (ARank != BRank)
-      return ARank < BRank;
-
-    if (AKind == SectionKind::Cold) {
-      if (AName.size() != BName.size())
-        return HotFunctionsAtEnd ? AName.size() > BName.size()
-                                 : AName.size() < BName.size();
-      if (AName != BName)
-        return HotFunctionsAtEnd ? AName > BName : AName < BName;
-    }
-
-    return false;
-  }
-
-private:
-  enum class SectionKind { Mover, Main, Warm, Cold, Other };
-
-  SectionKind getKind(StringRef Name) const {
-    if (HotText && Name == HotTextMoverSectionName)
-      return SectionKind::Mover;
-    if (Name == MainSectionName)
-      return SectionKind::Main;
-    if (Name == WarmSectionName)
-      return SectionKind::Warm;
-    if (Name.starts_with(ColdSectionName))
-      return SectionKind::Cold;
-    return SectionKind::Other;
-  }
-
-  unsigned getRank(SectionKind Kind) const {
-    if (Kind == SectionKind::Mover)
-      return 0;
-    if (HotFunctionsAtEnd) {
-      switch (Kind) {
-      case SectionKind::Other:
-        return 1;
-      case SectionKind::Cold:
-        return 2;
-      case SectionKind::Warm:
-        return 3;
-      case SectionKind::Main:
-        return 4;
-      case SectionKind::Mover:
-        llvm_unreachable("handled above");
-      }
-    }
-    switch (Kind) {
-    case SectionKind::Main:
-      return 1;
-    case SectionKind::Warm:
-      return 2;
-    case SectionKind::Cold:
-      return 3;
-    case SectionKind::Other:
-      return 4;
-    case SectionKind::Mover:
-      llvm_unreachable("handled above");
-    }
-    llvm_unreachable("unknown section kind");
-  }
-
-  StringRef ColdSectionName;
-  StringRef HotTextMoverSectionName;
-  StringRef MainSectionName;
-  StringRef WarmSectionName;
-  bool HotText;
-  bool HotFunctionsAtEnd;
-};
-
-} // namespace
-
 std::vector<BinarySection *> RewriteInstance::getCodeSections() {
   std::vector<BinarySection *> CodeSections;
   for (BinarySection &Section : BC->textSections())
     if (Section.hasValidSectionID())
       CodeSections.emplace_back(&Section);
 
-  const CodeSectionOrder CompareSections(
-      BC->getColdCodeSectionName(), BC->getHotTextMoverSectionName(),
-      BC->getMainCodeSectionName(), BC->getWarmCodeSectionName(), opts::HotText,
-      opts::HotFunctionsAtEnd);
-
   // Determine the order of sections.
-  llvm::stable_sort(CodeSections,
-                    [&](const BinarySection *A, const BinarySection *B) {
-                      return CompareSections(A->getName(), B->getName());
-                    });
+  llvm::stable_sort(
+      CodeSections, [&](const BinarySection *A, const BinarySection *B) {
+        return BC->compareSectionNames(A->getName(), B->getName());
+      });
 
 #ifndef NDEBUG
   // Verify that the order of sections and functions is consistent.
@@ -4477,7 +4383,7 @@ std::vector<BinarySection *> RewriteInstance::getCodeSections() {
 
   uint32_t LastIndex = 0;
   for (const BinaryFunction *BF : BC->getOutputBinaryFunctions()) {
-    if (!BF->isEmitted() || BF->isPatch())
+    if (!BF->isEmitted() || BF->isPatch() || BF->isThunk())
       continue;
 
     ErrorOr<BinarySection &> Sec = BF->getCodeSection();
diff --git a/bolt/test/AArch64/relax-branches-with-thunk-chain.s b/bolt/test/AArch64/relax-branches-with-thunk-chain.s
new file mode 100644
index 0000000000000..6c46afdb1c40a
--- /dev/null
+++ b/bolt/test/AArch64/relax-branches-with-thunk-chain.s
@@ -0,0 +1,309 @@
+## Check branch thunk chains with adjacent fragment clusters. Split functions
+## have small duplicated constant islands, while hot-only pad functions create
+## most of the hot-section distance.
+##
+## Input layout:
+##
+##   A  8MiB island, pad_hot_0 50MiB, B 8MiB island, pad_hot_1 50MiB,
+##   C  8MiB island, pad_hot_2 50MiB, D 8MiB island, pad_hot_3 50MiB
+##
+## With --split-functions, BOLT places the cold blocks of A/B/C/D in
+## .text.cold. With the default 120MiB function-fragment cluster size, branch
+## relaxation sees:
+##
+##   normal layout:
+##     cluster 0, .text:      A, pad_hot_0, B, pad_hot_1
+##     cluster 1, .text:      C, pad_hot_2, D, pad_hot_3
+##     cluster 2, .text.cold: A.cold, B.cold, C.cold, D.cold
+##
+##   --hot-functions-at-end:
+##     cluster 0, .text.cold: A.cold, B.cold, C.cold, D.cold
+##     cluster 1, .text:      A, pad_hot_0, B, pad_hot_1
+##     cluster 2, .text:      C, pad_hot_2, D, pad_hot_3
+
+# REQUIRES: system-linux
+
+# RUN: %clang %cflags -Wl,-q -Wl,-e,A %s -o %t -nostdlib
+# RUN: link_fdata --no-lbr %s %t %t.fdata
+# RUN: llvm-strip --strip-unneeded %t
+# RUN: llvm-bolt %t -o %t.bolt --data %t.fdata --split-functions \
+# RUN:   --compact-code-model --relax-exp \
+# RUN:   | FileCheck %s --check-prefix=CHECK-BOLT
+# RUN: llvm-bolt %t -o %t.hfe.bolt --data %t.fdata --split-functions \
+# RUN:   --compact-code-model --relax-exp --hot-functions-at-end \
+# RUN:   | FileCheck %s --check-prefix=CHECK-BOLT-HFE
+# RUN: llvm-readelf -S %t.bolt | FileCheck %s --check-prefix=CHECK-SECTIONS
+# RUN: llvm-objdump -d \
+# RUN:   --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64BranchForwardThunk_0,__AArch64BranchForwardThunk_1,__AArch64BranchForwardThunk_4,__AArch64BranchForwardThunk_5,__AArch64BranchForwardThunk_8,__AArch64BranchForwardThunk_10,__AArch64BranchBackwardThunk_2,__AArch64BranchBackwardThunk_3,__AArch64BranchBackwardThunk_6,__AArch64BranchBackwardThunk_7,__AArch64BranchBackwardThunk_9,__AArch64BranchBackwardThunk_11 \
+# RUN:   %t.bolt | FileCheck %s --check-prefix=CHECK-OUTPUT
+# RUN: llvm-objdump -d \
+# RUN:   --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64BranchForwardThunk_1,__AArch64BranchForwardThunk_3,__AArch64BranchForwardThunk_6,__AArch64BranchForwardThunk_7,__AArch64BranchForwardThunk_10,__AArch64BranchForwardThunk_11,__AArch64BranchBackwardThunk_0,__AArch64BranchBackwardThunk_2,__AArch64BranchBackwardThunk_4,__AArch64BranchBackwardThunk_5,__AArch64BranchBackwardThunk_8,__AArch64BranchBackwardThunk_9 \
+# RUN:   %t.hfe.bolt | FileCheck %s --check-prefix=CHECK-HFE-OUTPUT
+
+# CHECK-BOLT: BOLT-INFO: built 3 function fragment cluster(s)
+# CHECK-BOLT: BOLT-INFO: relaxed 8 cross-cluster branches
+# CHECK-BOLT: BOLT-INFO: 12 branch thunks created
+
+# CHECK-BOLT-HFE: BOLT-INFO: built 3 function fragment cluster(s)
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 8 cross-cluster branches
+# CHECK-BOLT-HFE: BOLT-INFO: 12 branch thunks created
+
+# CHECK-SECTIONS: .text
+# CHECK-SECTIONS: .text.cold
+
+  .text
+  .globl A
+  .type A, %function
+A:
+.A_entry:
+# FDATA: 1 A #.A_entry# 100
+  cbz x0, .A_ret
+  b .A_cold
+.A_ret:
+# FDATA: 1 A #.A_ret# 100
+  ret
+.A_cold:
+  mov x0, #1
+  b .A_ret
+  .space 0x800000
+  .size A, .-A
+
+  .globl pad_hot_0
+  .type pad_hot_0, %function
+pad_hot_0:
+.pad_hot_0_entry:
+# FDATA: 1 pad_hot_0 #.pad_hot_0_entry# 100
+  ret
+  .space 0x3200000
+  .size pad_hot_0, .-pad_hot_0
+
+  .globl B
+  .type B, %function
+B:
+.B_entry:
+# FDATA: 1 B #.B_entry# 100
+  cbz x0, .B_ret
+  b .B_cold
+.B_ret:
+# FDATA: 1 B #.B_ret# 100
+  ret
+.B_cold:
+  mov x0, #2
+  b .B_ret
+  .space 0x800000
+  .size B, .-B
+
+  .globl pad_hot_1
+  .type pad_hot_1, %function
+pad_hot_1:
+.pad_hot_1_entry:
+# FDATA: 1 pad_hot_1 #.pad_hot_1_entry# 100
+  ret
+  .space 0x3200000
+  .size pad_hot_1, .-pad_hot_1
+
+  .globl C
+  .type C, %function
+C:
+.C_entry:
+# FDATA: 1 C #.C_entry# 100
+  cbz x0, .C_ret
+  b .C_cold
+.C_ret:
+# FDATA: 1 C #.C_ret# 100
+  ret
+.C_cold:
+  mov x0, #3
+  b .C_ret
+  .space 0x800000
+  .size C, .-C
+
+  .globl pad_hot_2
+  .type pad_hot_2, %function
+pad_hot_2:
+.pad_hot_2_entry:
+# FDATA: 1 pad_hot_2 #.pad_hot_2_entry# 100
+  ret
+  .space 0x3200000
+  .size pad_hot_2, .-pad_hot_2
+
+  .globl D
+  .type D, %function
+D:
+.D_entry:
+# FDATA: 1 D #.D_entry# 100
+  cbz x0, .D_ret
+  b .D_cold
+.D_ret:
+# FDATA: 1 D #.D_ret# 100
+  ret
+.D_cold:
+  mov x0, #4
+  b .D_ret
+  .space 0x800000
+  .size D, .-D
+
+  .globl pad_hot_3
+  .type pad_hot_3, %function
+pad_hot_3:
+.pad_hot_3_entry:
+# FDATA: 1 pad_hot_3 #.pad_hot_3_entry# 100
+  ret
+  .space 0x3200000
+  .size pad_hot_3, .-pad_hot_3
+
+## Force relocation mode.
+  .reloc 0, R_AARCH64_NONE
+
+# CHECK-OUTPUT: Disassembly of section .text:
+
+# CHECK-OUTPUT:      <A>:
+# CHECK-OUTPUT-NEXT:                      {{.*}} cbnz x0, 0x[[A_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-OUTPUT-NEXT: [[A_RET:[0-9a-f]+]]: {{.*}} ret
+# CHECK-OUTPUT-NEXT: [[A_BR]]:            {{.*}} b    0x[[A_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_1>
+
+# CHECK-OUTPUT:      <B>:
+# CHECK-OUTPUT-NEXT:                      {{.*}} cbnz x0, 0x[[B_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-OUTPUT-NEXT: [[B_RET:[0-9a-f]+]]: {{.*}} ret
+# CHECK-OUTPUT-NEXT: [[B_BR]]:            {{.*}} b    0x[[B_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_5>
+
+# CHECK-OUTPUT:      <__AArch64BranchForwardThunk_1>:
+# CHECK-OUTPUT-NEXT: [[A_FW0]]:           {{.*}} b    0x[[A_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_0>
+
+# CHECK-OUTPUT:      <__AArch64BranchForwardThunk_5>:
+# CHECK-OUTPUT-NEXT: [[B_FW0]]:           {{.*}} b    0x[[B_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_4>
+
+# CHECK-OUTPUT:      <__AArch64BranchBackwardThunk_2>:
+# CHECK-OUTPUT-NEXT: [[A_BW1:[0-9a-f]+]]: {{.*}} b    0x[[A_RET]] <A+0x4>
+
+# CHECK-OUTPUT:      <__AArch64BranchBackwardThunk_6>:
+# CHECK-OUTPUT-NEXT: [[B_BW1:[0-9a-f]+]]: {{.*}} b    0x[[B_RET]] <B+0x4>
+
+# CHECK-OUTPUT:      <C>:
+# CHECK-OUTPUT-NEXT:                      {{.*}} cbnz x0, 0x[[C_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-OUTPUT-NEXT: [[C_RET:[0-9a-f]+]]: {{.*}} ret
+# CHECK-OUTPUT-NEXT: [[C_BR]]:            {{.*}} b    0x[[C_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_8>
+
+# CHECK-OUTPUT:      <D>:
+# CHECK-OUTPUT-NEXT:                      {{.*}} cbnz x0, 0x[[D_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-OUTPUT-NEXT: [[D_RET:[0-9a-f]+]]: {{.*}} ret
+# CHECK-OUTPUT-NEXT: [[D_BR]]:            {{.*}} b    0x[[D_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_10>
+
+# CHECK-OUTPUT:      <__AArch64BranchForwardThunk_0>:
+# CHECK-OUTPUT-NEXT: [[A_FW1]]:           {{.*}} b  0x[[A_COLD:[0-9a-f]+]] <A.cold.0>
+
+# CHECK-OUTPUT:      <__AArch64BranchForwardThunk_4>:
+# CHECK-OUTPUT-NEXT: [[B_FW1]]:           {{.*}} b  0x[[B_COLD:[0-9a-f]+]] <B.cold.0>
+
+# CHECK-OUTPUT:      <__AArch64BranchForwardThunk_8>:
+# CHECK-OUTPUT-NEXT: [[C_FW0]]:           {{.*}} b  0x[[C_COLD:[0-9a-f]+]] <C.cold.0>
+
+# CHECK-OUTPUT:      <__AArch64BranchForwardThunk_10>:
+# CHECK-OUTPUT-NEXT: [[D_FW0]]:           {{.*}} b  0x[[D_COLD:[0-9a-f]+]] <D.cold.0>
+
+# CHECK-OUTPUT: Disassembly of section .text.cold:
+
+# CHECK-OUTPUT:      <__AArch64BranchBackwardThunk_3>:
+# CHECK-OUTPUT-NEXT: [[A_BW0:[0-9a-f]+]]: {{.*}} b   0x[[A_BW1]] <__AArch64BranchBackwardThunk_2>
+
+# CHECK-OUTPUT:      <__AArch64BranchBackwardThunk_7>:
+# CHECK-OUTPUT-NEXT: [[B_BW0:[0-9a-f]+]]: {{.*}} b   0x[[B_BW1]] <__AArch64BranchBackwardThunk_6>
+
+# CHECK-OUTPUT:      <__AArch64BranchBackwardThunk_9>:
+# CHECK-OUTPUT-NEXT: [[C_BW0:[0-9a-f]+]]: {{.*}} b   0x[[C_RET]] <C+0x4>
+
+# CHECK-OUTPUT:      <__AArch64BranchBackwardThunk_11>:
+# CHECK-OUTPUT-NEXT: [[D_BW0:[0-9a-f]+]]: {{.*}} b   0x[[D_RET]] <D+0x4>
+
+# CHECK-OUTPUT:      <A.cold.0>:
+# CHECK-OUTPUT-NEXT: [[A_COLD]]: {{.*}} mov x0, #0x1
+# CHECK-OUTPUT-NEXT:             {{.*}} b   0x[[A_BW0]] <__AArch64BranchBackwardThunk_3>
+
+# CHECK-OUTPUT:      <B.cold.0>:
+# CHECK-OUTPUT-NEXT: [[B_COLD]]: {{.*}} mov x0, #0x2
+# CHECK-OUTPUT-NEXT:             {{.*}} b   0x[[B_BW0]] <__AArch64BranchBackwardThunk_7>
+
+# CHECK-OUTPUT:      <C.cold.0>:
+# CHECK-OUTPUT-NEXT: [[C_COLD]]: {{.*}} mov x0, #0x3
+# CHECK-OUTPUT-NEXT:             {{.*}} b   0x[[C_BW0]] <__AArch64BranchBackwardThunk_9>
+
+# CHECK-OUTPUT:      <D.cold.0>:
+# CHECK-OUTPUT-NEXT: [[D_COLD]]: {{.*}} mov x0, #0x4
+# CHECK-OUTPUT-NEXT:             {{.*}} b   0x[[D_BW0]] <__AArch64BranchBackwardThunk_11>
+
+
+# CHECK-HFE-OUTPUT: Disassembly of section .text.cold:
+
+# CHECK-HFE-OUTPUT:      <A.cold.0>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x1
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_A_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_1>
+
+# CHECK-HFE-OUTPUT:      <B.cold.0>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x2
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_B_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_3>
+
+# CHECK-HFE-OUTPUT:      <C.cold.0>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x3
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_C_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_7>
+
+# CHECK-HFE-OUTPUT:      <D.cold.0>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x4
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_D_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_11>
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_1>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_FW]]:             {{.*}} b   0x[[HFE_A_RET:[0-9a-f]+]] <A+0x4>
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_3>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_FW]]:             {{.*}} b   0x[[HFE_B_RET:[0-9a-f]+]] <B+0x4>
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_7>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_FW0]]:            {{.*}} b   0x[[HFE_C_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_6>
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_11>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_FW0]]:            {{.*}} b   0x[[HFE_D_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_10>
+
+# CHECK-HFE-OUTPUT: Disassembly of section .text:
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_0>:
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_A_COLD]] <A.cold.0>
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_2>:
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_B_COLD]] <B.cold.0>
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_4>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BW1:[0-9a-f]+]]:  {{.*}} b   0x[[HFE_C_COLD]] <C.cold.0>
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_8>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BW1:[0-9a-f]+]]:  {{.*}} b   0x[[HFE_D_COLD]] <D.cold.0>
+
+# CHECK-HFE-OUTPUT:      <A>:
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} cbnz x0, 0x[[HFE_A_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_RET]]:            {{.*}} ret
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_BR]]:             {{.*}} b    0x{{[0-9a-f]+}} <__AArch64BranchBackwardThunk_0>
+
+# CHECK-HFE-OUTPUT:      <B>:
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} cbnz x0, 0x[[HFE_B_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_RET]]:            {{.*}} ret
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_BR]]:             {{.*}} b    0x{{[0-9a-f]+}} <__AArch64BranchBackwardThunk_2>
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_6>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_FW1]]:            {{.*}} b    0x[[HFE_C_RET:[0-9a-f]+]] <C+0x4>
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_10>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_FW1]]:            {{.*}} b    0x[[HFE_D_RET:[0-9a-f]+]] <D+0x4>
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_5>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BW0:[0-9a-f]+]]:  {{.*}} b    0x[[HFE_C_BW1]] <__AArch64BranchBackwardThunk_4>
+
+# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_9>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BW0:[0-9a-f]+]]:  {{.*}} b    0x[[HFE_D_BW1]] <__AArch64BranchBackwardThunk_8>
+
+# CHECK-HFE-OUTPUT:      <C>:
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} cbnz x0, 0x[[HFE_C_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_RET]]:            {{.*}} ret
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BR]]:             {{.*}} b    0x[[HFE_C_BW0]] <__AArch64BranchBackwardThunk_5>
+
+# CHECK-HFE-OUTPUT:      <D>:
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} cbnz x0, 0x[[HFE_D_BR:[0-9a-f]+]] <{{.*}}>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_RET]]:            {{.*}} ret
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BR]]:             {{.*}} b    0x[[HFE_D_BW0]] <__AArch64BranchBackwardThunk_9>
diff --git a/bolt/test/AArch64/relax-exp.s b/bolt/test/AArch64/relax-calls.s
similarity index 86%
rename from bolt/test/AArch64/relax-exp.s
rename to bolt/test/AArch64/relax-calls.s
index f6394b7a39df7..3f2b0375e304b 100644
--- a/bolt/test/AArch64/relax-exp.s
+++ b/bolt/test/AArch64/relax-calls.s
@@ -55,17 +55,17 @@ hot:
 # CHECK-BOLT-LITE:     BOLT-INFO: 3 long thunks created
 
 ## Check the number of thunks created in other modes.
-# CHECK-BOLT: BOLT-INFO: 4 short thunks created
-# CHECK-BOLT: BOLT-INFO: 3 long thunks created
+# CHECK-BOLT: BOLT-INFO: 5 short thunks created
+# CHECK-BOLT: BOLT-INFO: 5 long thunks created
 
-# CHECK-BOLT-HOT-END: BOLT-INFO: 4 short thunks created
-# CHECK-BOLT-HOT-END: BOLT-INFO: 2 long thunks created
+# CHECK-BOLT-HOT-END: BOLT-INFO: 6 short thunks created
+# CHECK-BOLT-HOT-END: BOLT-INFO: 4 long thunks created
 
 ## Check that correct veneers are used depending on the target proximity.
 # CHECK-OUTPUT-LABEL: <hot>:
 # CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64ADRPThunk_foo>
-# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64Thunk_bar>
-# CHECK-OUTPUT-NEXT: bl {{.*}} <_start>
+# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64ADRPThunk_bar>
+# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64Thunk__start>
 
   .global _start
   .type _start, %function
diff --git a/bolt/test/AArch64/relax-cross-fragment-calls.s b/bolt/test/AArch64/relax-cross-fragment-calls.s
new file mode 100644
index 0000000000000..f99d30ee94f8d
--- /dev/null
+++ b/bolt/test/AArch64/relax-cross-fragment-calls.s
@@ -0,0 +1,216 @@
+## Check call relaxation with function fragment clusters. This uses split
+## functions with calls from hot and cold fragments so call thunks are placed
+## around the clusters.
+##
+## Input layout:
+##
+##   A  8MiB island, pad_hot_0 50MiB, B 8MiB island, pad_hot_1 50MiB,
+##   C  8MiB island, pad_hot_2 50MiB, D 8MiB island, pad_hot_3 50MiB
+##
+## With --split-functions, BOLT places the cold blocks of A/B/C/D in
+## .text.cold. With the default 120MiB function-fragment cluster size, call
+## relaxation sees:
+##
+##   normal layout:
+##     cluster 0, .text:      A, pad_hot_0, B, pad_hot_1
+##     cluster 1, .text:      C, pad_hot_2, D, pad_hot_3
+##     cluster 2, .text.cold: A.cold, B.cold, C.cold, D.cold
+##
+##   --hot-functions-at-end:
+##     cluster 0, .text.cold: A.cold, B.cold, C.cold, D.cold
+##     cluster 1, .text:      A, pad_hot_0, B, pad_hot_1
+##     cluster 2, .text:      C, pad_hot_2, D, pad_hot_3
+
+# REQUIRES: system-linux
+
+# RUN: %clang %cflags -Wl,-q -Wl,-e,A %s -o %t -nostdlib
+# RUN: link_fdata --no-lbr %s %t %t.fdata
+# RUN: llvm-strip --strip-unneeded %t
+# RUN: llvm-bolt %t -o %t.bolt --data %t.fdata --split-functions \
+# RUN:   --compact-code-model --relax-exp \
+# RUN:   | FileCheck %s --check-prefix=CHECK-BOLT
+# RUN: llvm-bolt %t -o %t.hfe.bolt --data %t.fdata --split-functions \
+# RUN:   --compact-code-model --relax-exp --hot-functions-at-end \
+# RUN:   | FileCheck %s --check-prefix=CHECK-BOLT-HFE
+# RUN: llvm-readelf -S %t.bolt | FileCheck %s --check-prefix=CHECK-SECTIONS
+# RUN: llvm-objdump -d \
+# RUN:   --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64Thunk_A,__AArch64Thunk_C,__AArch64ADRPThunk_A \
+# RUN:   %t.bolt | FileCheck %s --check-prefix=CHECK-OUTPUT
+# RUN: llvm-objdump -d \
+# RUN:   --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64Thunk_A,__AArch64Thunk_C,__AArch64ADRPThunk_C \
+# RUN:   %t.hfe.bolt | FileCheck %s --check-prefix=CHECK-HFE-OUTPUT
+
+# CHECK-BOLT: BOLT-INFO: built 3 function fragment cluster(s)
+# CHECK-BOLT: BOLT-INFO: relaxed 4 calls with thunks
+# CHECK-BOLT: BOLT-INFO: 3 short thunks created
+# CHECK-BOLT: BOLT-INFO: 1 long thunks created
+# CHECK-BOLT: BOLT-INFO: relaxed 8 cross-cluster branches
+# CHECK-BOLT: BOLT-INFO: 12 branch thunks created
+
+# CHECK-BOLT-HFE: BOLT-INFO: built 3 function fragment cluster(s)
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 4 calls with thunks
+# CHECK-BOLT-HFE: BOLT-INFO: 3 short thunks created
+# CHECK-BOLT-HFE: BOLT-INFO: 1 long thunks created
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 8 cross-cluster branches
+# CHECK-BOLT-HFE: BOLT-INFO: 12 branch thunks created
+
+# CHECK-SECTIONS: .text
+# CHECK-SECTIONS: .text.cold
+
+  .text
+  .globl A
+  .type A, %function
+A:
+.A_entry:
+# FDATA: 1 A #.A_entry# 100
+  bl B
+  bl C
+  cbz x0, .A_ret
+  b .A_cold
+.A_ret:
+# FDATA: 1 A #.A_ret# 100
+  ret
+.A_cold:
+  mov x0, #1
+  bl C
+  bl A
+  b .A_ret
+  .space 0x800000
+  .size A, .-A
+
+  .globl pad_hot_0
+  .type pad_hot_0, %function
+pad_hot_0:
+.pad_hot_0_entry:
+# FDATA: 1 pad_hot_0 #.pad_hot_0_entry# 100
+  ret
+  .space 0x3200000
+  .size pad_hot_0, .-pad_hot_0
+
+  .globl B
+  .type B, %function
+B:
+.B_entry:
+# FDATA: 1 B #.B_entry# 100
+  cbz x0, .B_ret
+  b .B_cold
+.B_ret:
+# FDATA: 1 B #.B_ret# 100
+  ret
+.B_cold:
+  mov x0, #2
+  b .B_ret
+  .space 0x800000
+  .size B, .-B
+
+  .globl pad_hot_1
+  .type pad_hot_1, %function
+pad_hot_1:
+.pad_hot_1_entry:
+# FDATA: 1 pad_hot_1 #.pad_hot_1_entry# 100
+  ret
+  .space 0x3200000
+  .size pad_hot_1, .-pad_hot_1
+
+  .globl C
+  .type C, %function
+C:
+.C_entry:
+# FDATA: 1 C #.C_entry# 100
+  bl A
+  cbz x0, .C_ret
+  b .C_cold
+.C_ret:
+# FDATA: 1 C #.C_ret# 100
+  ret
+.C_cold:
+  mov x0, #3
+  b .C_ret
+  .space 0x800000
+  .size C, .-C
+
+  .globl pad_hot_2
+  .type pad_hot_2, %function
+pad_hot_2:
+.pad_hot_2_entry:
+# FDATA: 1 pad_hot_2 #.pad_hot_2_entry# 100
+  ret
+  .space 0x3200000
+  .size pad_hot_2, .-pad_hot_2
+
+  .globl D
+  .type D, %function
+D:
+.D_entry:
+# FDATA: 1 D #.D_entry# 100
+  cbz x0, .D_ret
+  b .D_cold
+.D_ret:
+# FDATA: 1 D #.D_ret# 100
+  ret
+.D_cold:
+  mov x0, #4
+  b .D_ret
+  .space 0x800000
+  .size D, .-D
+
+  .globl pad_hot_3
+  .type pad_hot_3, %function
+pad_hot_3:
+.pad_hot_3_entry:
+# FDATA: 1 pad_hot_3 #.pad_hot_3_entry# 100
+  ret
+  .space 0x3200000
+  .size pad_hot_3, .-pad_hot_3
+
+## Force relocation mode.
+  .reloc 0, R_AARCH64_NONE
+
+# CHECK-OUTPUT:      <A>:
+# CHECK-OUTPUT-NEXT: {{.*}} bl {{.*}} <B>
+# CHECK-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_C>
+
+# CHECK-OUTPUT:      <__AArch64Thunk_C>:
+# CHECK-OUTPUT-NEXT: {{.*}} b {{.*}} <C>
+
+# CHECK-OUTPUT:      <__AArch64Thunk_A>:
+# CHECK-OUTPUT-NEXT: {{.*}} b {{.*}} <A>
+
+# CHECK-OUTPUT:      <C>:
+# CHECK-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_A>
+
+# CHECK-OUTPUT:      <__AArch64ADRPThunk_A>:
+# CHECK-OUTPUT-NEXT: {{.*}} adrp x16, {{.*}} <A>
+# CHECK-OUTPUT-NEXT: {{.*}} add x16, x16, #0x0
+# CHECK-OUTPUT-NEXT: {{.*}} br x16
+
+# CHECK-OUTPUT:      <A.cold.0>:
+# CHECK-OUTPUT-NEXT: {{.*}} mov x0, #0x1
+# CHECK-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_C>
+# CHECK-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64ADRPThunk_A>
+
+# CHECK-HFE-OUTPUT:      <A.cold.0>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} mov x0, #0x1
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64ADRPThunk_C>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_A>
+
+# CHECK-HFE-OUTPUT:      <__AArch64ADRPThunk_C>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} adrp x16, {{.*}}
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} add x16, x16, #0x{{[0-9a-f]+}}
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} br x16
+
+# CHECK-HFE-OUTPUT:      <__AArch64Thunk_A>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b {{.*}} <A>
+
+# CHECK-HFE-OUTPUT:      <A>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <B>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_C>
+
+# CHECK-HFE-OUTPUT:      <__AArch64Thunk_C>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b {{.*}} <C>
+
+# CHECK-HFE-OUTPUT:      <__AArch64Thunk_A>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} b {{.*}} <A>
+
+# CHECK-HFE-OUTPUT:      <C>:
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_A>

>From 6514e5993db0a696630c50fdc53432f088e3d356 Mon Sep 17 00:00:00 2001
From: Alexandros Lamprineas <alexandros.lamprineas at arm.com>
Date: Thu, 27 Aug 2026 09:00:27 +0000
Subject: [PATCH 2/2] allow clusters to span across sections

---
 bolt/include/bolt/Passes/LongJmp.h            | 18 +++--
 bolt/lib/Passes/LongJmp.cpp                   | 15 ++--
 .../AArch64/relax-branches-with-thunk-chain.s | 73 +++++++------------
 bolt/test/AArch64/relax-calls.s               | 14 ++--
 .../test/AArch64/relax-cross-fragment-calls.s | 34 ++++-----
 5 files changed, 65 insertions(+), 89 deletions(-)

diff --git a/bolt/include/bolt/Passes/LongJmp.h b/bolt/include/bolt/Passes/LongJmp.h
index 37c7e1452d5bf..e021e8fad503b 100644
--- a/bolt/include/bolt/Passes/LongJmp.h
+++ b/bolt/include/bolt/Passes/LongJmp.h
@@ -84,10 +84,13 @@ class LongJmpPass : public BinaryFunctionPass {
 
   /// A group of function fragments that are located within the longest direct
   /// branch/call instruction distance. Jumps within the cluster do not require
-  /// a thunk. The cluster may include thunks for jumps to targets outside.
+  /// a thunk. The cluster may span output sections and include thunks for jumps
+  /// to targets outside. Backward thunks are inserted before the cluster, while
+  /// forward thunks are inserted after it.
   struct FragmentCluster {
-    /// Output code section containing fragments in this cluster.
-    SmallString<32> SectionName;
+    /// Output code sections containing the first and last cluster fragments.
+    SmallString<32> StartSectionName;
+    SmallString<32> EndSectionName;
 
     /// Estimated size of the cluster in bytes.
     uint64_t Size{0};
@@ -115,12 +118,11 @@ class LongJmpPass : public BinaryFunctionPass {
     /// <Destination Symbol> -> <Thunk Function>.
     DenseMap<const MCSymbol *, BinaryFunction *> ForwardBranchThunks;
     DenseMap<const MCSymbol *, BinaryFunction *> BackwardBranchThunks;
-  };
 
-  /// Maximum size of combined regular functions in the cluster. Note that it's
-  /// less than 128MB, because the size of the cluster plus its thunks should be
-  /// less than 128MB.
-  static constexpr uint64_t MaxClusterSize = 120 * 1024 * 1024;
+    StringRef getThunkSectionName(bool IsForward) const {
+      return IsForward ? EndSectionName : StartSectionName;
+    }
+  };
 
   struct FragmentClusterLayout {
     SmallVector<FragmentCluster, 4> Clusters;
diff --git a/bolt/lib/Passes/LongJmp.cpp b/bolt/lib/Passes/LongJmp.cpp
index ccdb2ef37f77f..3a0dda76e0b08 100644
--- a/bolt/lib/Passes/LongJmp.cpp
+++ b/bolt/lib/Passes/LongJmp.cpp
@@ -40,6 +40,11 @@ static cl::opt<bool>
 static cl::opt<bool> RelaxPLT("relax-plt",
                               cl::desc("indicate PLT proximity to hot text"),
                               cl::init(true), cl::cat(BoltOptCategory));
+
+static cl::opt<unsigned long long> MaxClusterSize(
+    "max-cluster-size",
+    cl::desc("maximum estimated size of a function fragment cluster in bytes"),
+    cl::init(124 * 1024 * 1024), cl::cat(BoltOptCategory));
 }
 
 namespace llvm {
@@ -1079,15 +1084,15 @@ LongJmpPass::buildClusterLayout(BinaryContext &BC,
     const uint64_t FFSize = estimateFragmentSize(BF, FF);
 
     if (Layout.Clusters.empty() ||
-        Layout.Clusters.back().SectionName != Fragment.SectionName ||
-        Layout.Clusters.back().Size + FFSize > MaxClusterSize) {
+        Layout.Clusters.back().Size + FFSize > opts::MaxClusterSize) {
       Layout.Clusters.emplace_back(FragmentCluster());
       FragmentCluster &FC = Layout.Clusters.back();
-      FC.SectionName = Fragment.SectionName;
+      FC.StartSectionName = Fragment.SectionName;
       FC.FirstFunctionIndex = Fragment.FunctionIndex;
     }
 
     FragmentCluster &FC = Layout.Clusters.back();
+    FC.EndSectionName = Fragment.SectionName;
     FC.LastFunctionIndex = Fragment.FunctionIndex;
     ++FC.NumFragments;
     EstimatedSize += FFSize;
@@ -1249,7 +1254,7 @@ void LongJmpPass::relaxCalls(BinaryContext &BC,
     BinaryFunction *Thunk = createCallThunk(TargetSymbol, IsShort);
 
     FragmentCluster &ThunkCluster = Clusters[SourceCluster];
-    Thunk->setCodeSectionName(ThunkCluster.SectionName);
+    Thunk->setCodeSectionName(ThunkCluster.getThunkSectionName(IsForward));
     BinaryFunctionListType &ThunkList = IsForward
                                             ? ThunkCluster.ForwardThunkList
                                             : ThunkCluster.BackwardThunkList;
@@ -1355,7 +1360,7 @@ void LongJmpPass::relaxUnconditionalBranches(
           return It->second;
 
         BinaryFunction *Thunk = createBranchThunk(NextTarget, IsForward);
-        Thunk->setCodeSectionName(Cluster.SectionName);
+        Thunk->setCodeSectionName(Cluster.getThunkSectionName(IsForward));
         auto &ThunkList =
             IsForward ? Cluster.ForwardThunkList : Cluster.BackwardThunkList;
         ThunkList.push_back(Thunk);
diff --git a/bolt/test/AArch64/relax-branches-with-thunk-chain.s b/bolt/test/AArch64/relax-branches-with-thunk-chain.s
index 6c46afdb1c40a..1eddf8160fca9 100644
--- a/bolt/test/AArch64/relax-branches-with-thunk-chain.s
+++ b/bolt/test/AArch64/relax-branches-with-thunk-chain.s
@@ -8,8 +8,8 @@
 ##   C  8MiB island, pad_hot_2 50MiB, D 8MiB island, pad_hot_3 50MiB
 ##
 ## With --split-functions, BOLT places the cold blocks of A/B/C/D in
-## .text.cold. With the default 120MiB function-fragment cluster size, branch
-## relaxation sees:
+## .text.cold. With the default 124MiB function-fragment cluster size,
+## branch relaxation sees:
 ##
 ##   normal layout:
 ##     cluster 0, .text:      A, pad_hot_0, B, pad_hot_1
@@ -17,9 +17,10 @@
 ##     cluster 2, .text.cold: A.cold, B.cold, C.cold, D.cold
 ##
 ##   --hot-functions-at-end:
-##     cluster 0, .text.cold: A.cold, B.cold, C.cold, D.cold
-##     cluster 1, .text:      A, pad_hot_0, B, pad_hot_1
-##     cluster 2, .text:      C, pad_hot_2, D, pad_hot_3
+##     cluster 0: .text.cold A.cold, B.cold, C.cold, D.cold
+##                .text A, pad_hot_0, B
+##     cluster 1: .text pad_hot_1, C, pad_hot_2, D
+##     cluster 2: .text pad_hot_3
 
 # REQUIRES: system-linux
 
@@ -37,7 +38,7 @@
 # RUN:   --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64BranchForwardThunk_0,__AArch64BranchForwardThunk_1,__AArch64BranchForwardThunk_4,__AArch64BranchForwardThunk_5,__AArch64BranchForwardThunk_8,__AArch64BranchForwardThunk_10,__AArch64BranchBackwardThunk_2,__AArch64BranchBackwardThunk_3,__AArch64BranchBackwardThunk_6,__AArch64BranchBackwardThunk_7,__AArch64BranchBackwardThunk_9,__AArch64BranchBackwardThunk_11 \
 # RUN:   %t.bolt | FileCheck %s --check-prefix=CHECK-OUTPUT
 # RUN: llvm-objdump -d \
-# RUN:   --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64BranchForwardThunk_1,__AArch64BranchForwardThunk_3,__AArch64BranchForwardThunk_6,__AArch64BranchForwardThunk_7,__AArch64BranchForwardThunk_10,__AArch64BranchForwardThunk_11,__AArch64BranchBackwardThunk_0,__AArch64BranchBackwardThunk_2,__AArch64BranchBackwardThunk_4,__AArch64BranchBackwardThunk_5,__AArch64BranchBackwardThunk_8,__AArch64BranchBackwardThunk_9 \
+# RUN:   --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64BranchForwardThunk_1,__AArch64BranchForwardThunk_3,__AArch64BranchBackwardThunk_0,__AArch64BranchBackwardThunk_2 \
 # RUN:   %t.hfe.bolt | FileCheck %s --check-prefix=CHECK-HFE-OUTPUT
 
 # CHECK-BOLT: BOLT-INFO: built 3 function fragment cluster(s)
@@ -45,8 +46,8 @@
 # CHECK-BOLT: BOLT-INFO: 12 branch thunks created
 
 # CHECK-BOLT-HFE: BOLT-INFO: built 3 function fragment cluster(s)
-# CHECK-BOLT-HFE: BOLT-INFO: relaxed 8 cross-cluster branches
-# CHECK-BOLT-HFE: BOLT-INFO: 12 branch thunks created
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 4 cross-cluster branches
+# CHECK-BOLT-HFE: BOLT-INFO: 4 branch thunks created
 
 # CHECK-SECTIONS: .text
 # CHECK-SECTIONS: .text.cold
@@ -236,74 +237,50 @@ pad_hot_3:
 
 # CHECK-HFE-OUTPUT:      <A.cold.0>:
 # CHECK-HFE-OUTPUT-NEXT: [[HFE_A_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x1
-# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_A_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_1>
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_A_RET:[0-9a-f]+]] <A+0x4>
 
 # CHECK-HFE-OUTPUT:      <B.cold.0>:
 # CHECK-HFE-OUTPUT-NEXT: [[HFE_B_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x2
-# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_B_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_3>
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_B_RET:[0-9a-f]+]] <B+0x4>
 
 # CHECK-HFE-OUTPUT:      <C.cold.0>:
 # CHECK-HFE-OUTPUT-NEXT: [[HFE_C_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x3
-# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_C_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_7>
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_C_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_1>
 
 # CHECK-HFE-OUTPUT:      <D.cold.0>:
 # CHECK-HFE-OUTPUT-NEXT: [[HFE_D_COLD:[0-9a-f]+]]: {{.*}} mov x0, #0x4
-# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_D_FW0:[0-9a-f]+]] <__AArch64BranchForwardThunk_11>
-
-# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_1>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_FW]]:             {{.*}} b   0x[[HFE_A_RET:[0-9a-f]+]] <A+0x4>
-
-# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_3>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_FW]]:             {{.*}} b   0x[[HFE_B_RET:[0-9a-f]+]] <B+0x4>
-
-# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_7>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_FW0]]:            {{.*}} b   0x[[HFE_C_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_6>
-
-# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_11>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_FW0]]:            {{.*}} b   0x[[HFE_D_FW1:[0-9a-f]+]] <__AArch64BranchForwardThunk_10>
+# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_D_FW:[0-9a-f]+]] <__AArch64BranchForwardThunk_3>
 
 # CHECK-HFE-OUTPUT: Disassembly of section .text:
 
-# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_0>:
-# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_A_COLD]] <A.cold.0>
-
-# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_2>:
-# CHECK-HFE-OUTPUT-NEXT:                           {{.*}} b   0x[[HFE_B_COLD]] <B.cold.0>
-
-# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_4>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BW1:[0-9a-f]+]]:  {{.*}} b   0x[[HFE_C_COLD]] <C.cold.0>
-
-# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_8>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BW1:[0-9a-f]+]]:  {{.*}} b   0x[[HFE_D_COLD]] <D.cold.0>
-
 # CHECK-HFE-OUTPUT:      <A>:
 # CHECK-HFE-OUTPUT-NEXT:                           {{.*}} cbnz x0, 0x[[HFE_A_BR:[0-9a-f]+]] <{{.*}}>
 # CHECK-HFE-OUTPUT-NEXT: [[HFE_A_RET]]:            {{.*}} ret
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_BR]]:             {{.*}} b    0x{{[0-9a-f]+}} <__AArch64BranchBackwardThunk_0>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_A_BR]]:             {{.*}} b    0x[[HFE_A_COLD]] <A.cold.0>
 
 # CHECK-HFE-OUTPUT:      <B>:
 # CHECK-HFE-OUTPUT-NEXT:                           {{.*}} cbnz x0, 0x[[HFE_B_BR:[0-9a-f]+]] <{{.*}}>
 # CHECK-HFE-OUTPUT-NEXT: [[HFE_B_RET]]:            {{.*}} ret
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_BR]]:             {{.*}} b    0x{{[0-9a-f]+}} <__AArch64BranchBackwardThunk_2>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_B_BR]]:             {{.*}} b    0x[[HFE_B_COLD]] <B.cold.0>
 
-# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_6>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_FW1]]:            {{.*}} b    0x[[HFE_C_RET:[0-9a-f]+]] <C+0x4>
+# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_1>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_FW]]:             {{.*}} b    0x[[HFE_C_RET:[0-9a-f]+]] <C+0x4>
 
-# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_10>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_FW1]]:            {{.*}} b    0x[[HFE_D_RET:[0-9a-f]+]] <D+0x4>
+# CHECK-HFE-OUTPUT:      <__AArch64BranchForwardThunk_3>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_FW]]:             {{.*}} b    0x[[HFE_D_RET:[0-9a-f]+]] <D+0x4>
 
-# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_5>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BW0:[0-9a-f]+]]:  {{.*}} b    0x[[HFE_C_BW1]] <__AArch64BranchBackwardThunk_4>
+# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_0>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BW:[0-9a-f]+]]:   {{.*}} b    0x[[HFE_C_COLD]] <C.cold.0>
 
-# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_9>:
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BW0:[0-9a-f]+]]:  {{.*}} b    0x[[HFE_D_BW1]] <__AArch64BranchBackwardThunk_8>
+# CHECK-HFE-OUTPUT:      <__AArch64BranchBackwardThunk_2>:
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BW:[0-9a-f]+]]:   {{.*}} b    0x[[HFE_D_COLD]] <D.cold.0>
 
 # CHECK-HFE-OUTPUT:      <C>:
 # CHECK-HFE-OUTPUT-NEXT:                           {{.*}} cbnz x0, 0x[[HFE_C_BR:[0-9a-f]+]] <{{.*}}>
 # CHECK-HFE-OUTPUT-NEXT: [[HFE_C_RET]]:            {{.*}} ret
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BR]]:             {{.*}} b    0x[[HFE_C_BW0]] <__AArch64BranchBackwardThunk_5>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_C_BR]]:             {{.*}} b    0x[[HFE_C_BW]] <__AArch64BranchBackwardThunk_0>
 
 # CHECK-HFE-OUTPUT:      <D>:
 # CHECK-HFE-OUTPUT-NEXT:                           {{.*}} cbnz x0, 0x[[HFE_D_BR:[0-9a-f]+]] <{{.*}}>
 # CHECK-HFE-OUTPUT-NEXT: [[HFE_D_RET]]:            {{.*}} ret
-# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BR]]:             {{.*}} b    0x[[HFE_D_BW0]] <__AArch64BranchBackwardThunk_9>
+# CHECK-HFE-OUTPUT-NEXT: [[HFE_D_BR]]:             {{.*}} b    0x[[HFE_D_BW]] <__AArch64BranchBackwardThunk_2>
diff --git a/bolt/test/AArch64/relax-calls.s b/bolt/test/AArch64/relax-calls.s
index 3f2b0375e304b..767195f648c4a 100644
--- a/bolt/test/AArch64/relax-calls.s
+++ b/bolt/test/AArch64/relax-calls.s
@@ -55,17 +55,17 @@ hot:
 # CHECK-BOLT-LITE:     BOLT-INFO: 3 long thunks created
 
 ## Check the number of thunks created in other modes.
-# CHECK-BOLT: BOLT-INFO: 5 short thunks created
-# CHECK-BOLT: BOLT-INFO: 5 long thunks created
+# CHECK-BOLT: BOLT-INFO: 4 short thunks created
+# CHECK-BOLT: BOLT-INFO: 3 long thunks created
 
-# CHECK-BOLT-HOT-END: BOLT-INFO: 6 short thunks created
-# CHECK-BOLT-HOT-END: BOLT-INFO: 4 long thunks created
+# CHECK-BOLT-HOT-END: BOLT-INFO: 4 short thunks created
+# CHECK-BOLT-HOT-END: BOLT-INFO: 2 long thunks created
 
 ## Check that correct veneers are used depending on the target proximity.
 # CHECK-OUTPUT-LABEL: <hot>:
-# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64ADRPThunk_foo>
-# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64ADRPThunk_bar>
-# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64Thunk__start>
+# CHECK-OUTPUT-NEXT: bl {{.*}} <foo>
+# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64Thunk_bar>
+# CHECK-OUTPUT-NEXT: bl {{.*}} <__AArch64ADRPThunk__start>
 
   .global _start
   .type _start, %function
diff --git a/bolt/test/AArch64/relax-cross-fragment-calls.s b/bolt/test/AArch64/relax-cross-fragment-calls.s
index f99d30ee94f8d..b1522e2db38bc 100644
--- a/bolt/test/AArch64/relax-cross-fragment-calls.s
+++ b/bolt/test/AArch64/relax-cross-fragment-calls.s
@@ -8,8 +8,8 @@
 ##   C  8MiB island, pad_hot_2 50MiB, D 8MiB island, pad_hot_3 50MiB
 ##
 ## With --split-functions, BOLT places the cold blocks of A/B/C/D in
-## .text.cold. With the default 120MiB function-fragment cluster size, call
-## relaxation sees:
+## .text.cold. With the default 124MiB function-fragment cluster size,
+## call relaxation sees:
 ##
 ##   normal layout:
 ##     cluster 0, .text:      A, pad_hot_0, B, pad_hot_1
@@ -17,9 +17,10 @@
 ##     cluster 2, .text.cold: A.cold, B.cold, C.cold, D.cold
 ##
 ##   --hot-functions-at-end:
-##     cluster 0, .text.cold: A.cold, B.cold, C.cold, D.cold
-##     cluster 1, .text:      A, pad_hot_0, B, pad_hot_1
-##     cluster 2, .text:      C, pad_hot_2, D, pad_hot_3
+##     cluster 0: .text.cold A.cold, B.cold, C.cold, D.cold
+##                .text A, pad_hot_0, B
+##     cluster 1: .text pad_hot_1, C, pad_hot_2, D
+##     cluster 2: .text pad_hot_3
 
 # REQUIRES: system-linux
 
@@ -37,7 +38,7 @@
 # RUN:   --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64Thunk_A,__AArch64Thunk_C,__AArch64ADRPThunk_A \
 # RUN:   %t.bolt | FileCheck %s --check-prefix=CHECK-OUTPUT
 # RUN: llvm-objdump -d \
-# RUN:   --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64Thunk_A,__AArch64Thunk_C,__AArch64ADRPThunk_C \
+# RUN:   --disassemble-symbols=A,B,C,D,A.cold.0,B.cold.0,C.cold.0,D.cold.0,__AArch64Thunk_A,__AArch64Thunk_C \
 # RUN:   %t.hfe.bolt | FileCheck %s --check-prefix=CHECK-HFE-OUTPUT
 
 # CHECK-BOLT: BOLT-INFO: built 3 function fragment cluster(s)
@@ -48,11 +49,10 @@
 # CHECK-BOLT: BOLT-INFO: 12 branch thunks created
 
 # CHECK-BOLT-HFE: BOLT-INFO: built 3 function fragment cluster(s)
-# CHECK-BOLT-HFE: BOLT-INFO: relaxed 4 calls with thunks
-# CHECK-BOLT-HFE: BOLT-INFO: 3 short thunks created
-# CHECK-BOLT-HFE: BOLT-INFO: 1 long thunks created
-# CHECK-BOLT-HFE: BOLT-INFO: relaxed 8 cross-cluster branches
-# CHECK-BOLT-HFE: BOLT-INFO: 12 branch thunks created
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 3 calls with thunks
+# CHECK-BOLT-HFE: BOLT-INFO: 2 short thunks created
+# CHECK-BOLT-HFE: BOLT-INFO: relaxed 4 cross-cluster branches
+# CHECK-BOLT-HFE: BOLT-INFO: 4 branch thunks created
 
 # CHECK-SECTIONS: .text
 # CHECK-SECTIONS: .text.cold
@@ -191,16 +191,8 @@ pad_hot_3:
 
 # CHECK-HFE-OUTPUT:      <A.cold.0>:
 # CHECK-HFE-OUTPUT-NEXT: {{.*}} mov x0, #0x1
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64ADRPThunk_C>
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_A>
-
-# CHECK-HFE-OUTPUT:      <__AArch64ADRPThunk_C>:
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} adrp x16, {{.*}}
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} add x16, x16, #0x{{[0-9a-f]+}}
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} br x16
-
-# CHECK-HFE-OUTPUT:      <__AArch64Thunk_A>:
-# CHECK-HFE-OUTPUT-NEXT: {{.*}} b {{.*}} <A>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <__AArch64Thunk_C>
+# CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <A>
 
 # CHECK-HFE-OUTPUT:      <A>:
 # CHECK-HFE-OUTPUT-NEXT: {{.*}} bl {{.*}} <B>



More information about the llvm-commits mailing list