[llvm] [Mips][BranchFolding] Enable safe tail merging (PR #225382)

Jiaxun Yang via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 22 05:27:25 PDT 2026


https://github.com/FlyGoat updated https://github.com/llvm/llvm-project/pull/225382

>From cea0c6c0b85514d4e0613cb07d183ffc22072d05 Mon Sep 17 00:00:00 2001
From: Jiaxun Yang <jiaxun.yang at flygoat.com>
Date: Tue, 22 Sep 2026 12:44:24 +0100
Subject: [PATCH] [Mips][BranchFolding] Enable safe tail merging

MIPS disables tail merging when the long-branch pass is enabled because
branch expansion may clobber $at. This also blocks merges where $at is
dead at the new branch.

Use physical-register liveness to reject tails that need $at, allowing
safe merges for standard MIPS and microMIPS. Keep MIPS16 tail merging
disabled because its far branches clobber $ra.

Check both candidate tails before selecting a common tail. The existing
split check is only reached when creating a new block; reusing an entire
block can still introduce a branch without passing that check.

Assisted-by: chatgpt-5.6-luna # Testcases
---
 llvm/lib/CodeGen/BranchFolding.cpp            | 12 +--
 llvm/lib/Target/Mips/MipsInstrInfo.cpp        | 19 ++++
 llvm/lib/Target/Mips/MipsInstrInfo.h          |  3 +
 llvm/lib/Target/Mips/MipsTargetMachine.cpp    |  5 --
 .../Mips/compact-branch-combine-never.ll      |  5 +-
 .../CodeGen/Mips/compact-branch-combine.ll    |  5 +-
 llvm/test/CodeGen/Mips/disable-tail-merge.ll  | 33 -------
 .../CodeGen/Mips/llvm-ir/forbidden-slot-ir.ll |  3 +-
 .../CodeGen/Mips/mips1-load-in-delay-slot.ll  |  3 +-
 llvm/test/CodeGen/Mips/pseudo-jump-fill.ll    | 11 +--
 llvm/test/CodeGen/Mips/tail-merge.ll          | 88 +++++++++++++++++++
 11 files changed, 131 insertions(+), 56 deletions(-)
 delete mode 100644 llvm/test/CodeGen/Mips/disable-tail-merge.ll
 create mode 100644 llvm/test/CodeGen/Mips/tail-merge.ll

diff --git a/llvm/lib/CodeGen/BranchFolding.cpp b/llvm/lib/CodeGen/BranchFolding.cpp
index aef63f6bdc1c52..3dafdcb6d65aa1 100644
--- a/llvm/lib/CodeGen/BranchFolding.cpp
+++ b/llvm/lib/CodeGen/BranchFolding.cpp
@@ -754,11 +754,13 @@ unsigned BranchFolder::ComputeSameTails(unsigned CurHash,
     for (MPIterator I = std::prev(CurMPIter); I->getHash() == CurHash; --I) {
       unsigned CommonTailLen;
       if (ProfitableToMerge(CurMPIter->getBlock(), I->getBlock(),
-                            MinCommonTailLength,
-                            CommonTailLen, TrialBBI1, TrialBBI2,
-                            SuccBB, PredBB,
-                            EHScopeMembership,
-                            AfterBlockPlacement, MBBFreqInfo, PSI)) {
+                            MinCommonTailLength, CommonTailLen, TrialBBI1,
+                            TrialBBI2, SuccBB, PredBB, EHScopeMembership,
+                            AfterBlockPlacement, MBBFreqInfo, PSI) &&
+          // Check both tails even if one is an entire block: replacing the
+          // other tail can introduce a branch at its start.
+          TII->isLegalToSplitMBBAt(*CurMPIter->getBlock(), TrialBBI1) &&
+          TII->isLegalToSplitMBBAt(*I->getBlock(), TrialBBI2)) {
         if (CommonTailLen > maxCommonTailLength) {
           SameTails.clear();
           maxCommonTailLength = CommonTailLen;
diff --git a/llvm/lib/Target/Mips/MipsInstrInfo.cpp b/llvm/lib/Target/Mips/MipsInstrInfo.cpp
index 3dcd9c712b90d0..c11bf9af98c54d 100644
--- a/llvm/lib/Target/Mips/MipsInstrInfo.cpp
+++ b/llvm/lib/Target/Mips/MipsInstrInfo.cpp
@@ -16,12 +16,14 @@
 #include "Mips.h"
 #include "MipsSubtarget.h"
 #include "llvm/ADT/SmallVector.h"
+#include "llvm/CodeGen/LivePhysRegs.h"
 #include "llvm/CodeGen/MachineBasicBlock.h"
 #include "llvm/CodeGen/MachineFrameInfo.h"
 #include "llvm/CodeGen/MachineFunction.h"
 #include "llvm/CodeGen/MachineInstr.h"
 #include "llvm/CodeGen/MachineInstrBuilder.h"
 #include "llvm/CodeGen/MachineOperand.h"
+#include "llvm/CodeGen/MachineRegisterInfo.h"
 #include "llvm/CodeGen/TargetOpcodes.h"
 #include "llvm/CodeGen/TargetSubtargetInfo.h"
 #include "llvm/IR/DebugInfoMetadata.h"
@@ -57,6 +59,23 @@ const MipsInstrInfo *MipsInstrInfo::create(MipsSubtarget &STI) {
   return createMipsSEInstrInfo(STI);
 }
 
+bool MipsInstrInfo::isLegalToSplitMBBAt(
+    MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI) const {
+  // Keep tail merging disabled for MIPS16, whose far branches clobber $ra.
+  if (Subtarget.inMips16Mode())
+    return false;
+
+  // Long branch expansion clobbers $at. Do not let tail merging introduce a
+  // branch while $at (or its 64-bit super-register) is live.
+  if (!MBB.getParent()->getRegInfo().tracksLiveness())
+    return false;
+  LivePhysRegs LiveRegs(*Subtarget.getRegisterInfo());
+  LiveRegs.addLiveOuts(MBB);
+  for (auto I = MBB.end(); I != MBBI;)
+    LiveRegs.stepBackward(*--I);
+  return !LiveRegs.contains(Mips::AT);
+}
+
 bool MipsInstrInfo::isZeroImm(const MachineOperand &op) const {
   return op.isImm() && op.getImm() == 0;
 }
diff --git a/llvm/lib/Target/Mips/MipsInstrInfo.h b/llvm/lib/Target/Mips/MipsInstrInfo.h
index 8772fc8ee28ef1..15b292d14bd434 100644
--- a/llvm/lib/Target/Mips/MipsInstrInfo.h
+++ b/llvm/lib/Target/Mips/MipsInstrInfo.h
@@ -77,6 +77,9 @@ class MipsInstrInfo : public MipsGenInstrInfo {
                         const DebugLoc &DL,
                         int *BytesAdded = nullptr) const override;
 
+  bool isLegalToSplitMBBAt(MachineBasicBlock &MBB,
+                           MachineBasicBlock::iterator MBBI) const override;
+
   bool
   reverseBranchCondition(SmallVectorImpl<MachineOperand> &Cond) const override;
 
diff --git a/llvm/lib/Target/Mips/MipsTargetMachine.cpp b/llvm/lib/Target/Mips/MipsTargetMachine.cpp
index 3b4a6aa0b7a543..f2e0a0aa145c56 100644
--- a/llvm/lib/Target/Mips/MipsTargetMachine.cpp
+++ b/llvm/lib/Target/Mips/MipsTargetMachine.cpp
@@ -191,11 +191,6 @@ class MipsPassConfig : public TargetPassConfig {
 public:
   MipsPassConfig(MipsTargetMachine &TM, PassManagerBase &PM)
       : TargetPassConfig(TM, PM) {
-    // The current implementation of long branch pass requires a scratch
-    // register ($at) to be available before branch instructions. Tail merging
-    // can break this requirement, so disable it when long branch pass is
-    // enabled.
-    EnableTailMerge = !getMipsSubtarget().enableLongBranchPass();
     EnableLoopTermFold = true;
   }
 
diff --git a/llvm/test/CodeGen/Mips/compact-branch-combine-never.ll b/llvm/test/CodeGen/Mips/compact-branch-combine-never.ll
index 74e908d08248c2..fb05d3fc690142 100644
--- a/llvm/test/CodeGen/Mips/compact-branch-combine-never.ll
+++ b/llvm/test/CodeGen/Mips/compact-branch-combine-never.ll
@@ -1,6 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -mtriple=mipsel -mcpu=mips32r6 -mips-compact-branches=never < %s | FileCheck %s --check-prefix=MIPS32R6
-; RUN: llc -mtriple=mips64el -mcpu=mips64r6 -mips-compact-branches=never < %s | FileCheck %s --check-prefix=MIPS64R6
+; Keep the branches being tested from folding into a common return.
+; RUN: llc -enable-tail-merge=false -mtriple=mipsel -mcpu=mips32r6 -mips-compact-branches=never < %s | FileCheck %s --check-prefix=MIPS32R6
+; RUN: llc -enable-tail-merge=false -mtriple=mips64el -mcpu=mips64r6 -mips-compact-branches=never < %s | FileCheck %s --check-prefix=MIPS64R6
 
 ;; Test checking we respect mips-compact-branches=never
 ;; The patterns set + branch should be disabled and not emit compact branches
diff --git a/llvm/test/CodeGen/Mips/compact-branch-combine.ll b/llvm/test/CodeGen/Mips/compact-branch-combine.ll
index 27a48e509deaba..9b7755734a5f9e 100644
--- a/llvm/test/CodeGen/Mips/compact-branch-combine.ll
+++ b/llvm/test/CodeGen/Mips/compact-branch-combine.ll
@@ -1,6 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -mtriple=mipsel -mcpu=mips32r6 < %s | FileCheck %s --check-prefix=MIPS32R6
-; RUN: llc -mtriple=mips64el -mcpu=mips64r6 < %s | FileCheck %s --check-prefix=MIPS64R6
+; Keep the branches being tested from folding into a common return.
+; RUN: llc -enable-tail-merge=false -mtriple=mipsel -mcpu=mips32r6 < %s | FileCheck %s --check-prefix=MIPS32R6
+; RUN: llc -enable-tail-merge=false -mtriple=mips64el -mcpu=mips64r6 < %s | FileCheck %s --check-prefix=MIPS64R6
 
 ;; Each function is a single compare + branch
 ;; Checking each pattern for compact branch is selected
diff --git a/llvm/test/CodeGen/Mips/disable-tail-merge.ll b/llvm/test/CodeGen/Mips/disable-tail-merge.ll
deleted file mode 100644
index 8e6174c4b3c483..00000000000000
--- a/llvm/test/CodeGen/Mips/disable-tail-merge.ll
+++ /dev/null
@@ -1,33 +0,0 @@
-; RUN: llc -mtriple=mipsel < %s | FileCheck %s
-
- at g0 = common global i32 0, align 4
- at g1 = common global i32 0, align 4
-
-; CHECK: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
-; CHECK: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
-
-define i32 @test1(i32 %a) {
-entry:
-  %tobool = icmp eq i32 %a, 0
-  %0 = load i32, ptr @g0, align 4
-  br i1 %tobool, label %if.else, label %if.then
-
-if.then:
-  %add = add nsw i32 %0, 1
-  store i32 %add, ptr @g0, align 4
-  %1 = load i32, ptr @g1, align 4
-  %add1 = add nsw i32 %1, 23
-  br label %if.end
-
-if.else:
-  %add2 = add nsw i32 %0, 11
-  store i32 %add2, ptr @g0, align 4
-  %2 = load i32, ptr @g1, align 4
-  %add3 = add nsw i32 %2, 23
-  br label %if.end
-
-if.end:
-  %storemerge = phi i32 [ %add3, %if.else ], [ %add1, %if.then ]
-  store i32 %storemerge, ptr @g1, align 4
-  ret i32 %storemerge
-}
diff --git a/llvm/test/CodeGen/Mips/llvm-ir/forbidden-slot-ir.ll b/llvm/test/CodeGen/Mips/llvm-ir/forbidden-slot-ir.ll
index 3e6826f0cd1d13..78f5f1e5ef4438 100644
--- a/llvm/test/CodeGen/Mips/llvm-ir/forbidden-slot-ir.ll
+++ b/llvm/test/CodeGen/Mips/llvm-ir/forbidden-slot-ir.ll
@@ -1,6 +1,7 @@
+; Keep the branch layout being tested independent of tail merging.
 target triple = "mipsisa32r6el-unknown-linux-gnu"
 
-; RUN: llc -filetype=asm %s -o - | FileCheck %s --check-prefix=MIPSELR6
+; RUN: llc -enable-tail-merge=false -filetype=asm %s -o - | FileCheck %s --check-prefix=MIPSELR6
 ; Function Attrs: noinline nounwind optnone uwtable
 define i1 @foo0() nounwind {
 ; MIPSELR6:      bnezc	$1, $BB0_2
diff --git a/llvm/test/CodeGen/Mips/mips1-load-in-delay-slot.ll b/llvm/test/CodeGen/Mips/mips1-load-in-delay-slot.ll
index d0cb5601cfd866..c94864930c489d 100644
--- a/llvm/test/CodeGen/Mips/mips1-load-in-delay-slot.ll
+++ b/llvm/test/CodeGen/Mips/mips1-load-in-delay-slot.ll
@@ -1,5 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc < %s -mtriple=mipsel-sony-psx -mcpu=mips1 -relocation-model=pic | FileCheck %s -check-prefixes=MIPS1-PSX
+; Keep the branch whose delay slot could otherwise take the load.
+; RUN: llc < %s -mtriple=mipsel-sony-psx -mcpu=mips1 -relocation-model=pic -enable-tail-merge=false | FileCheck %s -check-prefixes=MIPS1-PSX
 
 @data = internal global <{ [1 x i8] }> <{ [1 x i8] c"\00" }>, align 4
 
diff --git a/llvm/test/CodeGen/Mips/pseudo-jump-fill.ll b/llvm/test/CodeGen/Mips/pseudo-jump-fill.ll
index 16186c9be9e265..afcd1a2f715003 100644
--- a/llvm/test/CodeGen/Mips/pseudo-jump-fill.ll
+++ b/llvm/test/CodeGen/Mips/pseudo-jump-fill.ll
@@ -12,7 +12,7 @@ define i32 @test(i32 signext %x, i32 signext %c) {
 ; CHECK-NEXT:    addiu $2, $2, %lo(_gp_disp)
 ; CHECK-NEXT:    addiur2 $5, $5, -1
 ; CHECK-NEXT:    sltiu $1, $5, 4
-; CHECK-NEXT:    beqz $1, $BB0_6
+; CHECK-NEXT:    beqz $1, $BB0_5
 ; CHECK-NEXT:    addu $3, $2, $25
 ; CHECK-NEXT:  # %bb.1: # %entry
 ; CHECK-NEXT:    li16 $2, 0
@@ -27,16 +27,13 @@ define i32 @test(i32 signext %x, i32 signext %c) {
 ; CHECK-NEXT:    addiur2 $2, $4, 1
 ; CHECK-NEXT:    jrc $ra
 ; CHECK-NEXT:  $BB0_3: # %sw.bb3
+; CHECK-NEXT:    b $BB0_5
 ; CHECK-NEXT:    addius5 $4, 2
-; CHECK-NEXT:    move $2, $4
-; CHECK-NEXT:    jrc $ra
 ; CHECK-NEXT:  $BB0_4: # %sw.bb5
 ; CHECK-NEXT:    addius5 $4, 3
+; CHECK-NEXT:  $BB0_5: # %sw.epilog
 ; CHECK-NEXT:    move $2, $4
-; CHECK-NEXT:  $BB0_5: # %for.cond.cleanup
-; CHECK-NEXT:    jrc $ra
-; CHECK-NEXT:  $BB0_6:
-; CHECK-NEXT:    move $2, $4
+; CHECK-NEXT:  $BB0_6: # %for.cond.cleanup
 ; CHECK-NEXT:    jrc $ra
 entry:
   switch i32 %c, label %sw.epilog [
diff --git a/llvm/test/CodeGen/Mips/tail-merge.ll b/llvm/test/CodeGen/Mips/tail-merge.ll
new file mode 100644
index 00000000000000..c18c547e609d04
--- /dev/null
+++ b/llvm/test/CodeGen/Mips/tail-merge.ll
@@ -0,0 +1,88 @@
+; RUN: split-file %s %t
+; RUN: llc -mtriple=mipsel -verify-machineinstrs < %t/test.ll | FileCheck %s
+; RUN: llc -mtriple=mipsel -relocation-model=pic -force-mips-long-branch -verify-machineinstrs < %t/test.ll | FileCheck %s
+; RUN: llc -mtriple=mipsel -mattr=+reserve-gpr1 -verify-machineinstrs < %t/test.ll | FileCheck %s --check-prefix=SAFE
+; RUN: llc -mtriple=mipsel -run-pass=branch-folder -enable-tail-merge -verify-machineinstrs %t/reuse.mir -o - | FileCheck %s --check-prefix=REUSE
+; RUN: llc -mtriple=mipsel -mattr=+mips16,+soft-float -verify-machineinstrs < %t/test.ll | FileCheck %s --check-prefix=MIPS16
+
+; Keep the two tails separate when merging would leave $at live across a branch.
+; Reserving $at makes the same tail safe to merge.
+; SAFE: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
+; SAFE-NOT: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
+
+; MIPS16 keeps tail merging disabled.
+; MIPS16: addiu ${{[0-9]+}}, 23
+; MIPS16: addiu ${{[0-9]+}}, 23
+
+;--- test.ll
+
+ at g0 = common global i32 0, align 4
+ at g1 = common global i32 0, align 4
+
+; CHECK: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
+; CHECK: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
+
+define i32 @test1(i32 %a) {
+entry:
+  %tobool = icmp eq i32 %a, 0
+  %0 = load i32, ptr @g0, align 4
+  br i1 %tobool, label %if.else, label %if.then
+
+if.then:
+  %add = add nsw i32 %0, 1
+  store i32 %add, ptr @g0, align 4
+  %1 = load i32, ptr @g1, align 4
+  %add1 = add nsw i32 %1, 23
+  br label %if.end
+
+if.else:
+  %add2 = add nsw i32 %0, 11
+  store i32 %add2, ptr @g0, align 4
+  %2 = load i32, ptr @g1, align 4
+  %add3 = add nsw i32 %2, 23
+  br label %if.end
+
+if.end:
+  %storemerge = phi i32 [ %add3, %if.else ], [ %add1, %if.then ]
+  store i32 %storemerge, ptr @g1, align 4
+  ret i32 %storemerge
+}
+
+;--- reuse.mir
+# A whole-block common tail must obey the same legality check as a split tail.
+# The first shared instruction does not use $at, but a later instruction does.
+# REUSE-LABEL: name: reuse
+# REUSE: $at = ADDiu $a1, {{[12]}}
+# REUSE-NEXT: $v0 = LW $a3, 0
+# REUSE-NEXT: $v0 = ADDu $v0, $at
+# REUSE: $at = ADDiu $a1, {{[12]}}
+# REUSE-NEXT: $v0 = LW $a3, 0
+# REUSE-NEXT: $v0 = ADDu $v0, $at
+---
+name: reuse
+tracksRegLiveness: true
+body: |
+  bb.0:
+    successors: %bb.1, %bb.2
+    liveins: $a0, $a1, $a3
+    BEQ $a0, $zero, %bb.2, implicit-def $at
+    B %bb.1, implicit-def $at
+
+  bb.1:
+    liveins: $a1, $a3
+    $at = ADDiu $a1, 1
+    $v0 = LW $a3, 0
+    $v0 = ADDu $v0, $at
+    RetRA implicit $v0
+
+  bb.2:
+    successors: %bb.3
+    liveins: $a1, $a3
+    $at = ADDiu $a1, 2
+
+  bb.3:
+    liveins: $at, $a3
+    $v0 = LW $a3, 0
+    $v0 = ADDu $v0, $at
+    RetRA implicit $v0
+...



More information about the llvm-commits mailing list