[llvm] [Mips][BranchFolding] Enable safe tail merging (PR #225382)
Jiaxun Yang via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 22 05:27:25 PDT 2026
https://github.com/FlyGoat updated https://github.com/llvm/llvm-project/pull/225382
>From cea0c6c0b85514d4e0613cb07d183ffc22072d05 Mon Sep 17 00:00:00 2001
From: Jiaxun Yang <jiaxun.yang at flygoat.com>
Date: Tue, 22 Sep 2026 12:44:24 +0100
Subject: [PATCH] [Mips][BranchFolding] Enable safe tail merging
MIPS disables tail merging when the long-branch pass is enabled because
branch expansion may clobber $at. This also blocks merges where $at is
dead at the new branch.
Use physical-register liveness to reject tails that need $at, allowing
safe merges for standard MIPS and microMIPS. Keep MIPS16 tail merging
disabled because its far branches clobber $ra.
Check both candidate tails before selecting a common tail. The existing
split check is only reached when creating a new block; reusing an entire
block can still introduce a branch without passing that check.
Assisted-by: chatgpt-5.6-luna # Testcases
---
llvm/lib/CodeGen/BranchFolding.cpp | 12 +--
llvm/lib/Target/Mips/MipsInstrInfo.cpp | 19 ++++
llvm/lib/Target/Mips/MipsInstrInfo.h | 3 +
llvm/lib/Target/Mips/MipsTargetMachine.cpp | 5 --
.../Mips/compact-branch-combine-never.ll | 5 +-
.../CodeGen/Mips/compact-branch-combine.ll | 5 +-
llvm/test/CodeGen/Mips/disable-tail-merge.ll | 33 -------
.../CodeGen/Mips/llvm-ir/forbidden-slot-ir.ll | 3 +-
.../CodeGen/Mips/mips1-load-in-delay-slot.ll | 3 +-
llvm/test/CodeGen/Mips/pseudo-jump-fill.ll | 11 +--
llvm/test/CodeGen/Mips/tail-merge.ll | 88 +++++++++++++++++++
11 files changed, 131 insertions(+), 56 deletions(-)
delete mode 100644 llvm/test/CodeGen/Mips/disable-tail-merge.ll
create mode 100644 llvm/test/CodeGen/Mips/tail-merge.ll
diff --git a/llvm/lib/CodeGen/BranchFolding.cpp b/llvm/lib/CodeGen/BranchFolding.cpp
index aef63f6bdc1c52..3dafdcb6d65aa1 100644
--- a/llvm/lib/CodeGen/BranchFolding.cpp
+++ b/llvm/lib/CodeGen/BranchFolding.cpp
@@ -754,11 +754,13 @@ unsigned BranchFolder::ComputeSameTails(unsigned CurHash,
for (MPIterator I = std::prev(CurMPIter); I->getHash() == CurHash; --I) {
unsigned CommonTailLen;
if (ProfitableToMerge(CurMPIter->getBlock(), I->getBlock(),
- MinCommonTailLength,
- CommonTailLen, TrialBBI1, TrialBBI2,
- SuccBB, PredBB,
- EHScopeMembership,
- AfterBlockPlacement, MBBFreqInfo, PSI)) {
+ MinCommonTailLength, CommonTailLen, TrialBBI1,
+ TrialBBI2, SuccBB, PredBB, EHScopeMembership,
+ AfterBlockPlacement, MBBFreqInfo, PSI) &&
+ // Check both tails even if one is an entire block: replacing the
+ // other tail can introduce a branch at its start.
+ TII->isLegalToSplitMBBAt(*CurMPIter->getBlock(), TrialBBI1) &&
+ TII->isLegalToSplitMBBAt(*I->getBlock(), TrialBBI2)) {
if (CommonTailLen > maxCommonTailLength) {
SameTails.clear();
maxCommonTailLength = CommonTailLen;
diff --git a/llvm/lib/Target/Mips/MipsInstrInfo.cpp b/llvm/lib/Target/Mips/MipsInstrInfo.cpp
index 3dcd9c712b90d0..c11bf9af98c54d 100644
--- a/llvm/lib/Target/Mips/MipsInstrInfo.cpp
+++ b/llvm/lib/Target/Mips/MipsInstrInfo.cpp
@@ -16,12 +16,14 @@
#include "Mips.h"
#include "MipsSubtarget.h"
#include "llvm/ADT/SmallVector.h"
+#include "llvm/CodeGen/LivePhysRegs.h"
#include "llvm/CodeGen/MachineBasicBlock.h"
#include "llvm/CodeGen/MachineFrameInfo.h"
#include "llvm/CodeGen/MachineFunction.h"
#include "llvm/CodeGen/MachineInstr.h"
#include "llvm/CodeGen/MachineInstrBuilder.h"
#include "llvm/CodeGen/MachineOperand.h"
+#include "llvm/CodeGen/MachineRegisterInfo.h"
#include "llvm/CodeGen/TargetOpcodes.h"
#include "llvm/CodeGen/TargetSubtargetInfo.h"
#include "llvm/IR/DebugInfoMetadata.h"
@@ -57,6 +59,23 @@ const MipsInstrInfo *MipsInstrInfo::create(MipsSubtarget &STI) {
return createMipsSEInstrInfo(STI);
}
+bool MipsInstrInfo::isLegalToSplitMBBAt(
+ MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI) const {
+ // Keep tail merging disabled for MIPS16, whose far branches clobber $ra.
+ if (Subtarget.inMips16Mode())
+ return false;
+
+ // Long branch expansion clobbers $at. Do not let tail merging introduce a
+ // branch while $at (or its 64-bit super-register) is live.
+ if (!MBB.getParent()->getRegInfo().tracksLiveness())
+ return false;
+ LivePhysRegs LiveRegs(*Subtarget.getRegisterInfo());
+ LiveRegs.addLiveOuts(MBB);
+ for (auto I = MBB.end(); I != MBBI;)
+ LiveRegs.stepBackward(*--I);
+ return !LiveRegs.contains(Mips::AT);
+}
+
bool MipsInstrInfo::isZeroImm(const MachineOperand &op) const {
return op.isImm() && op.getImm() == 0;
}
diff --git a/llvm/lib/Target/Mips/MipsInstrInfo.h b/llvm/lib/Target/Mips/MipsInstrInfo.h
index 8772fc8ee28ef1..15b292d14bd434 100644
--- a/llvm/lib/Target/Mips/MipsInstrInfo.h
+++ b/llvm/lib/Target/Mips/MipsInstrInfo.h
@@ -77,6 +77,9 @@ class MipsInstrInfo : public MipsGenInstrInfo {
const DebugLoc &DL,
int *BytesAdded = nullptr) const override;
+ bool isLegalToSplitMBBAt(MachineBasicBlock &MBB,
+ MachineBasicBlock::iterator MBBI) const override;
+
bool
reverseBranchCondition(SmallVectorImpl<MachineOperand> &Cond) const override;
diff --git a/llvm/lib/Target/Mips/MipsTargetMachine.cpp b/llvm/lib/Target/Mips/MipsTargetMachine.cpp
index 3b4a6aa0b7a543..f2e0a0aa145c56 100644
--- a/llvm/lib/Target/Mips/MipsTargetMachine.cpp
+++ b/llvm/lib/Target/Mips/MipsTargetMachine.cpp
@@ -191,11 +191,6 @@ class MipsPassConfig : public TargetPassConfig {
public:
MipsPassConfig(MipsTargetMachine &TM, PassManagerBase &PM)
: TargetPassConfig(TM, PM) {
- // The current implementation of long branch pass requires a scratch
- // register ($at) to be available before branch instructions. Tail merging
- // can break this requirement, so disable it when long branch pass is
- // enabled.
- EnableTailMerge = !getMipsSubtarget().enableLongBranchPass();
EnableLoopTermFold = true;
}
diff --git a/llvm/test/CodeGen/Mips/compact-branch-combine-never.ll b/llvm/test/CodeGen/Mips/compact-branch-combine-never.ll
index 74e908d08248c2..fb05d3fc690142 100644
--- a/llvm/test/CodeGen/Mips/compact-branch-combine-never.ll
+++ b/llvm/test/CodeGen/Mips/compact-branch-combine-never.ll
@@ -1,6 +1,7 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -mtriple=mipsel -mcpu=mips32r6 -mips-compact-branches=never < %s | FileCheck %s --check-prefix=MIPS32R6
-; RUN: llc -mtriple=mips64el -mcpu=mips64r6 -mips-compact-branches=never < %s | FileCheck %s --check-prefix=MIPS64R6
+; Keep the branches being tested from folding into a common return.
+; RUN: llc -enable-tail-merge=false -mtriple=mipsel -mcpu=mips32r6 -mips-compact-branches=never < %s | FileCheck %s --check-prefix=MIPS32R6
+; RUN: llc -enable-tail-merge=false -mtriple=mips64el -mcpu=mips64r6 -mips-compact-branches=never < %s | FileCheck %s --check-prefix=MIPS64R6
;; Test checking we respect mips-compact-branches=never
;; The patterns set + branch should be disabled and not emit compact branches
diff --git a/llvm/test/CodeGen/Mips/compact-branch-combine.ll b/llvm/test/CodeGen/Mips/compact-branch-combine.ll
index 27a48e509deaba..9b7755734a5f9e 100644
--- a/llvm/test/CodeGen/Mips/compact-branch-combine.ll
+++ b/llvm/test/CodeGen/Mips/compact-branch-combine.ll
@@ -1,6 +1,7 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -mtriple=mipsel -mcpu=mips32r6 < %s | FileCheck %s --check-prefix=MIPS32R6
-; RUN: llc -mtriple=mips64el -mcpu=mips64r6 < %s | FileCheck %s --check-prefix=MIPS64R6
+; Keep the branches being tested from folding into a common return.
+; RUN: llc -enable-tail-merge=false -mtriple=mipsel -mcpu=mips32r6 < %s | FileCheck %s --check-prefix=MIPS32R6
+; RUN: llc -enable-tail-merge=false -mtriple=mips64el -mcpu=mips64r6 < %s | FileCheck %s --check-prefix=MIPS64R6
;; Each function is a single compare + branch
;; Checking each pattern for compact branch is selected
diff --git a/llvm/test/CodeGen/Mips/disable-tail-merge.ll b/llvm/test/CodeGen/Mips/disable-tail-merge.ll
deleted file mode 100644
index 8e6174c4b3c483..00000000000000
--- a/llvm/test/CodeGen/Mips/disable-tail-merge.ll
+++ /dev/null
@@ -1,33 +0,0 @@
-; RUN: llc -mtriple=mipsel < %s | FileCheck %s
-
- at g0 = common global i32 0, align 4
- at g1 = common global i32 0, align 4
-
-; CHECK: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
-; CHECK: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
-
-define i32 @test1(i32 %a) {
-entry:
- %tobool = icmp eq i32 %a, 0
- %0 = load i32, ptr @g0, align 4
- br i1 %tobool, label %if.else, label %if.then
-
-if.then:
- %add = add nsw i32 %0, 1
- store i32 %add, ptr @g0, align 4
- %1 = load i32, ptr @g1, align 4
- %add1 = add nsw i32 %1, 23
- br label %if.end
-
-if.else:
- %add2 = add nsw i32 %0, 11
- store i32 %add2, ptr @g0, align 4
- %2 = load i32, ptr @g1, align 4
- %add3 = add nsw i32 %2, 23
- br label %if.end
-
-if.end:
- %storemerge = phi i32 [ %add3, %if.else ], [ %add1, %if.then ]
- store i32 %storemerge, ptr @g1, align 4
- ret i32 %storemerge
-}
diff --git a/llvm/test/CodeGen/Mips/llvm-ir/forbidden-slot-ir.ll b/llvm/test/CodeGen/Mips/llvm-ir/forbidden-slot-ir.ll
index 3e6826f0cd1d13..78f5f1e5ef4438 100644
--- a/llvm/test/CodeGen/Mips/llvm-ir/forbidden-slot-ir.ll
+++ b/llvm/test/CodeGen/Mips/llvm-ir/forbidden-slot-ir.ll
@@ -1,6 +1,7 @@
+; Keep the branch layout being tested independent of tail merging.
target triple = "mipsisa32r6el-unknown-linux-gnu"
-; RUN: llc -filetype=asm %s -o - | FileCheck %s --check-prefix=MIPSELR6
+; RUN: llc -enable-tail-merge=false -filetype=asm %s -o - | FileCheck %s --check-prefix=MIPSELR6
; Function Attrs: noinline nounwind optnone uwtable
define i1 @foo0() nounwind {
; MIPSELR6: bnezc $1, $BB0_2
diff --git a/llvm/test/CodeGen/Mips/mips1-load-in-delay-slot.ll b/llvm/test/CodeGen/Mips/mips1-load-in-delay-slot.ll
index d0cb5601cfd866..c94864930c489d 100644
--- a/llvm/test/CodeGen/Mips/mips1-load-in-delay-slot.ll
+++ b/llvm/test/CodeGen/Mips/mips1-load-in-delay-slot.ll
@@ -1,5 +1,6 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc < %s -mtriple=mipsel-sony-psx -mcpu=mips1 -relocation-model=pic | FileCheck %s -check-prefixes=MIPS1-PSX
+; Keep the branch whose delay slot could otherwise take the load.
+; RUN: llc < %s -mtriple=mipsel-sony-psx -mcpu=mips1 -relocation-model=pic -enable-tail-merge=false | FileCheck %s -check-prefixes=MIPS1-PSX
@data = internal global <{ [1 x i8] }> <{ [1 x i8] c"\00" }>, align 4
diff --git a/llvm/test/CodeGen/Mips/pseudo-jump-fill.ll b/llvm/test/CodeGen/Mips/pseudo-jump-fill.ll
index 16186c9be9e265..afcd1a2f715003 100644
--- a/llvm/test/CodeGen/Mips/pseudo-jump-fill.ll
+++ b/llvm/test/CodeGen/Mips/pseudo-jump-fill.ll
@@ -12,7 +12,7 @@ define i32 @test(i32 signext %x, i32 signext %c) {
; CHECK-NEXT: addiu $2, $2, %lo(_gp_disp)
; CHECK-NEXT: addiur2 $5, $5, -1
; CHECK-NEXT: sltiu $1, $5, 4
-; CHECK-NEXT: beqz $1, $BB0_6
+; CHECK-NEXT: beqz $1, $BB0_5
; CHECK-NEXT: addu $3, $2, $25
; CHECK-NEXT: # %bb.1: # %entry
; CHECK-NEXT: li16 $2, 0
@@ -27,16 +27,13 @@ define i32 @test(i32 signext %x, i32 signext %c) {
; CHECK-NEXT: addiur2 $2, $4, 1
; CHECK-NEXT: jrc $ra
; CHECK-NEXT: $BB0_3: # %sw.bb3
+; CHECK-NEXT: b $BB0_5
; CHECK-NEXT: addius5 $4, 2
-; CHECK-NEXT: move $2, $4
-; CHECK-NEXT: jrc $ra
; CHECK-NEXT: $BB0_4: # %sw.bb5
; CHECK-NEXT: addius5 $4, 3
+; CHECK-NEXT: $BB0_5: # %sw.epilog
; CHECK-NEXT: move $2, $4
-; CHECK-NEXT: $BB0_5: # %for.cond.cleanup
-; CHECK-NEXT: jrc $ra
-; CHECK-NEXT: $BB0_6:
-; CHECK-NEXT: move $2, $4
+; CHECK-NEXT: $BB0_6: # %for.cond.cleanup
; CHECK-NEXT: jrc $ra
entry:
switch i32 %c, label %sw.epilog [
diff --git a/llvm/test/CodeGen/Mips/tail-merge.ll b/llvm/test/CodeGen/Mips/tail-merge.ll
new file mode 100644
index 00000000000000..c18c547e609d04
--- /dev/null
+++ b/llvm/test/CodeGen/Mips/tail-merge.ll
@@ -0,0 +1,88 @@
+; RUN: split-file %s %t
+; RUN: llc -mtriple=mipsel -verify-machineinstrs < %t/test.ll | FileCheck %s
+; RUN: llc -mtriple=mipsel -relocation-model=pic -force-mips-long-branch -verify-machineinstrs < %t/test.ll | FileCheck %s
+; RUN: llc -mtriple=mipsel -mattr=+reserve-gpr1 -verify-machineinstrs < %t/test.ll | FileCheck %s --check-prefix=SAFE
+; RUN: llc -mtriple=mipsel -run-pass=branch-folder -enable-tail-merge -verify-machineinstrs %t/reuse.mir -o - | FileCheck %s --check-prefix=REUSE
+; RUN: llc -mtriple=mipsel -mattr=+mips16,+soft-float -verify-machineinstrs < %t/test.ll | FileCheck %s --check-prefix=MIPS16
+
+; Keep the two tails separate when merging would leave $at live across a branch.
+; Reserving $at makes the same tail safe to merge.
+; SAFE: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
+; SAFE-NOT: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
+
+; MIPS16 keeps tail merging disabled.
+; MIPS16: addiu ${{[0-9]+}}, 23
+; MIPS16: addiu ${{[0-9]+}}, 23
+
+;--- test.ll
+
+ at g0 = common global i32 0, align 4
+ at g1 = common global i32 0, align 4
+
+; CHECK: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
+; CHECK: addiu ${{[0-9]+}}, ${{[0-9]+}}, 23
+
+define i32 @test1(i32 %a) {
+entry:
+ %tobool = icmp eq i32 %a, 0
+ %0 = load i32, ptr @g0, align 4
+ br i1 %tobool, label %if.else, label %if.then
+
+if.then:
+ %add = add nsw i32 %0, 1
+ store i32 %add, ptr @g0, align 4
+ %1 = load i32, ptr @g1, align 4
+ %add1 = add nsw i32 %1, 23
+ br label %if.end
+
+if.else:
+ %add2 = add nsw i32 %0, 11
+ store i32 %add2, ptr @g0, align 4
+ %2 = load i32, ptr @g1, align 4
+ %add3 = add nsw i32 %2, 23
+ br label %if.end
+
+if.end:
+ %storemerge = phi i32 [ %add3, %if.else ], [ %add1, %if.then ]
+ store i32 %storemerge, ptr @g1, align 4
+ ret i32 %storemerge
+}
+
+;--- reuse.mir
+# A whole-block common tail must obey the same legality check as a split tail.
+# The first shared instruction does not use $at, but a later instruction does.
+# REUSE-LABEL: name: reuse
+# REUSE: $at = ADDiu $a1, {{[12]}}
+# REUSE-NEXT: $v0 = LW $a3, 0
+# REUSE-NEXT: $v0 = ADDu $v0, $at
+# REUSE: $at = ADDiu $a1, {{[12]}}
+# REUSE-NEXT: $v0 = LW $a3, 0
+# REUSE-NEXT: $v0 = ADDu $v0, $at
+---
+name: reuse
+tracksRegLiveness: true
+body: |
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $a0, $a1, $a3
+ BEQ $a0, $zero, %bb.2, implicit-def $at
+ B %bb.1, implicit-def $at
+
+ bb.1:
+ liveins: $a1, $a3
+ $at = ADDiu $a1, 1
+ $v0 = LW $a3, 0
+ $v0 = ADDu $v0, $at
+ RetRA implicit $v0
+
+ bb.2:
+ successors: %bb.3
+ liveins: $a1, $a3
+ $at = ADDiu $a1, 2
+
+ bb.3:
+ liveins: $at, $a3
+ $v0 = LW $a3, 0
+ $v0 = ADDu $v0, $at
+ RetRA implicit $v0
+...
More information about the llvm-commits
mailing list