[llvm] [CodeGen] Don't rotate shared increment blocks to loop top (PR #219126)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 23 20:50:03 PDT 2026
https://github.com/compilersutra updated https://github.com/llvm/llvm-project/pull/219126
>From b447cb07b291cd689a565f1aff6c703706a50d0f Mon Sep 17 00:00:00 2001
From: compilersutra <osc at compilersutra.com>
Date: Thu, 27 Aug 2026 11:52:21 +0530
Subject: [PATCH] [CodeGen] Don't rotate shared increment blocks to loop top
canMoveBottomBlockToTop only looked at a single predecessor, so a
diamond like c == ',' || c == '\n' got rotated and inverted the
likely fallthrough.
Fixes #218248
---
llvm/lib/CodeGen/MachineBlockPlacement.cpp | 35 ++++++++++----
llvm/test/CodeGen/AArch64/peephole-and-tst.ll | 22 ++++-----
.../AArch64/regalloc-spill-weight-basic.ll | 47 +++++++++----------
.../block-placement-loop-top-multi-succ.ll | 44 +++++++++++++++++
llvm/test/CodeGen/X86/fold-loop-of-urem.ll | 26 +++++-----
5 files changed, 116 insertions(+), 58 deletions(-)
create mode 100644 llvm/test/CodeGen/X86/block-placement-loop-top-multi-succ.ll
diff --git a/llvm/lib/CodeGen/MachineBlockPlacement.cpp b/llvm/lib/CodeGen/MachineBlockPlacement.cpp
index a748150d98c64..84215791515f6 100644
--- a/llvm/lib/CodeGen/MachineBlockPlacement.cpp
+++ b/llvm/lib/CodeGen/MachineBlockPlacement.cpp
@@ -2007,19 +2007,27 @@ void MachineBlockPlacement::buildChain(const MachineBasicBlock *HeadBB,
// If BB is moved before OldTop, Pred needs a taken branch to BB, and it can't
// layout the other successor below it, so it can't reduce taken branch.
// In this case we keep its original layout.
+//
+// Pred may not be BB's only predecessor: several 2-way branches can share the
+// same BB (both sides of `c == ',' || c == '\n'`). Treat any such Pred the
+// same way.
bool MachineBlockPlacement::canMoveBottomBlockToTop(
const MachineBasicBlock *BottomBlock, const MachineBasicBlock *OldTop) {
- if (BottomBlock->pred_size() != 1)
- return true;
- MachineBasicBlock *Pred = *BottomBlock->pred_begin();
- if (Pred->succ_size() != 2)
- return true;
+ // BB may have several predecessors that share this diamond (e.g. comma and
+ // newline both jump to the same increment). Reject the rotate if any of them
+ // has OldTop as the other successor.
+ for (const MachineBasicBlock *Pred : BottomBlock->predecessors()) {
+ if (Pred == OldTop)
+ continue;
+ if (Pred->succ_size() != 2)
+ continue;
- MachineBasicBlock *OtherBB = *Pred->succ_begin();
- if (OtherBB == BottomBlock)
- OtherBB = *Pred->succ_rbegin();
- if (OtherBB == OldTop)
- return false;
+ const MachineBasicBlock *OtherBB = *Pred->succ_begin();
+ if (OtherBB == BottomBlock)
+ OtherBB = *Pred->succ_rbegin();
+ if (OtherBB == OldTop)
+ return false;
+ }
return true;
}
@@ -2228,6 +2236,13 @@ MachineBasicBlock *MachineBlockPlacement::findBestLoopTopHelper(
*BestPred->pred_begin() != L.getHeader())
BestPred = *BestPred->pred_begin();
+ // The walk-back can land on a block that was not the predecessor FallThrough
+ // gains considered. Re-check the diamond-into-top shape.
+ if (!canMoveBottomBlockToTop(BestPred, OldTop)) {
+ LLVM_DEBUG(dbgs() << " final top unchanged\n");
+ return OldTop;
+ }
+
LLVM_DEBUG(dbgs() << " final top: " << getBlockName(BestPred) << "\n");
return BestPred;
}
diff --git a/llvm/test/CodeGen/AArch64/peephole-and-tst.ll b/llvm/test/CodeGen/AArch64/peephole-and-tst.ll
index 6449f5d5f07d3..628d0b4a59c64 100644
--- a/llvm/test/CodeGen/AArch64/peephole-and-tst.ll
+++ b/llvm/test/CodeGen/AArch64/peephole-and-tst.ll
@@ -39,26 +39,26 @@ define i32 @test_func_i32_two_uses(i32 %in, i32 %bit, i32 %mask) {
; CHECK-GI-NEXT: ldr x8, [x8, :got_lo12:ptr_wrapper]
; CHECK-GI-NEXT: ldr x9, [x8]
; CHECK-GI-NEXT: mov w8, wzr
-; CHECK-GI-NEXT: b .LBB0_3
-; CHECK-GI-NEXT: .LBB0_1: // in Loop: Header=BB0_3 Depth=1
-; CHECK-GI-NEXT: str xzr, [x9, #8]
-; CHECK-GI-NEXT: .LBB0_2: // in Loop: Header=BB0_3 Depth=1
+; CHECK-GI-NEXT: b .LBB0_2
+; CHECK-GI-NEXT: .LBB0_1: // in Loop: Header=BB0_2 Depth=1
; CHECK-GI-NEXT: lsl w1, w1, #1
; CHECK-GI-NEXT: cbz w1, .LBB0_6
-; CHECK-GI-NEXT: .LBB0_3: // %do.body
+; CHECK-GI-NEXT: .LBB0_2: // %do.body
; CHECK-GI-NEXT: // =>This Inner Loop Header: Depth=1
; CHECK-GI-NEXT: and w10, w1, w0
; CHECK-GI-NEXT: tst w1, w0
; CHECK-GI-NEXT: and w11, w2, w0
; CHECK-GI-NEXT: cinc w8, w8, ne
; CHECK-GI-NEXT: cmp w10, w11
-; CHECK-GI-NEXT: b.eq .LBB0_1
+; CHECK-GI-NEXT: b.eq .LBB0_5
+; CHECK-GI-NEXT: // %bb.3: // %do.body
+; CHECK-GI-NEXT: // in Loop: Header=BB0_2 Depth=1
+; CHECK-GI-NEXT: cbnz w2, .LBB0_5
; CHECK-GI-NEXT: // %bb.4: // %do.body
-; CHECK-GI-NEXT: // in Loop: Header=BB0_3 Depth=1
-; CHECK-GI-NEXT: cbnz w2, .LBB0_1
-; CHECK-GI-NEXT: // %bb.5: // %do.body
-; CHECK-GI-NEXT: // in Loop: Header=BB0_3 Depth=1
-; CHECK-GI-NEXT: cbz w10, .LBB0_2
+; CHECK-GI-NEXT: // in Loop: Header=BB0_2 Depth=1
+; CHECK-GI-NEXT: cbz w10, .LBB0_1
+; CHECK-GI-NEXT: .LBB0_5: // in Loop: Header=BB0_2 Depth=1
+; CHECK-GI-NEXT: str xzr, [x9, #8]
; CHECK-GI-NEXT: b .LBB0_1
; CHECK-GI-NEXT: .LBB0_6: // %do.end
; CHECK-GI-NEXT: mov w0, w8
diff --git a/llvm/test/CodeGen/AArch64/regalloc-spill-weight-basic.ll b/llvm/test/CodeGen/AArch64/regalloc-spill-weight-basic.ll
index 5c3bd984087ec..697b00da5a0ce 100644
--- a/llvm/test/CodeGen/AArch64/regalloc-spill-weight-basic.ll
+++ b/llvm/test/CodeGen/AArch64/regalloc-spill-weight-basic.ll
@@ -102,37 +102,36 @@ define void @optspeed(i32 %arg, i32 %arg1, ptr %arg2, ptr %arg3, ptr %arg4, i32
; CHECK-NEXT: mov x20, x3
; CHECK-NEXT: mov x23, x2
; CHECK-NEXT: mov w19, w1
-; CHECK-NEXT: b .LBB1_2
-; CHECK-NEXT: .LBB1_1: // %bb10
-; CHECK-NEXT: // in Loop: Header=BB1_2 Depth=1
-; CHECK-NEXT: mov w0, w22
-; CHECK-NEXT: mov x1, x20
-; CHECK-NEXT: str wzr, [x21]
-; CHECK-NEXT: bl foo
-; CHECK-NEXT: .LBB1_2: // %bb8
+; CHECK-NEXT: .LBB1_1: // %bb8
; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
; CHECK-NEXT: cmp w19, #33
-; CHECK-NEXT: b.gt .LBB1_6
+; CHECK-NEXT: b.gt .LBB1_5
+; CHECK-NEXT: // %bb.2: // %bb8
+; CHECK-NEXT: // in Loop: Header=BB1_1 Depth=1
+; CHECK-NEXT: cbz w19, .LBB1_1
; CHECK-NEXT: // %bb.3: // %bb8
-; CHECK-NEXT: // in Loop: Header=BB1_2 Depth=1
-; CHECK-NEXT: cbz w19, .LBB1_2
-; CHECK-NEXT: // %bb.4: // %bb8
-; CHECK-NEXT: // in Loop: Header=BB1_2 Depth=1
+; CHECK-NEXT: // in Loop: Header=BB1_1 Depth=1
; CHECK-NEXT: cmp w19, #10
-; CHECK-NEXT: b.ne .LBB1_2
-; CHECK-NEXT: // %bb.5: // %bb9
-; CHECK-NEXT: // in Loop: Header=BB1_2 Depth=1
+; CHECK-NEXT: b.ne .LBB1_1
+; CHECK-NEXT: // %bb.4: // %bb9
+; CHECK-NEXT: // in Loop: Header=BB1_1 Depth=1
; CHECK-NEXT: str wzr, [x23]
-; CHECK-NEXT: b .LBB1_2
-; CHECK-NEXT: .LBB1_6: // %bb8
-; CHECK-NEXT: // in Loop: Header=BB1_2 Depth=1
+; CHECK-NEXT: b .LBB1_1
+; CHECK-NEXT: .LBB1_5: // %bb8
+; CHECK-NEXT: // in Loop: Header=BB1_1 Depth=1
; CHECK-NEXT: cmp w19, #34
-; CHECK-NEXT: b.eq .LBB1_1
-; CHECK-NEXT: // %bb.7: // %bb8
-; CHECK-NEXT: // in Loop: Header=BB1_2 Depth=1
+; CHECK-NEXT: b.eq .LBB1_7
+; CHECK-NEXT: // %bb.6: // %bb8
+; CHECK-NEXT: // in Loop: Header=BB1_1 Depth=1
; CHECK-NEXT: cmp w19, #39
-; CHECK-NEXT: b.eq .LBB1_1
-; CHECK-NEXT: b .LBB1_2
+; CHECK-NEXT: b.ne .LBB1_1
+; CHECK-NEXT: .LBB1_7: // %bb10
+; CHECK-NEXT: // in Loop: Header=BB1_1 Depth=1
+; CHECK-NEXT: mov w0, w22
+; CHECK-NEXT: mov x1, x20
+; CHECK-NEXT: str wzr, [x21]
+; CHECK-NEXT: bl foo
+; CHECK-NEXT: b .LBB1_1
bb:
br label %bb7
diff --git a/llvm/test/CodeGen/X86/block-placement-loop-top-multi-succ.ll b/llvm/test/CodeGen/X86/block-placement-loop-top-multi-succ.ll
new file mode 100644
index 0000000000000..dfe2ffccc89ac
--- /dev/null
+++ b/llvm/test/CodeGen/X86/block-placement-loop-top-multi-succ.ll
@@ -0,0 +1,44 @@
+; RUN: llc -O2 -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s
+;
+; Do not rotate a single-successor increment block in front of the latch when
+; a 2-way predecessor already has the latch as its other successor:
+;
+; header --> check_nl --> latch --> header
+; \ | ^
+; \ v |
+; ----> inc --------
+;
+; Rotating inc to the loop top inverts the likely fallthrough of check_nl
+; (cmpb $10 / je to inc) and increases branch misses. See GH218248.
+
+; CHECK-LABEL: skip_sep:
+; CHECK: %check_nl
+; CHECK: cmpb $10, %cl
+; CHECK-NEXT: jne
+
+define ptr @skip_sep(ptr %p, ptr %end) {
+entry:
+ br label %header
+
+header:
+ %q = phi ptr [ %p, %entry ], [ %q.next, %latch ]
+ %c = load i8, ptr %q
+ %is_comma = icmp eq i8 %c, 44
+ br i1 %is_comma, label %inc, label %check_nl
+
+check_nl:
+ %is_nl = icmp eq i8 %c, 10
+ br i1 %is_nl, label %inc, label %latch
+
+inc:
+ %q.inc = getelementptr i8, ptr %q, i64 1
+ br label %latch
+
+latch:
+ %q.next = phi ptr [ %q, %check_nl ], [ %q.inc, %inc ]
+ %done = icmp uge ptr %q.next, %end
+ br i1 %done, label %exit, label %header
+
+exit:
+ ret ptr %q.next
+}
diff --git a/llvm/test/CodeGen/X86/fold-loop-of-urem.ll b/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
index f3b9af4eb08e8..a0519033981e1 100644
--- a/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
+++ b/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
@@ -289,10 +289,6 @@ define void @simple_urem_to_sel_nested2(i32 %N, i32 %rem_amt) nounwind {
; CHECK-NEXT: xorl %r12d, %r12d
; CHECK-NEXT: jmp .LBB4_2
; CHECK-NEXT: .p2align 4
-; CHECK-NEXT: .LBB4_5: # %for.body1
-; CHECK-NEXT: # in Loop: Header=BB4_2 Depth=1
-; CHECK-NEXT: movl %r14d, %edi
-; CHECK-NEXT: callq use.i32 at PLT
; CHECK-NEXT: .LBB4_6: # %for.body.tail
; CHECK-NEXT: # in Loop: Header=BB4_2 Depth=1
; CHECK-NEXT: incl %r14d
@@ -315,7 +311,11 @@ define void @simple_urem_to_sel_nested2(i32 %N, i32 %rem_amt) nounwind {
; CHECK-NEXT: # in Loop: Header=BB4_2 Depth=1
; CHECK-NEXT: callq get.i1 at PLT
; CHECK-NEXT: testb $1, %al
-; CHECK-NEXT: jne .LBB4_5
+; CHECK-NEXT: je .LBB4_6
+; CHECK-NEXT: .LBB4_5: # %for.body1
+; CHECK-NEXT: # in Loop: Header=BB4_2 Depth=1
+; CHECK-NEXT: movl %r14d, %edi
+; CHECK-NEXT: callq use.i32 at PLT
; CHECK-NEXT: jmp .LBB4_6
; CHECK-NEXT: .LBB4_7:
; CHECK-NEXT: popq %rbx
@@ -364,13 +364,6 @@ define void @simple_urem_fail_bad_incr3(i32 %N, i32 %rem_amt) nounwind {
; CHECK-NEXT: movl %esi, %ebx
; CHECK-NEXT: jmp .LBB5_2
; CHECK-NEXT: .p2align 4
-; CHECK-NEXT: .LBB5_6: # %for.body1
-; CHECK-NEXT: # in Loop: Header=BB5_2 Depth=1
-; CHECK-NEXT: movl %ebp, %eax
-; CHECK-NEXT: xorl %edx, %edx
-; CHECK-NEXT: divl %ebx
-; CHECK-NEXT: movl %edx, %edi
-; CHECK-NEXT: callq use.i32 at PLT
; CHECK-NEXT: .LBB5_7: # %for.body.tail
; CHECK-NEXT: # in Loop: Header=BB5_2 Depth=1
; CHECK-NEXT: callq get.i1 at PLT
@@ -398,7 +391,14 @@ define void @simple_urem_fail_bad_incr3(i32 %N, i32 %rem_amt) nounwind {
; CHECK-NEXT: xorl %ebp, %ebp
; CHECK-NEXT: callq get.i1 at PLT
; CHECK-NEXT: testb $1, %al
-; CHECK-NEXT: jne .LBB5_6
+; CHECK-NEXT: je .LBB5_7
+; CHECK-NEXT: .LBB5_6: # %for.body1
+; CHECK-NEXT: # in Loop: Header=BB5_2 Depth=1
+; CHECK-NEXT: movl %ebp, %eax
+; CHECK-NEXT: xorl %edx, %edx
+; CHECK-NEXT: divl %ebx
+; CHECK-NEXT: movl %edx, %edi
+; CHECK-NEXT: callq use.i32 at PLT
; CHECK-NEXT: jmp .LBB5_7
; CHECK-NEXT: .LBB5_8:
; CHECK-NEXT: popq %rbx
More information about the llvm-commits
mailing list