[llvm] [CodeGen] Don't rotate shared increment blocks to loop top (PR #219126)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 23 20:50:03 PDT 2026


https://github.com/compilersutra updated https://github.com/llvm/llvm-project/pull/219126

>From b447cb07b291cd689a565f1aff6c703706a50d0f Mon Sep 17 00:00:00 2001
From: compilersutra <osc at compilersutra.com>
Date: Thu, 27 Aug 2026 11:52:21 +0530
Subject: [PATCH] [CodeGen] Don't rotate shared increment blocks to loop top

canMoveBottomBlockToTop only looked at a single predecessor, so a
diamond like c == ',' || c == '\n' got rotated and inverted the
likely fallthrough.

Fixes #218248
---
 llvm/lib/CodeGen/MachineBlockPlacement.cpp    | 35 ++++++++++----
 llvm/test/CodeGen/AArch64/peephole-and-tst.ll | 22 ++++-----
 .../AArch64/regalloc-spill-weight-basic.ll    | 47 +++++++++----------
 .../block-placement-loop-top-multi-succ.ll    | 44 +++++++++++++++++
 llvm/test/CodeGen/X86/fold-loop-of-urem.ll    | 26 +++++-----
 5 files changed, 116 insertions(+), 58 deletions(-)
 create mode 100644 llvm/test/CodeGen/X86/block-placement-loop-top-multi-succ.ll

diff --git a/llvm/lib/CodeGen/MachineBlockPlacement.cpp b/llvm/lib/CodeGen/MachineBlockPlacement.cpp
index a748150d98c64..84215791515f6 100644
--- a/llvm/lib/CodeGen/MachineBlockPlacement.cpp
+++ b/llvm/lib/CodeGen/MachineBlockPlacement.cpp
@@ -2007,19 +2007,27 @@ void MachineBlockPlacement::buildChain(const MachineBasicBlock *HeadBB,
 // If BB is moved before OldTop, Pred needs a taken branch to BB, and it can't
 // layout the other successor below it, so it can't reduce taken branch.
 // In this case we keep its original layout.
+//
+// Pred may not be BB's only predecessor: several 2-way branches can share the
+// same BB (both sides of `c == ',' || c == '\n'`). Treat any such Pred the
+// same way.
 bool MachineBlockPlacement::canMoveBottomBlockToTop(
     const MachineBasicBlock *BottomBlock, const MachineBasicBlock *OldTop) {
-  if (BottomBlock->pred_size() != 1)
-    return true;
-  MachineBasicBlock *Pred = *BottomBlock->pred_begin();
-  if (Pred->succ_size() != 2)
-    return true;
+  // BB may have several predecessors that share this diamond (e.g. comma and
+  // newline both jump to the same increment). Reject the rotate if any of them
+  // has OldTop as the other successor.
+  for (const MachineBasicBlock *Pred : BottomBlock->predecessors()) {
+    if (Pred == OldTop)
+      continue;
+    if (Pred->succ_size() != 2)
+      continue;
 
-  MachineBasicBlock *OtherBB = *Pred->succ_begin();
-  if (OtherBB == BottomBlock)
-    OtherBB = *Pred->succ_rbegin();
-  if (OtherBB == OldTop)
-    return false;
+    const MachineBasicBlock *OtherBB = *Pred->succ_begin();
+    if (OtherBB == BottomBlock)
+      OtherBB = *Pred->succ_rbegin();
+    if (OtherBB == OldTop)
+      return false;
+  }
 
   return true;
 }
@@ -2228,6 +2236,13 @@ MachineBasicBlock *MachineBlockPlacement::findBestLoopTopHelper(
          *BestPred->pred_begin() != L.getHeader())
     BestPred = *BestPred->pred_begin();
 
+  // The walk-back can land on a block that was not the predecessor FallThrough
+  // gains considered. Re-check the diamond-into-top shape.
+  if (!canMoveBottomBlockToTop(BestPred, OldTop)) {
+    LLVM_DEBUG(dbgs() << "    final top unchanged\n");
+    return OldTop;
+  }
+
   LLVM_DEBUG(dbgs() << "    final top: " << getBlockName(BestPred) << "\n");
   return BestPred;
 }
diff --git a/llvm/test/CodeGen/AArch64/peephole-and-tst.ll b/llvm/test/CodeGen/AArch64/peephole-and-tst.ll
index 6449f5d5f07d3..628d0b4a59c64 100644
--- a/llvm/test/CodeGen/AArch64/peephole-and-tst.ll
+++ b/llvm/test/CodeGen/AArch64/peephole-and-tst.ll
@@ -39,26 +39,26 @@ define i32 @test_func_i32_two_uses(i32 %in, i32 %bit, i32 %mask) {
 ; CHECK-GI-NEXT:    ldr x8, [x8, :got_lo12:ptr_wrapper]
 ; CHECK-GI-NEXT:    ldr x9, [x8]
 ; CHECK-GI-NEXT:    mov w8, wzr
-; CHECK-GI-NEXT:    b .LBB0_3
-; CHECK-GI-NEXT:  .LBB0_1: // in Loop: Header=BB0_3 Depth=1
-; CHECK-GI-NEXT:    str xzr, [x9, #8]
-; CHECK-GI-NEXT:  .LBB0_2: // in Loop: Header=BB0_3 Depth=1
+; CHECK-GI-NEXT:    b .LBB0_2
+; CHECK-GI-NEXT:  .LBB0_1: // in Loop: Header=BB0_2 Depth=1
 ; CHECK-GI-NEXT:    lsl w1, w1, #1
 ; CHECK-GI-NEXT:    cbz w1, .LBB0_6
-; CHECK-GI-NEXT:  .LBB0_3: // %do.body
+; CHECK-GI-NEXT:  .LBB0_2: // %do.body
 ; CHECK-GI-NEXT:    // =>This Inner Loop Header: Depth=1
 ; CHECK-GI-NEXT:    and w10, w1, w0
 ; CHECK-GI-NEXT:    tst w1, w0
 ; CHECK-GI-NEXT:    and w11, w2, w0
 ; CHECK-GI-NEXT:    cinc w8, w8, ne
 ; CHECK-GI-NEXT:    cmp w10, w11
-; CHECK-GI-NEXT:    b.eq .LBB0_1
+; CHECK-GI-NEXT:    b.eq .LBB0_5
+; CHECK-GI-NEXT:  // %bb.3: // %do.body
+; CHECK-GI-NEXT:    // in Loop: Header=BB0_2 Depth=1
+; CHECK-GI-NEXT:    cbnz w2, .LBB0_5
 ; CHECK-GI-NEXT:  // %bb.4: // %do.body
-; CHECK-GI-NEXT:    // in Loop: Header=BB0_3 Depth=1
-; CHECK-GI-NEXT:    cbnz w2, .LBB0_1
-; CHECK-GI-NEXT:  // %bb.5: // %do.body
-; CHECK-GI-NEXT:    // in Loop: Header=BB0_3 Depth=1
-; CHECK-GI-NEXT:    cbz w10, .LBB0_2
+; CHECK-GI-NEXT:    // in Loop: Header=BB0_2 Depth=1
+; CHECK-GI-NEXT:    cbz w10, .LBB0_1
+; CHECK-GI-NEXT:  .LBB0_5: // in Loop: Header=BB0_2 Depth=1
+; CHECK-GI-NEXT:    str xzr, [x9, #8]
 ; CHECK-GI-NEXT:    b .LBB0_1
 ; CHECK-GI-NEXT:  .LBB0_6: // %do.end
 ; CHECK-GI-NEXT:    mov w0, w8
diff --git a/llvm/test/CodeGen/AArch64/regalloc-spill-weight-basic.ll b/llvm/test/CodeGen/AArch64/regalloc-spill-weight-basic.ll
index 5c3bd984087ec..697b00da5a0ce 100644
--- a/llvm/test/CodeGen/AArch64/regalloc-spill-weight-basic.ll
+++ b/llvm/test/CodeGen/AArch64/regalloc-spill-weight-basic.ll
@@ -102,37 +102,36 @@ define void @optspeed(i32 %arg, i32 %arg1, ptr %arg2, ptr %arg3, ptr %arg4, i32
 ; CHECK-NEXT:    mov x20, x3
 ; CHECK-NEXT:    mov x23, x2
 ; CHECK-NEXT:    mov w19, w1
-; CHECK-NEXT:    b .LBB1_2
-; CHECK-NEXT:  .LBB1_1: // %bb10
-; CHECK-NEXT:    // in Loop: Header=BB1_2 Depth=1
-; CHECK-NEXT:    mov w0, w22
-; CHECK-NEXT:    mov x1, x20
-; CHECK-NEXT:    str wzr, [x21]
-; CHECK-NEXT:    bl foo
-; CHECK-NEXT:  .LBB1_2: // %bb8
+; CHECK-NEXT:  .LBB1_1: // %bb8
 ; CHECK-NEXT:    // =>This Inner Loop Header: Depth=1
 ; CHECK-NEXT:    cmp w19, #33
-; CHECK-NEXT:    b.gt .LBB1_6
+; CHECK-NEXT:    b.gt .LBB1_5
+; CHECK-NEXT:  // %bb.2: // %bb8
+; CHECK-NEXT:    // in Loop: Header=BB1_1 Depth=1
+; CHECK-NEXT:    cbz w19, .LBB1_1
 ; CHECK-NEXT:  // %bb.3: // %bb8
-; CHECK-NEXT:    // in Loop: Header=BB1_2 Depth=1
-; CHECK-NEXT:    cbz w19, .LBB1_2
-; CHECK-NEXT:  // %bb.4: // %bb8
-; CHECK-NEXT:    // in Loop: Header=BB1_2 Depth=1
+; CHECK-NEXT:    // in Loop: Header=BB1_1 Depth=1
 ; CHECK-NEXT:    cmp w19, #10
-; CHECK-NEXT:    b.ne .LBB1_2
-; CHECK-NEXT:  // %bb.5: // %bb9
-; CHECK-NEXT:    // in Loop: Header=BB1_2 Depth=1
+; CHECK-NEXT:    b.ne .LBB1_1
+; CHECK-NEXT:  // %bb.4: // %bb9
+; CHECK-NEXT:    // in Loop: Header=BB1_1 Depth=1
 ; CHECK-NEXT:    str wzr, [x23]
-; CHECK-NEXT:    b .LBB1_2
-; CHECK-NEXT:  .LBB1_6: // %bb8
-; CHECK-NEXT:    // in Loop: Header=BB1_2 Depth=1
+; CHECK-NEXT:    b .LBB1_1
+; CHECK-NEXT:  .LBB1_5: // %bb8
+; CHECK-NEXT:    // in Loop: Header=BB1_1 Depth=1
 ; CHECK-NEXT:    cmp w19, #34
-; CHECK-NEXT:    b.eq .LBB1_1
-; CHECK-NEXT:  // %bb.7: // %bb8
-; CHECK-NEXT:    // in Loop: Header=BB1_2 Depth=1
+; CHECK-NEXT:    b.eq .LBB1_7
+; CHECK-NEXT:  // %bb.6: // %bb8
+; CHECK-NEXT:    // in Loop: Header=BB1_1 Depth=1
 ; CHECK-NEXT:    cmp w19, #39
-; CHECK-NEXT:    b.eq .LBB1_1
-; CHECK-NEXT:    b .LBB1_2
+; CHECK-NEXT:    b.ne .LBB1_1
+; CHECK-NEXT:  .LBB1_7: // %bb10
+; CHECK-NEXT:    // in Loop: Header=BB1_1 Depth=1
+; CHECK-NEXT:    mov w0, w22
+; CHECK-NEXT:    mov x1, x20
+; CHECK-NEXT:    str wzr, [x21]
+; CHECK-NEXT:    bl foo
+; CHECK-NEXT:    b .LBB1_1
 bb:
   br label %bb7
 
diff --git a/llvm/test/CodeGen/X86/block-placement-loop-top-multi-succ.ll b/llvm/test/CodeGen/X86/block-placement-loop-top-multi-succ.ll
new file mode 100644
index 0000000000000..dfe2ffccc89ac
--- /dev/null
+++ b/llvm/test/CodeGen/X86/block-placement-loop-top-multi-succ.ll
@@ -0,0 +1,44 @@
+; RUN: llc -O2 -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s
+;
+; Do not rotate a single-successor increment block in front of the latch when
+; a 2-way predecessor already has the latch as its other successor:
+;
+;   header --> check_nl --> latch --> header
+;        \        |          ^
+;         \       v          |
+;          ----> inc --------
+;
+; Rotating inc to the loop top inverts the likely fallthrough of check_nl
+; (cmpb $10 / je to inc) and increases branch misses. See GH218248.
+
+; CHECK-LABEL: skip_sep:
+; CHECK:       %check_nl
+; CHECK:       cmpb $10, %cl
+; CHECK-NEXT:  jne
+
+define ptr @skip_sep(ptr %p, ptr %end) {
+entry:
+  br label %header
+
+header:
+  %q = phi ptr [ %p, %entry ], [ %q.next, %latch ]
+  %c = load i8, ptr %q
+  %is_comma = icmp eq i8 %c, 44
+  br i1 %is_comma, label %inc, label %check_nl
+
+check_nl:
+  %is_nl = icmp eq i8 %c, 10
+  br i1 %is_nl, label %inc, label %latch
+
+inc:
+  %q.inc = getelementptr i8, ptr %q, i64 1
+  br label %latch
+
+latch:
+  %q.next = phi ptr [ %q, %check_nl ], [ %q.inc, %inc ]
+  %done = icmp uge ptr %q.next, %end
+  br i1 %done, label %exit, label %header
+
+exit:
+  ret ptr %q.next
+}
diff --git a/llvm/test/CodeGen/X86/fold-loop-of-urem.ll b/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
index f3b9af4eb08e8..a0519033981e1 100644
--- a/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
+++ b/llvm/test/CodeGen/X86/fold-loop-of-urem.ll
@@ -289,10 +289,6 @@ define void @simple_urem_to_sel_nested2(i32 %N, i32 %rem_amt) nounwind {
 ; CHECK-NEXT:    xorl %r12d, %r12d
 ; CHECK-NEXT:    jmp .LBB4_2
 ; CHECK-NEXT:    .p2align 4
-; CHECK-NEXT:  .LBB4_5: # %for.body1
-; CHECK-NEXT:    # in Loop: Header=BB4_2 Depth=1
-; CHECK-NEXT:    movl %r14d, %edi
-; CHECK-NEXT:    callq use.i32 at PLT
 ; CHECK-NEXT:  .LBB4_6: # %for.body.tail
 ; CHECK-NEXT:    # in Loop: Header=BB4_2 Depth=1
 ; CHECK-NEXT:    incl %r14d
@@ -315,7 +311,11 @@ define void @simple_urem_to_sel_nested2(i32 %N, i32 %rem_amt) nounwind {
 ; CHECK-NEXT:    # in Loop: Header=BB4_2 Depth=1
 ; CHECK-NEXT:    callq get.i1 at PLT
 ; CHECK-NEXT:    testb $1, %al
-; CHECK-NEXT:    jne .LBB4_5
+; CHECK-NEXT:    je .LBB4_6
+; CHECK-NEXT:  .LBB4_5: # %for.body1
+; CHECK-NEXT:    # in Loop: Header=BB4_2 Depth=1
+; CHECK-NEXT:    movl %r14d, %edi
+; CHECK-NEXT:    callq use.i32 at PLT
 ; CHECK-NEXT:    jmp .LBB4_6
 ; CHECK-NEXT:  .LBB4_7:
 ; CHECK-NEXT:    popq %rbx
@@ -364,13 +364,6 @@ define void @simple_urem_fail_bad_incr3(i32 %N, i32 %rem_amt) nounwind {
 ; CHECK-NEXT:    movl %esi, %ebx
 ; CHECK-NEXT:    jmp .LBB5_2
 ; CHECK-NEXT:    .p2align 4
-; CHECK-NEXT:  .LBB5_6: # %for.body1
-; CHECK-NEXT:    # in Loop: Header=BB5_2 Depth=1
-; CHECK-NEXT:    movl %ebp, %eax
-; CHECK-NEXT:    xorl %edx, %edx
-; CHECK-NEXT:    divl %ebx
-; CHECK-NEXT:    movl %edx, %edi
-; CHECK-NEXT:    callq use.i32 at PLT
 ; CHECK-NEXT:  .LBB5_7: # %for.body.tail
 ; CHECK-NEXT:    # in Loop: Header=BB5_2 Depth=1
 ; CHECK-NEXT:    callq get.i1 at PLT
@@ -398,7 +391,14 @@ define void @simple_urem_fail_bad_incr3(i32 %N, i32 %rem_amt) nounwind {
 ; CHECK-NEXT:    xorl %ebp, %ebp
 ; CHECK-NEXT:    callq get.i1 at PLT
 ; CHECK-NEXT:    testb $1, %al
-; CHECK-NEXT:    jne .LBB5_6
+; CHECK-NEXT:    je .LBB5_7
+; CHECK-NEXT:  .LBB5_6: # %for.body1
+; CHECK-NEXT:    # in Loop: Header=BB5_2 Depth=1
+; CHECK-NEXT:    movl %ebp, %eax
+; CHECK-NEXT:    xorl %edx, %edx
+; CHECK-NEXT:    divl %ebx
+; CHECK-NEXT:    movl %edx, %edi
+; CHECK-NEXT:    callq use.i32 at PLT
 ; CHECK-NEXT:    jmp .LBB5_7
 ; CHECK-NEXT:  .LBB5_8:
 ; CHECK-NEXT:    popq %rbx



More information about the llvm-commits mailing list