[llvm] [AArch64] Don't merge branch conditions that both compare memory loads (PR #206504)

Kunal Pathak via llvm-commits llvm-commits at lists.llvm.org
Mon Jun 29 07:57:25 PDT 2026


https://github.com/kunalspathak created https://github.com/llvm/llvm-project/pull/206504

`shouldKeepJumpConditionsTogether` decides whether to fold two integer branch conditions into a CMP/CCMP chain by pricing the RHS dependency-chain latency. That ignores register pressure: when both conditions compare loaded values, merging pins all the loaded operands live at once to feed the chain instead of consuming them at each split compare-and-branch. On a load-store target that extends their live ranges across the region the branch dominates.

In SPEC CPU2026 709.cactus_r, the interior-point bounds guard ahead of the ML_CCZ4 stencil kernels merges this way and the live-range extension cascades into ~900 extra spills in the kernel body, regressing ~8% on Neoverse V2 at -O3 (branch misses unchanged).

Decline to merge when both sides are integer compares of loaded values, mirroring the machine CCMP pass (which won't speculate loads).

>From f0f1d583ab174dfea2a3cfd8f6c967cc15b3637f Mon Sep 17 00:00:00 2001
From: Kunal Pathak <kupathak at meta.com>
Date: Mon, 29 Jun 2026 07:46:58 -0700
Subject: [PATCH] [AArch64] Don't merge branch conditions that both compare
 memory loads

shouldKeepJumpConditionsTogether decides whether to fold two integer branch
conditions into a CMP/CCMP chain by pricing the RHS dependency-chain latency.
That ignores register pressure: when both conditions compare loaded values,
merging pins all the loaded operands live at once to feed the chain instead
of consuming them at each split compare-and-branch. On a load-store target
that extends their live ranges across the region the branch dominates.

In SPEC CPU2026 709.cactus_r, the interior-point bounds guard ahead of the
ML_CCZ4 stencil kernels merges this way and the live-range extension cascades
into ~900 extra spills in the kernel body, regressing ~8% on Neoverse V2 at
-O3 (branch misses unchanged).

Decline to merge when both sides are integer compares of loaded values,
mirroring the machine CCMP pass (which won't speculate loads) and the
existing FP/CBZ-TBNZ carve-outs.
---
 .../Target/AArch64/AArch64ISelLowering.cpp    |  25 +++++
 .../br-cond-merging-loaded-operands.ll        |  99 ++++++++++++++++
 llvm/test/CodeGen/AArch64/ragreedy-csr.ll     | 106 +++++++++---------
 3 files changed, 179 insertions(+), 51 deletions(-)
 create mode 100644 llvm/test/CodeGen/AArch64/br-cond-merging-loaded-operands.ll

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index e883c8bb5e96e..a93a5a2e7dee4 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -32130,6 +32130,31 @@ AArch64TargetLowering::getJumpConditionMergingParams(Instruction::BinaryOps Opc,
     return false;
   };
 
+  // Returns true if \p V is an integer comparison whose operands trace back to
+  // a memory load. Merging such a condition forces the loaded value to be held
+  // in a register up to the single merged branch (AArch64 is load-store, so it
+  // cannot fold the load into the compare the way x86 can). When the branch
+  // dominates a large region -- e.g. a Cactus/Kranc interior-point bounds guard
+  // (imin[d] < imax[d]) that dominates a 10k+ instruction stencil kernel -- the
+  // extended live ranges cascade into stack spills throughout the body. The
+  // latency-based cost model below cannot see register pressure, so carve it
+  // out.
+  auto ComparesLoadedValue = [](const Value *V) {
+    const auto *Cmp = dyn_cast<ICmpInst>(V);
+    if (!Cmp)
+      return false;
+    for (const Value *Op : {Cmp->getOperand(0), Cmp->getOperand(1)}) {
+      const Value *Stripped = Op;
+      while (const auto *Cast = dyn_cast<CastInst>(Stripped))
+        Stripped = Cast->getOperand(0);
+      if (isa<LoadInst>(Stripped))
+        return true;
+    }
+    return false;
+  };
+  if (ComparesLoadedValue(Lhs) && ComparesLoadedValue(Rhs))
+    return {-1, -1, -1};
+
   int BaseCost = BrMergingBaseCostThresh.getValue();
   // CCMP folds the second compare and the branch into a single cheap op, so
   // merging is worth tolerating extra speculated work on the RHS dependency
diff --git a/llvm/test/CodeGen/AArch64/br-cond-merging-loaded-operands.ll b/llvm/test/CodeGen/AArch64/br-cond-merging-loaded-operands.ll
new file mode 100644
index 0000000000000..e9440367fa4e0
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/br-cond-merging-loaded-operands.ll
@@ -0,0 +1,99 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=aarch64-linux-gnu | FileCheck %s
+
+; getJumpConditionMergingParams keeps two integer branch conditions split when
+; both compare values loaded from memory: merging would pin the loaded operands
+; in registers for the CMP/CCMP chain, which the latency-based heuristic prices
+; as cheap but can cause spills under register pressure. Other conditions still
+; merge into a CCMP chain.
+
+declare void @sink_a()
+declare void @sink_b()
+
+; Both sides compare loads (e.g. a bounds guard imin[d] < imax[d]): keep split.
+define void @both_loaded(ptr %imin, ptr %imax) {
+; CHECK-LABEL: both_loaded:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    ldr w8, [x0]
+; CHECK-NEXT:    ldr w9, [x1]
+; CHECK-NEXT:    cmp w8, w9
+; CHECK-NEXT:    b.ge .LBB0_3
+; CHECK-NEXT:  // %bb.1: // %entry
+; CHECK-NEXT:    ldr w8, [x0, #4]
+; CHECK-NEXT:    ldr w9, [x1, #4]
+; CHECK-NEXT:    cmp w8, w9
+; CHECK-NEXT:    b.ge .LBB0_3
+; CHECK-NEXT:  // %bb.2: // %taken
+; CHECK-NEXT:    b sink_a
+; CHECK-NEXT:  .LBB0_3: // %fall
+; CHECK-NEXT:    b sink_b
+entry:
+  %imin0 = load i32, ptr %imin
+  %imax0 = load i32, ptr %imax
+  %c0 = icmp slt i32 %imin0, %imax0
+  %imin1p = getelementptr inbounds i32, ptr %imin, i64 1
+  %imax1p = getelementptr inbounds i32, ptr %imax, i64 1
+  %imin1 = load i32, ptr %imin1p
+  %imax1 = load i32, ptr %imax1p
+  %c1 = icmp slt i32 %imin1, %imax1
+  %and = and i1 %c0, %c1
+  br i1 %and, label %taken, label %fall
+taken:
+  tail call void @sink_a()
+  ret void
+fall:
+  tail call void @sink_b()
+  ret void
+}
+
+; Negative control: register operands, not loads -> still merge into a CCMP chain.
+define void @both_reg(i32 %a0, i32 %b0, i32 %a1, i32 %b1) {
+; CHECK-LABEL: both_reg:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    cmp w2, w3
+; CHECK-NEXT:    ccmp w0, w1, #0, lt
+; CHECK-NEXT:    b.ge .LBB1_2
+; CHECK-NEXT:  // %bb.1: // %taken
+; CHECK-NEXT:    b sink_a
+; CHECK-NEXT:  .LBB1_2: // %fall
+; CHECK-NEXT:    b sink_b
+entry:
+  %c0 = icmp slt i32 %a0, %b0
+  %c1 = icmp slt i32 %a1, %b1
+  %and = and i1 %c0, %c1
+  br i1 %and, label %taken, label %fall
+taken:
+  tail call void @sink_a()
+  ret void
+fall:
+  tail call void @sink_b()
+  ret void
+}
+
+; Only one side is a load -> "both" requirement not met -> still merges.
+define void @one_loaded(ptr %imin, ptr %imax, i32 %a1, i32 %b1) {
+; CHECK-LABEL: one_loaded:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    ldr w8, [x1]
+; CHECK-NEXT:    ldr w9, [x0]
+; CHECK-NEXT:    cmp w2, w3
+; CHECK-NEXT:    ccmp w9, w8, #0, lt
+; CHECK-NEXT:    b.ge .LBB2_2
+; CHECK-NEXT:  // %bb.1: // %taken
+; CHECK-NEXT:    b sink_a
+; CHECK-NEXT:  .LBB2_2: // %fall
+; CHECK-NEXT:    b sink_b
+entry:
+  %imin0 = load i32, ptr %imin
+  %imax0 = load i32, ptr %imax
+  %c0 = icmp slt i32 %imin0, %imax0
+  %c1 = icmp slt i32 %a1, %b1
+  %and = and i1 %c0, %c1
+  br i1 %and, label %taken, label %fall
+taken:
+  tail call void @sink_a()
+  ret void
+fall:
+  tail call void @sink_b()
+  ret void
+}
diff --git a/llvm/test/CodeGen/AArch64/ragreedy-csr.ll b/llvm/test/CodeGen/AArch64/ragreedy-csr.ll
index 4ac7d0f1c34e6..249a08118386f 100644
--- a/llvm/test/CodeGen/AArch64/ragreedy-csr.ll
+++ b/llvm/test/CodeGen/AArch64/ragreedy-csr.ll
@@ -24,7 +24,7 @@ define fastcc i32 @prune_match(ptr nocapture readonly %a, ptr nocapture readonly
 ; CHECK-NEXT:    ldrh w8, [x0]
 ; CHECK-NEXT:    ldrh w9, [x1]
 ; CHECK-NEXT:    cmp w8, w9
-; CHECK-NEXT:    b.ne LBB0_42
+; CHECK-NEXT:    b.ne LBB0_44
 ; CHECK-NEXT:  ; %bb.1: ; %if.end
 ; CHECK-NEXT:    sub sp, sp, #64
 ; CHECK-NEXT:    stp x29, x30, [sp, #48] ; 16-byte Folded Spill
@@ -80,7 +80,7 @@ define fastcc i32 @prune_match(ptr nocapture readonly %a, ptr nocapture readonly
 ; CHECK-NEXT:    ldrb w8, [x9, x11]
 ; CHECK-NEXT:    ldrb w15, [x10, x11]
 ; CHECK-NEXT:    cmp w8, w15
-; CHECK-NEXT:    b.ne LBB0_25
+; CHECK-NEXT:    b.ne LBB0_32
 ; CHECK-NEXT:  ; %bb.7: ; %if.end17
 ; CHECK-NEXT:    add x11, x11, #1
 ; CHECK-NEXT:    ldrsb x8, [x9, x11]
@@ -138,12 +138,12 @@ define fastcc i32 @prune_match(ptr nocapture readonly %a, ptr nocapture readonly
 ; CHECK-NEXT:    ; in Loop: Header=BB0_13 Depth=1
 ; CHECK-NEXT:    cmp w8, #94
 ; CHECK-NEXT:    ccmp w8, w11, #0, ne
-; CHECK-NEXT:    b.ne LBB0_25
+; CHECK-NEXT:    b.ne LBB0_32
 ; CHECK-NEXT:  LBB0_17: ; %if.then83
 ; CHECK-NEXT:    ; in Loop: Header=BB0_13 Depth=1
 ; CHECK-NEXT:    ldrb w8, [x10], #1
 ; CHECK-NEXT:    cbnz w8, LBB0_13
-; CHECK-NEXT:    b LBB0_26
+; CHECK-NEXT:    b LBB0_33
 ; CHECK-NEXT:  LBB0_18: ; %land.lhs.true28
 ; CHECK-NEXT:    cbz w8, LBB0_22
 ; CHECK-NEXT:  ; %bb.19: ; %land.lhs.true28
@@ -157,95 +157,99 @@ define fastcc i32 @prune_match(ptr nocapture readonly %a, ptr nocapture readonly
 ; CHECK-NEXT:    sub x12, x9, x12
 ; CHECK-NEXT:    add x12, x12, x11
 ; CHECK-NEXT:    cmp x12, #1
-; CHECK-NEXT:    b.ne LBB0_39
+; CHECK-NEXT:    b.ne LBB0_41
 ; CHECK-NEXT:  LBB0_22:
 ; CHECK-NEXT:    mov w0, #1 ; =0x1
-; CHECK-NEXT:    b LBB0_26
+; CHECK-NEXT:    b LBB0_33
 ; CHECK-NEXT:  LBB0_23: ; %if.else88
+; CHECK-NEXT:    cmp w12, #1
+; CHECK-NEXT:    b.ne LBB0_34
+; CHECK-NEXT:  ; %bb.24: ; %if.else88
 ; CHECK-NEXT:    cmp w13, #2
-; CHECK-NEXT:    ccmp w12, #1, #0, eq
-; CHECK-NEXT:    b.eq LBB0_27
-; CHECK-NEXT:  ; %bb.24: ; %if.else123
-; CHECK-NEXT:    cmp w12, #2
-; CHECK-NEXT:    ccmp w13, #1, #0, eq
-; CHECK-NEXT:    b.eq LBB0_34
-; CHECK-NEXT:  LBB0_25:
-; CHECK-NEXT:    mov w0, wzr
-; CHECK-NEXT:  LBB0_26:
-; CHECK-NEXT:    ldp x29, x30, [sp, #48] ; 16-byte Folded Reload
-; CHECK-NEXT:    add sp, sp, #64
-; CHECK-NEXT:    ret
-; CHECK-NEXT:  LBB0_27: ; %while.cond95.preheader
+; CHECK-NEXT:    b.ne LBB0_34
+; CHECK-NEXT:  ; %bb.25: ; %while.cond95.preheader
 ; CHECK-NEXT:    ldrb w12, [x9, x11]
 ; CHECK-NEXT:    cbz w12, LBB0_22
-; CHECK-NEXT:  ; %bb.28: ; %land.rhs99.preheader
+; CHECK-NEXT:  ; %bb.26: ; %land.rhs99.preheader
 ; CHECK-NEXT:    mov x8, xzr
 ; CHECK-NEXT:    mov w0, #1 ; =0x1
-; CHECK-NEXT:    b LBB0_30
-; CHECK-NEXT:  LBB0_29: ; %if.then117
-; CHECK-NEXT:    ; in Loop: Header=BB0_30 Depth=1
+; CHECK-NEXT:    b LBB0_28
+; CHECK-NEXT:  LBB0_27: ; %if.then117
+; CHECK-NEXT:    ; in Loop: Header=BB0_28 Depth=1
 ; CHECK-NEXT:    add x12, x9, x8
 ; CHECK-NEXT:    add x8, x8, #1
 ; CHECK-NEXT:    add x12, x12, x11
 ; CHECK-NEXT:    ldrb w12, [x12, #1]
-; CHECK-NEXT:    cbz w12, LBB0_26
-; CHECK-NEXT:  LBB0_30: ; %land.rhs99
+; CHECK-NEXT:    cbz w12, LBB0_33
+; CHECK-NEXT:  LBB0_28: ; %land.rhs99
 ; CHECK-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; CHECK-NEXT:    add x13, x10, x8
 ; CHECK-NEXT:    ldrb w13, [x13, x11]
 ; CHECK-NEXT:    cbz w13, LBB0_22
-; CHECK-NEXT:  ; %bb.31: ; %while.body104
-; CHECK-NEXT:    ; in Loop: Header=BB0_30 Depth=1
+; CHECK-NEXT:  ; %bb.29: ; %while.body104
+; CHECK-NEXT:    ; in Loop: Header=BB0_28 Depth=1
 ; CHECK-NEXT:    cmp w12, w13
-; CHECK-NEXT:    b.eq LBB0_29
-; CHECK-NEXT:  ; %bb.32: ; %while.body104
-; CHECK-NEXT:    ; in Loop: Header=BB0_30 Depth=1
+; CHECK-NEXT:    b.eq LBB0_27
+; CHECK-NEXT:  ; %bb.30: ; %while.body104
+; CHECK-NEXT:    ; in Loop: Header=BB0_28 Depth=1
 ; CHECK-NEXT:    cmp w12, #42
-; CHECK-NEXT:    b.eq LBB0_29
-; CHECK-NEXT:  ; %bb.33: ; %while.body104
-; CHECK-NEXT:    ; in Loop: Header=BB0_30 Depth=1
+; CHECK-NEXT:    b.eq LBB0_27
+; CHECK-NEXT:  ; %bb.31: ; %while.body104
+; CHECK-NEXT:    ; in Loop: Header=BB0_28 Depth=1
 ; CHECK-NEXT:    cmp w13, #94
-; CHECK-NEXT:    b.eq LBB0_29
-; CHECK-NEXT:    b LBB0_25
-; CHECK-NEXT:  LBB0_34: ; %while.cond130.preheader
+; CHECK-NEXT:    b.eq LBB0_27
+; CHECK-NEXT:  LBB0_32:
+; CHECK-NEXT:    mov w0, wzr
+; CHECK-NEXT:  LBB0_33:
+; CHECK-NEXT:    ldp x29, x30, [sp, #48] ; 16-byte Folded Reload
+; CHECK-NEXT:    add sp, sp, #64
+; CHECK-NEXT:    ret
+; CHECK-NEXT:  LBB0_34: ; %if.else123
+; CHECK-NEXT:    cmp w13, #1
+; CHECK-NEXT:    mov w0, wzr
+; CHECK-NEXT:    b.ne LBB0_33
+; CHECK-NEXT:  ; %bb.35: ; %if.else123
+; CHECK-NEXT:    cmp w12, #2
+; CHECK-NEXT:    b.ne LBB0_33
+; CHECK-NEXT:  ; %bb.36: ; %while.cond130.preheader
 ; CHECK-NEXT:    ldrb w12, [x9, x11]
 ; CHECK-NEXT:    cbz w12, LBB0_22
-; CHECK-NEXT:  ; %bb.35: ; %land.rhs134.preheader
+; CHECK-NEXT:  ; %bb.37: ; %land.rhs134.preheader
 ; CHECK-NEXT:    mov x8, xzr
 ; CHECK-NEXT:    mov w13, #42 ; =0x2a
 ; CHECK-NEXT:    mov w0, #1 ; =0x1
-; CHECK-NEXT:  LBB0_36: ; %land.rhs134
+; CHECK-NEXT:  LBB0_38: ; %land.rhs134
 ; CHECK-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; CHECK-NEXT:    add x14, x10, x8
 ; CHECK-NEXT:    ldrb w14, [x14, x11]
 ; CHECK-NEXT:    cbz w14, LBB0_22
-; CHECK-NEXT:  ; %bb.37: ; %while.body139
-; CHECK-NEXT:    ; in Loop: Header=BB0_36 Depth=1
+; CHECK-NEXT:  ; %bb.39: ; %while.body139
+; CHECK-NEXT:    ; in Loop: Header=BB0_38 Depth=1
 ; CHECK-NEXT:    cmp w12, #94
 ; CHECK-NEXT:    ccmp w14, w13, #4, ne
 ; CHECK-NEXT:    ccmp w12, w14, #4, ne
-; CHECK-NEXT:    b.ne LBB0_25
-; CHECK-NEXT:  ; %bb.38: ; %if.then152
-; CHECK-NEXT:    ; in Loop: Header=BB0_36 Depth=1
+; CHECK-NEXT:    b.ne LBB0_32
+; CHECK-NEXT:  ; %bb.40: ; %if.then152
+; CHECK-NEXT:    ; in Loop: Header=BB0_38 Depth=1
 ; CHECK-NEXT:    add x12, x9, x8
 ; CHECK-NEXT:    add x8, x8, #1
 ; CHECK-NEXT:    add x12, x12, x11
 ; CHECK-NEXT:    ldrb w12, [x12, #1]
-; CHECK-NEXT:    cbnz w12, LBB0_36
-; CHECK-NEXT:    b LBB0_26
-; CHECK-NEXT:  LBB0_39: ; %lor.lhs.false47
+; CHECK-NEXT:    cbnz w12, LBB0_38
+; CHECK-NEXT:    b LBB0_33
+; CHECK-NEXT:  LBB0_41: ; %lor.lhs.false47
 ; CHECK-NEXT:    cmp x12, #2
 ; CHECK-NEXT:    b.ne LBB0_11
-; CHECK-NEXT:  ; %bb.40: ; %land.lhs.true52
+; CHECK-NEXT:  ; %bb.42: ; %land.lhs.true52
 ; CHECK-NEXT:    add x12, x9, x11
 ; CHECK-NEXT:    mov w0, #1 ; =0x1
 ; CHECK-NEXT:    ldurb w12, [x12, #-1]
 ; CHECK-NEXT:    cmp w12, #73
-; CHECK-NEXT:    b.eq LBB0_26
-; CHECK-NEXT:  ; %bb.41: ; %land.lhs.true52
-; CHECK-NEXT:    cbz w8, LBB0_26
+; CHECK-NEXT:    b.eq LBB0_33
+; CHECK-NEXT:  ; %bb.43: ; %land.lhs.true52
+; CHECK-NEXT:    cbz w8, LBB0_33
 ; CHECK-NEXT:    b LBB0_12
-; CHECK-NEXT:  LBB0_42:
+; CHECK-NEXT:  LBB0_44:
 ; CHECK-NEXT:    mov w0, wzr
 ; CHECK-NEXT:    ret
 ; CHECK-NEXT:    .loh AdrpLdrGot Lloh0, Lloh1



More information about the llvm-commits mailing list