[llvm] [AArch64] Don't merge branch conditions that both compare memory loads (PR #206504)
Kunal Pathak via llvm-commits
llvm-commits at lists.llvm.org
Mon Jun 29 07:57:25 PDT 2026
https://github.com/kunalspathak created https://github.com/llvm/llvm-project/pull/206504
`shouldKeepJumpConditionsTogether` decides whether to fold two integer branch conditions into a CMP/CCMP chain by pricing the RHS dependency-chain latency. That ignores register pressure: when both conditions compare loaded values, merging pins all the loaded operands live at once to feed the chain instead of consuming them at each split compare-and-branch. On a load-store target that extends their live ranges across the region the branch dominates.
In SPEC CPU2026 709.cactus_r, the interior-point bounds guard ahead of the ML_CCZ4 stencil kernels merges this way and the live-range extension cascades into ~900 extra spills in the kernel body, regressing ~8% on Neoverse V2 at -O3 (branch misses unchanged).
Decline to merge when both sides are integer compares of loaded values, mirroring the machine CCMP pass (which won't speculate loads).
>From f0f1d583ab174dfea2a3cfd8f6c967cc15b3637f Mon Sep 17 00:00:00 2001
From: Kunal Pathak <kupathak at meta.com>
Date: Mon, 29 Jun 2026 07:46:58 -0700
Subject: [PATCH] [AArch64] Don't merge branch conditions that both compare
memory loads
shouldKeepJumpConditionsTogether decides whether to fold two integer branch
conditions into a CMP/CCMP chain by pricing the RHS dependency-chain latency.
That ignores register pressure: when both conditions compare loaded values,
merging pins all the loaded operands live at once to feed the chain instead
of consuming them at each split compare-and-branch. On a load-store target
that extends their live ranges across the region the branch dominates.
In SPEC CPU2026 709.cactus_r, the interior-point bounds guard ahead of the
ML_CCZ4 stencil kernels merges this way and the live-range extension cascades
into ~900 extra spills in the kernel body, regressing ~8% on Neoverse V2 at
-O3 (branch misses unchanged).
Decline to merge when both sides are integer compares of loaded values,
mirroring the machine CCMP pass (which won't speculate loads) and the
existing FP/CBZ-TBNZ carve-outs.
---
.../Target/AArch64/AArch64ISelLowering.cpp | 25 +++++
.../br-cond-merging-loaded-operands.ll | 99 ++++++++++++++++
llvm/test/CodeGen/AArch64/ragreedy-csr.ll | 106 +++++++++---------
3 files changed, 179 insertions(+), 51 deletions(-)
create mode 100644 llvm/test/CodeGen/AArch64/br-cond-merging-loaded-operands.ll
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index e883c8bb5e96e..a93a5a2e7dee4 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -32130,6 +32130,31 @@ AArch64TargetLowering::getJumpConditionMergingParams(Instruction::BinaryOps Opc,
return false;
};
+ // Returns true if \p V is an integer comparison whose operands trace back to
+ // a memory load. Merging such a condition forces the loaded value to be held
+ // in a register up to the single merged branch (AArch64 is load-store, so it
+ // cannot fold the load into the compare the way x86 can). When the branch
+ // dominates a large region -- e.g. a Cactus/Kranc interior-point bounds guard
+ // (imin[d] < imax[d]) that dominates a 10k+ instruction stencil kernel -- the
+ // extended live ranges cascade into stack spills throughout the body. The
+ // latency-based cost model below cannot see register pressure, so carve it
+ // out.
+ auto ComparesLoadedValue = [](const Value *V) {
+ const auto *Cmp = dyn_cast<ICmpInst>(V);
+ if (!Cmp)
+ return false;
+ for (const Value *Op : {Cmp->getOperand(0), Cmp->getOperand(1)}) {
+ const Value *Stripped = Op;
+ while (const auto *Cast = dyn_cast<CastInst>(Stripped))
+ Stripped = Cast->getOperand(0);
+ if (isa<LoadInst>(Stripped))
+ return true;
+ }
+ return false;
+ };
+ if (ComparesLoadedValue(Lhs) && ComparesLoadedValue(Rhs))
+ return {-1, -1, -1};
+
int BaseCost = BrMergingBaseCostThresh.getValue();
// CCMP folds the second compare and the branch into a single cheap op, so
// merging is worth tolerating extra speculated work on the RHS dependency
diff --git a/llvm/test/CodeGen/AArch64/br-cond-merging-loaded-operands.ll b/llvm/test/CodeGen/AArch64/br-cond-merging-loaded-operands.ll
new file mode 100644
index 0000000000000..e9440367fa4e0
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/br-cond-merging-loaded-operands.ll
@@ -0,0 +1,99 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=aarch64-linux-gnu | FileCheck %s
+
+; getJumpConditionMergingParams keeps two integer branch conditions split when
+; both compare values loaded from memory: merging would pin the loaded operands
+; in registers for the CMP/CCMP chain, which the latency-based heuristic prices
+; as cheap but can cause spills under register pressure. Other conditions still
+; merge into a CCMP chain.
+
+declare void @sink_a()
+declare void @sink_b()
+
+; Both sides compare loads (e.g. a bounds guard imin[d] < imax[d]): keep split.
+define void @both_loaded(ptr %imin, ptr %imax) {
+; CHECK-LABEL: both_loaded:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: ldr w8, [x0]
+; CHECK-NEXT: ldr w9, [x1]
+; CHECK-NEXT: cmp w8, w9
+; CHECK-NEXT: b.ge .LBB0_3
+; CHECK-NEXT: // %bb.1: // %entry
+; CHECK-NEXT: ldr w8, [x0, #4]
+; CHECK-NEXT: ldr w9, [x1, #4]
+; CHECK-NEXT: cmp w8, w9
+; CHECK-NEXT: b.ge .LBB0_3
+; CHECK-NEXT: // %bb.2: // %taken
+; CHECK-NEXT: b sink_a
+; CHECK-NEXT: .LBB0_3: // %fall
+; CHECK-NEXT: b sink_b
+entry:
+ %imin0 = load i32, ptr %imin
+ %imax0 = load i32, ptr %imax
+ %c0 = icmp slt i32 %imin0, %imax0
+ %imin1p = getelementptr inbounds i32, ptr %imin, i64 1
+ %imax1p = getelementptr inbounds i32, ptr %imax, i64 1
+ %imin1 = load i32, ptr %imin1p
+ %imax1 = load i32, ptr %imax1p
+ %c1 = icmp slt i32 %imin1, %imax1
+ %and = and i1 %c0, %c1
+ br i1 %and, label %taken, label %fall
+taken:
+ tail call void @sink_a()
+ ret void
+fall:
+ tail call void @sink_b()
+ ret void
+}
+
+; Negative control: register operands, not loads -> still merge into a CCMP chain.
+define void @both_reg(i32 %a0, i32 %b0, i32 %a1, i32 %b1) {
+; CHECK-LABEL: both_reg:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: cmp w2, w3
+; CHECK-NEXT: ccmp w0, w1, #0, lt
+; CHECK-NEXT: b.ge .LBB1_2
+; CHECK-NEXT: // %bb.1: // %taken
+; CHECK-NEXT: b sink_a
+; CHECK-NEXT: .LBB1_2: // %fall
+; CHECK-NEXT: b sink_b
+entry:
+ %c0 = icmp slt i32 %a0, %b0
+ %c1 = icmp slt i32 %a1, %b1
+ %and = and i1 %c0, %c1
+ br i1 %and, label %taken, label %fall
+taken:
+ tail call void @sink_a()
+ ret void
+fall:
+ tail call void @sink_b()
+ ret void
+}
+
+; Only one side is a load -> "both" requirement not met -> still merges.
+define void @one_loaded(ptr %imin, ptr %imax, i32 %a1, i32 %b1) {
+; CHECK-LABEL: one_loaded:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: ldr w8, [x1]
+; CHECK-NEXT: ldr w9, [x0]
+; CHECK-NEXT: cmp w2, w3
+; CHECK-NEXT: ccmp w9, w8, #0, lt
+; CHECK-NEXT: b.ge .LBB2_2
+; CHECK-NEXT: // %bb.1: // %taken
+; CHECK-NEXT: b sink_a
+; CHECK-NEXT: .LBB2_2: // %fall
+; CHECK-NEXT: b sink_b
+entry:
+ %imin0 = load i32, ptr %imin
+ %imax0 = load i32, ptr %imax
+ %c0 = icmp slt i32 %imin0, %imax0
+ %c1 = icmp slt i32 %a1, %b1
+ %and = and i1 %c0, %c1
+ br i1 %and, label %taken, label %fall
+taken:
+ tail call void @sink_a()
+ ret void
+fall:
+ tail call void @sink_b()
+ ret void
+}
diff --git a/llvm/test/CodeGen/AArch64/ragreedy-csr.ll b/llvm/test/CodeGen/AArch64/ragreedy-csr.ll
index 4ac7d0f1c34e6..249a08118386f 100644
--- a/llvm/test/CodeGen/AArch64/ragreedy-csr.ll
+++ b/llvm/test/CodeGen/AArch64/ragreedy-csr.ll
@@ -24,7 +24,7 @@ define fastcc i32 @prune_match(ptr nocapture readonly %a, ptr nocapture readonly
; CHECK-NEXT: ldrh w8, [x0]
; CHECK-NEXT: ldrh w9, [x1]
; CHECK-NEXT: cmp w8, w9
-; CHECK-NEXT: b.ne LBB0_42
+; CHECK-NEXT: b.ne LBB0_44
; CHECK-NEXT: ; %bb.1: ; %if.end
; CHECK-NEXT: sub sp, sp, #64
; CHECK-NEXT: stp x29, x30, [sp, #48] ; 16-byte Folded Spill
@@ -80,7 +80,7 @@ define fastcc i32 @prune_match(ptr nocapture readonly %a, ptr nocapture readonly
; CHECK-NEXT: ldrb w8, [x9, x11]
; CHECK-NEXT: ldrb w15, [x10, x11]
; CHECK-NEXT: cmp w8, w15
-; CHECK-NEXT: b.ne LBB0_25
+; CHECK-NEXT: b.ne LBB0_32
; CHECK-NEXT: ; %bb.7: ; %if.end17
; CHECK-NEXT: add x11, x11, #1
; CHECK-NEXT: ldrsb x8, [x9, x11]
@@ -138,12 +138,12 @@ define fastcc i32 @prune_match(ptr nocapture readonly %a, ptr nocapture readonly
; CHECK-NEXT: ; in Loop: Header=BB0_13 Depth=1
; CHECK-NEXT: cmp w8, #94
; CHECK-NEXT: ccmp w8, w11, #0, ne
-; CHECK-NEXT: b.ne LBB0_25
+; CHECK-NEXT: b.ne LBB0_32
; CHECK-NEXT: LBB0_17: ; %if.then83
; CHECK-NEXT: ; in Loop: Header=BB0_13 Depth=1
; CHECK-NEXT: ldrb w8, [x10], #1
; CHECK-NEXT: cbnz w8, LBB0_13
-; CHECK-NEXT: b LBB0_26
+; CHECK-NEXT: b LBB0_33
; CHECK-NEXT: LBB0_18: ; %land.lhs.true28
; CHECK-NEXT: cbz w8, LBB0_22
; CHECK-NEXT: ; %bb.19: ; %land.lhs.true28
@@ -157,95 +157,99 @@ define fastcc i32 @prune_match(ptr nocapture readonly %a, ptr nocapture readonly
; CHECK-NEXT: sub x12, x9, x12
; CHECK-NEXT: add x12, x12, x11
; CHECK-NEXT: cmp x12, #1
-; CHECK-NEXT: b.ne LBB0_39
+; CHECK-NEXT: b.ne LBB0_41
; CHECK-NEXT: LBB0_22:
; CHECK-NEXT: mov w0, #1 ; =0x1
-; CHECK-NEXT: b LBB0_26
+; CHECK-NEXT: b LBB0_33
; CHECK-NEXT: LBB0_23: ; %if.else88
+; CHECK-NEXT: cmp w12, #1
+; CHECK-NEXT: b.ne LBB0_34
+; CHECK-NEXT: ; %bb.24: ; %if.else88
; CHECK-NEXT: cmp w13, #2
-; CHECK-NEXT: ccmp w12, #1, #0, eq
-; CHECK-NEXT: b.eq LBB0_27
-; CHECK-NEXT: ; %bb.24: ; %if.else123
-; CHECK-NEXT: cmp w12, #2
-; CHECK-NEXT: ccmp w13, #1, #0, eq
-; CHECK-NEXT: b.eq LBB0_34
-; CHECK-NEXT: LBB0_25:
-; CHECK-NEXT: mov w0, wzr
-; CHECK-NEXT: LBB0_26:
-; CHECK-NEXT: ldp x29, x30, [sp, #48] ; 16-byte Folded Reload
-; CHECK-NEXT: add sp, sp, #64
-; CHECK-NEXT: ret
-; CHECK-NEXT: LBB0_27: ; %while.cond95.preheader
+; CHECK-NEXT: b.ne LBB0_34
+; CHECK-NEXT: ; %bb.25: ; %while.cond95.preheader
; CHECK-NEXT: ldrb w12, [x9, x11]
; CHECK-NEXT: cbz w12, LBB0_22
-; CHECK-NEXT: ; %bb.28: ; %land.rhs99.preheader
+; CHECK-NEXT: ; %bb.26: ; %land.rhs99.preheader
; CHECK-NEXT: mov x8, xzr
; CHECK-NEXT: mov w0, #1 ; =0x1
-; CHECK-NEXT: b LBB0_30
-; CHECK-NEXT: LBB0_29: ; %if.then117
-; CHECK-NEXT: ; in Loop: Header=BB0_30 Depth=1
+; CHECK-NEXT: b LBB0_28
+; CHECK-NEXT: LBB0_27: ; %if.then117
+; CHECK-NEXT: ; in Loop: Header=BB0_28 Depth=1
; CHECK-NEXT: add x12, x9, x8
; CHECK-NEXT: add x8, x8, #1
; CHECK-NEXT: add x12, x12, x11
; CHECK-NEXT: ldrb w12, [x12, #1]
-; CHECK-NEXT: cbz w12, LBB0_26
-; CHECK-NEXT: LBB0_30: ; %land.rhs99
+; CHECK-NEXT: cbz w12, LBB0_33
+; CHECK-NEXT: LBB0_28: ; %land.rhs99
; CHECK-NEXT: ; =>This Inner Loop Header: Depth=1
; CHECK-NEXT: add x13, x10, x8
; CHECK-NEXT: ldrb w13, [x13, x11]
; CHECK-NEXT: cbz w13, LBB0_22
-; CHECK-NEXT: ; %bb.31: ; %while.body104
-; CHECK-NEXT: ; in Loop: Header=BB0_30 Depth=1
+; CHECK-NEXT: ; %bb.29: ; %while.body104
+; CHECK-NEXT: ; in Loop: Header=BB0_28 Depth=1
; CHECK-NEXT: cmp w12, w13
-; CHECK-NEXT: b.eq LBB0_29
-; CHECK-NEXT: ; %bb.32: ; %while.body104
-; CHECK-NEXT: ; in Loop: Header=BB0_30 Depth=1
+; CHECK-NEXT: b.eq LBB0_27
+; CHECK-NEXT: ; %bb.30: ; %while.body104
+; CHECK-NEXT: ; in Loop: Header=BB0_28 Depth=1
; CHECK-NEXT: cmp w12, #42
-; CHECK-NEXT: b.eq LBB0_29
-; CHECK-NEXT: ; %bb.33: ; %while.body104
-; CHECK-NEXT: ; in Loop: Header=BB0_30 Depth=1
+; CHECK-NEXT: b.eq LBB0_27
+; CHECK-NEXT: ; %bb.31: ; %while.body104
+; CHECK-NEXT: ; in Loop: Header=BB0_28 Depth=1
; CHECK-NEXT: cmp w13, #94
-; CHECK-NEXT: b.eq LBB0_29
-; CHECK-NEXT: b LBB0_25
-; CHECK-NEXT: LBB0_34: ; %while.cond130.preheader
+; CHECK-NEXT: b.eq LBB0_27
+; CHECK-NEXT: LBB0_32:
+; CHECK-NEXT: mov w0, wzr
+; CHECK-NEXT: LBB0_33:
+; CHECK-NEXT: ldp x29, x30, [sp, #48] ; 16-byte Folded Reload
+; CHECK-NEXT: add sp, sp, #64
+; CHECK-NEXT: ret
+; CHECK-NEXT: LBB0_34: ; %if.else123
+; CHECK-NEXT: cmp w13, #1
+; CHECK-NEXT: mov w0, wzr
+; CHECK-NEXT: b.ne LBB0_33
+; CHECK-NEXT: ; %bb.35: ; %if.else123
+; CHECK-NEXT: cmp w12, #2
+; CHECK-NEXT: b.ne LBB0_33
+; CHECK-NEXT: ; %bb.36: ; %while.cond130.preheader
; CHECK-NEXT: ldrb w12, [x9, x11]
; CHECK-NEXT: cbz w12, LBB0_22
-; CHECK-NEXT: ; %bb.35: ; %land.rhs134.preheader
+; CHECK-NEXT: ; %bb.37: ; %land.rhs134.preheader
; CHECK-NEXT: mov x8, xzr
; CHECK-NEXT: mov w13, #42 ; =0x2a
; CHECK-NEXT: mov w0, #1 ; =0x1
-; CHECK-NEXT: LBB0_36: ; %land.rhs134
+; CHECK-NEXT: LBB0_38: ; %land.rhs134
; CHECK-NEXT: ; =>This Inner Loop Header: Depth=1
; CHECK-NEXT: add x14, x10, x8
; CHECK-NEXT: ldrb w14, [x14, x11]
; CHECK-NEXT: cbz w14, LBB0_22
-; CHECK-NEXT: ; %bb.37: ; %while.body139
-; CHECK-NEXT: ; in Loop: Header=BB0_36 Depth=1
+; CHECK-NEXT: ; %bb.39: ; %while.body139
+; CHECK-NEXT: ; in Loop: Header=BB0_38 Depth=1
; CHECK-NEXT: cmp w12, #94
; CHECK-NEXT: ccmp w14, w13, #4, ne
; CHECK-NEXT: ccmp w12, w14, #4, ne
-; CHECK-NEXT: b.ne LBB0_25
-; CHECK-NEXT: ; %bb.38: ; %if.then152
-; CHECK-NEXT: ; in Loop: Header=BB0_36 Depth=1
+; CHECK-NEXT: b.ne LBB0_32
+; CHECK-NEXT: ; %bb.40: ; %if.then152
+; CHECK-NEXT: ; in Loop: Header=BB0_38 Depth=1
; CHECK-NEXT: add x12, x9, x8
; CHECK-NEXT: add x8, x8, #1
; CHECK-NEXT: add x12, x12, x11
; CHECK-NEXT: ldrb w12, [x12, #1]
-; CHECK-NEXT: cbnz w12, LBB0_36
-; CHECK-NEXT: b LBB0_26
-; CHECK-NEXT: LBB0_39: ; %lor.lhs.false47
+; CHECK-NEXT: cbnz w12, LBB0_38
+; CHECK-NEXT: b LBB0_33
+; CHECK-NEXT: LBB0_41: ; %lor.lhs.false47
; CHECK-NEXT: cmp x12, #2
; CHECK-NEXT: b.ne LBB0_11
-; CHECK-NEXT: ; %bb.40: ; %land.lhs.true52
+; CHECK-NEXT: ; %bb.42: ; %land.lhs.true52
; CHECK-NEXT: add x12, x9, x11
; CHECK-NEXT: mov w0, #1 ; =0x1
; CHECK-NEXT: ldurb w12, [x12, #-1]
; CHECK-NEXT: cmp w12, #73
-; CHECK-NEXT: b.eq LBB0_26
-; CHECK-NEXT: ; %bb.41: ; %land.lhs.true52
-; CHECK-NEXT: cbz w8, LBB0_26
+; CHECK-NEXT: b.eq LBB0_33
+; CHECK-NEXT: ; %bb.43: ; %land.lhs.true52
+; CHECK-NEXT: cbz w8, LBB0_33
; CHECK-NEXT: b LBB0_12
-; CHECK-NEXT: LBB0_42:
+; CHECK-NEXT: LBB0_44:
; CHECK-NEXT: mov w0, wzr
; CHECK-NEXT: ret
; CHECK-NEXT: .loh AdrpLdrGot Lloh0, Lloh1
More information about the llvm-commits
mailing list