[llvm] [AArch64] Lower an equality branch against a logical immediate to EOR + CBZ/CBNZ (PR #223123)
Andrew Gaul via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 14 10:48:03 PDT 2026
https://github.com/gaul updated https://github.com/llvm/llvm-project/pull/223123
>From 004f8a7378117b3e0aa87ee127a29b7d743095e9 Mon Sep 17 00:00:00 2001
From: Andrew Gaul <andrew at gaul.org>
Date: Fri, 11 Sep 2026 21:52:07 -0700
Subject: [PATCH 1/2] [AArch64] Add tests for equality branches against logical
immediates (NFC)
Branches on x == C and x != C where C is not a legal CMP/CMN immediate:
constants that are logical immediates (INT64_MIN, the i32 sign bit,
0xffffff, 0x7fffffff), the same constant reused across a call, a loop
whose constant is hoisted into a callee-saved register, and the shapes
that must keep their current lowering -- a CMP or CMN immediate, a
constant that is neither kind of immediate, an ordered condition, a
value-producing compare, two tests merged into a conditional-compare
chain, a constant shared with a select, and a function under speculative
load hardening. The checks record the current codegen, which materializes
the constant for every compare.
Co-Authored-By: Claude Fable 5.1 <noreply at anthropic.com>
Claude-Session: https://claude.ai/code/session_01KZMjy58X6Z2sH8fd4SSRi4
---
.../CodeGen/AArch64/branch-eq-logical-imm.ll | 512 ++++++++++++++++++
1 file changed, 512 insertions(+)
create mode 100644 llvm/test/CodeGen/AArch64/branch-eq-logical-imm.ll
diff --git a/llvm/test/CodeGen/AArch64/branch-eq-logical-imm.ll b/llvm/test/CodeGen/AArch64/branch-eq-logical-imm.ll
new file mode 100644
index 0000000000000..f75d436f19605
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/branch-eq-logical-imm.ll
@@ -0,0 +1,512 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=aarch64-linux-gnu < %s | FileCheck %s
+
+; Equality branches against constants that no CMP/CMN immediate encodes.
+; The sign bit is the common one: rustc's enum niches and INT64_MIN sentinels.
+
+declare void @g()
+declare void @h()
+
+define void @br_eq_int64_min(i64 %x) {
+; CHECK-LABEL: br_eq_int64_min:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: mov x8, #-9223372036854775808 // =0x8000000000000000
+; CHECK-NEXT: cmp x0, x8
+; CHECK-NEXT: b.ne .LBB0_2
+; CHECK-NEXT: // %bb.1: // %then
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: bl g
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: .LBB0_2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %c = icmp eq i64 %x, -9223372036854775808
+ br i1 %c, label %then, label %exit
+then:
+ call void @g()
+ br label %exit
+exit:
+ ret void
+}
+
+define void @br_ne_int64_min(i64 %x) {
+; CHECK-LABEL: br_ne_int64_min:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: mov x8, #-9223372036854775808 // =0x8000000000000000
+; CHECK-NEXT: cmp x0, x8
+; CHECK-NEXT: b.eq .LBB1_2
+; CHECK-NEXT: // %bb.1: // %then
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: bl g
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: .LBB1_2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %c = icmp ne i64 %x, -9223372036854775808
+ br i1 %c, label %then, label %exit
+then:
+ call void @g()
+ br label %exit
+exit:
+ ret void
+}
+
+define void @br_eq_i32_signbit(i32 %x) {
+; CHECK-LABEL: br_eq_i32_signbit:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: mov w8, #-2147483648 // =0x80000000
+; CHECK-NEXT: cmp w0, w8
+; CHECK-NEXT: b.ne .LBB2_2
+; CHECK-NEXT: // %bb.1: // %then
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: bl g
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: .LBB2_2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %c = icmp eq i32 %x, -2147483648
+ br i1 %c, label %then, label %exit
+then:
+ call void @g()
+ br label %exit
+exit:
+ ret void
+}
+
+define void @br_ne_i32_ffffff(i32 %x) {
+; CHECK-LABEL: br_ne_i32_ffffff:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: mov w8, #16777215 // =0xffffff
+; CHECK-NEXT: cmp w0, w8
+; CHECK-NEXT: b.eq .LBB3_2
+; CHECK-NEXT: // %bb.1: // %then
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: bl g
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: .LBB3_2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %c = icmp ne i32 %x, 16777215
+ br i1 %c, label %then, label %exit
+then:
+ call void @g()
+ br label %exit
+exit:
+ ret void
+}
+
+define void @br_eq_i64_int32_max(i64 %x) {
+; CHECK-LABEL: br_eq_i64_int32_max:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: mov w8, #2147483647 // =0x7fffffff
+; CHECK-NEXT: cmp x0, x8
+; CHECK-NEXT: b.ne .LBB4_2
+; CHECK-NEXT: // %bb.1: // %then
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: bl g
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: .LBB4_2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %c = icmp eq i64 %x, 2147483647
+ br i1 %c, label %then, label %exit
+then:
+ call void @g()
+ br label %exit
+exit:
+ ret void
+}
+
+; The constant is used twice, across a call.
+define void @br_two_uses(i64 %x, i64 %y) {
+; CHECK-LABEL: br_two_uses:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: str x30, [sp, #-32]! // 8-byte Folded Spill
+; CHECK-NEXT: stp x20, x19, [sp, #16] // 16-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 32
+; CHECK-NEXT: .cfi_offset w19, -8
+; CHECK-NEXT: .cfi_offset w20, -16
+; CHECK-NEXT: .cfi_offset w30, -32
+; CHECK-NEXT: mov x20, #-9223372036854775808 // =0x8000000000000000
+; CHECK-NEXT: mov x19, x1
+; CHECK-NEXT: cmp x0, x20
+; CHECK-NEXT: b.ne .LBB5_2
+; CHECK-NEXT: // %bb.1: // %callg
+; CHECK-NEXT: bl g
+; CHECK-NEXT: .LBB5_2: // %second
+; CHECK-NEXT: cmp x19, x20
+; CHECK-NEXT: b.ne .LBB5_4
+; CHECK-NEXT: // %bb.3: // %callh
+; CHECK-NEXT: bl h
+; CHECK-NEXT: .LBB5_4: // %exit
+; CHECK-NEXT: ldp x20, x19, [sp, #16] // 16-byte Folded Reload
+; CHECK-NEXT: ldr x30, [sp], #32 // 8-byte Folded Reload
+; CHECK-NEXT: ret
+entry:
+ %cx = icmp eq i64 %x, -9223372036854775808
+ br i1 %cx, label %callg, label %second
+callg:
+ call void @g()
+ br label %second
+second:
+ %cy = icmp eq i64 %y, -9223372036854775808
+ br i1 %cy, label %callh, label %exit
+callh:
+ call void @h()
+ br label %exit
+exit:
+ ret void
+}
+
+; A CMP immediate keeps the compare.
+define void @br_eq_imm12(i64 %x) {
+; CHECK-LABEL: br_eq_imm12:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: cmp x0, #100
+; CHECK-NEXT: b.ne .LBB6_2
+; CHECK-NEXT: // %bb.1: // %then
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: bl g
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: .LBB6_2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %c = icmp eq i64 %x, 100
+ br i1 %c, label %then, label %exit
+then:
+ call void @g()
+ br label %exit
+exit:
+ ret void
+}
+
+; A negative constant whose magnitude encodes keeps CMN.
+define void @br_eq_neg5(i64 %x) {
+; CHECK-LABEL: br_eq_neg5:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: cmn x0, #5
+; CHECK-NEXT: b.ne .LBB7_2
+; CHECK-NEXT: // %bb.1: // %then
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: bl g
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: .LBB7_2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %c = icmp eq i64 %x, -5
+ br i1 %c, label %then, label %exit
+then:
+ call void @g()
+ br label %exit
+exit:
+ ret void
+}
+
+; Not a logical immediate: the constant is materialized.
+define void @br_eq_0x1234(i64 %x) {
+; CHECK-LABEL: br_eq_0x1234:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: mov w8, #4660 // =0x1234
+; CHECK-NEXT: cmp x0, x8
+; CHECK-NEXT: b.ne .LBB8_2
+; CHECK-NEXT: // %bb.1: // %then
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: bl g
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: .LBB8_2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %c = icmp eq i64 %x, 4660
+ br i1 %c, label %then, label %exit
+then:
+ call void @g()
+ br label %exit
+exit:
+ ret void
+}
+
+; An ordered condition needs the subtraction.
+define void @br_sgt_int32_max(i64 %x) {
+; CHECK-LABEL: br_sgt_int32_max:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: mov w8, #-2147483648 // =0x80000000
+; CHECK-NEXT: cmp x0, x8
+; CHECK-NEXT: b.lt .LBB9_2
+; CHECK-NEXT: // %bb.1: // %then
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: bl g
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: .LBB9_2: // %exit
+; CHECK-NEXT: ret
+entry:
+ %c = icmp sgt i64 %x, 2147483647
+ br i1 %c, label %then, label %exit
+then:
+ call void @g()
+ br label %exit
+exit:
+ ret void
+}
+
+; Value-producing compares.
+define i1 @setcc_int64_min(i64 %x) {
+; CHECK-LABEL: setcc_int64_min:
+; CHECK: // %bb.0:
+; CHECK-NEXT: mov x8, #-9223372036854775808 // =0x8000000000000000
+; CHECK-NEXT: cmp x0, x8
+; CHECK-NEXT: cset w0, eq
+; CHECK-NEXT: ret
+ %c = icmp eq i64 %x, -9223372036854775808
+ ret i1 %c
+}
+
+define i64 @select_int64_min(i64 %x, i64 %a, i64 %b) {
+; CHECK-LABEL: select_int64_min:
+; CHECK: // %bb.0:
+; CHECK-NEXT: mov x8, #-9223372036854775808 // =0x8000000000000000
+; CHECK-NEXT: cmp x0, x8
+; CHECK-NEXT: csel x0, x1, x2, eq
+; CHECK-NEXT: ret
+ %c = icmp eq i64 %x, -9223372036854775808
+ %r = select i1 %c, i64 %a, i64 %b
+ ret i64 %r
+}
+
+; Three tests lowered from a switch and merged by the conditional-compare
+; pass: the compares' flags feed a CCMP chain, not a branch.
+define i64 @br_ccmp_chain(i64 %x0, i64 %n) {
+; CHECK-LABEL: br_ccmp_chain:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: stp x30, x25, [sp, #-64]! // 16-byte Folded Spill
+; CHECK-NEXT: stp x24, x23, [sp, #16] // 16-byte Folded Spill
+; CHECK-NEXT: stp x22, x21, [sp, #32] // 16-byte Folded Spill
+; CHECK-NEXT: stp x20, x19, [sp, #48] // 16-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 64
+; CHECK-NEXT: .cfi_offset w19, -8
+; CHECK-NEXT: .cfi_offset w20, -16
+; CHECK-NEXT: .cfi_offset w21, -24
+; CHECK-NEXT: .cfi_offset w22, -32
+; CHECK-NEXT: .cfi_offset w23, -40
+; CHECK-NEXT: .cfi_offset w24, -48
+; CHECK-NEXT: .cfi_offset w25, -56
+; CHECK-NEXT: .cfi_offset w30, -64
+; CHECK-NEXT: cmp x1, #1
+; CHECK-NEXT: mov x19, x0
+; CHECK-NEXT: b.lt .LBB12_5
+; CHECK-NEXT: // %bb.1: // %loop.preheader
+; CHECK-NEXT: mov x21, #32557 // =0x7f2d
+; CHECK-NEXT: mov x22, #33103 // =0x814f
+; CHECK-NEXT: mov x20, x1
+; CHECK-NEXT: movk x21, #19605, lsl #16
+; CHECK-NEXT: movk x22, #63335, lsl #16
+; CHECK-NEXT: mov w23, #65535 // =0xffff
+; CHECK-NEXT: movk x21, #62509, lsl #32
+; CHECK-NEXT: movk x22, #31614, lsl #32
+; CHECK-NEXT: mov x24, #9223372036854775807 // =0x7fffffffffffffff
+; CHECK-NEXT: movk x21, #22609, lsl #48
+; CHECK-NEXT: movk x22, #5125, lsl #48
+; CHECK-NEXT: mov w25, #-1 // =0xffffffff
+; CHECK-NEXT: b .LBB12_3
+; CHECK-NEXT: .LBB12_2: // %latch
+; CHECK-NEXT: // in Loop: Header=BB12_3 Depth=1
+; CHECK-NEXT: subs x20, x20, #1
+; CHECK-NEXT: b.eq .LBB12_5
+; CHECK-NEXT: .LBB12_3: // %loop
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: madd x19, x19, x21, x22
+; CHECK-NEXT: cmp x19, x23
+; CHECK-NEXT: ccmp x19, x24, #4, ne
+; CHECK-NEXT: ccmp x19, x25, #4, ne
+; CHECK-NEXT: b.ne .LBB12_2
+; CHECK-NEXT: // %bb.4: // %call
+; CHECK-NEXT: // in Loop: Header=BB12_3 Depth=1
+; CHECK-NEXT: bl g
+; CHECK-NEXT: b .LBB12_2
+; CHECK-NEXT: .LBB12_5: // %exit
+; CHECK-NEXT: mov x0, x19
+; CHECK-NEXT: ldp x20, x19, [sp, #48] // 16-byte Folded Reload
+; CHECK-NEXT: ldp x22, x21, [sp, #32] // 16-byte Folded Reload
+; CHECK-NEXT: ldp x24, x23, [sp, #16] // 16-byte Folded Reload
+; CHECK-NEXT: ldp x30, x25, [sp], #64 // 16-byte Folded Reload
+; CHECK-NEXT: ret
+entry:
+ %cmp0 = icmp sgt i64 %n, 0
+ br i1 %cmp0, label %loop, label %exit
+loop:
+ %i = phi i64 [ %inc, %latch ], [ 0, %entry ]
+ %x = phi i64 [ %next, %latch ], [ %x0, %entry ]
+ %mul = mul nsw i64 %x, 6364136223846793005
+ %next = add nsw i64 %mul, 1442695040888963407
+ switch i64 %next, label %latch [
+ i64 9223372036854775807, label %call
+ i64 4294967295, label %call
+ i64 65535, label %call
+ ]
+call:
+ call void @g()
+ br label %latch
+latch:
+ %inc = add nuw nsw i64 %i, 1
+ %done = icmp eq i64 %inc, %n
+ br i1 %done, label %exit, label %loop
+exit:
+ %r = phi i64 [ %x0, %entry ], [ %next, %latch ]
+ ret i64 %r
+}
+
+; The constant is shared with a select.
+define i64 @br_shared_with_select(i64 %x, i64 %y, i64 %a, i64 %b) {
+; CHECK-LABEL: br_shared_with_select:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: str x30, [sp, #-48]! // 8-byte Folded Spill
+; CHECK-NEXT: stp x22, x21, [sp, #16] // 16-byte Folded Spill
+; CHECK-NEXT: stp x20, x19, [sp, #32] // 16-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 48
+; CHECK-NEXT: .cfi_offset w19, -8
+; CHECK-NEXT: .cfi_offset w20, -16
+; CHECK-NEXT: .cfi_offset w21, -24
+; CHECK-NEXT: .cfi_offset w22, -32
+; CHECK-NEXT: .cfi_offset w30, -48
+; CHECK-NEXT: mov x22, #-9223372036854775808 // =0x8000000000000000
+; CHECK-NEXT: mov x19, x3
+; CHECK-NEXT: mov x20, x2
+; CHECK-NEXT: cmp x0, x22
+; CHECK-NEXT: mov x21, x1
+; CHECK-NEXT: b.ne .LBB13_2
+; CHECK-NEXT: // %bb.1: // %then
+; CHECK-NEXT: bl g
+; CHECK-NEXT: .LBB13_2: // %exit
+; CHECK-NEXT: cmp x21, x22
+; CHECK-NEXT: ldp x22, x21, [sp, #16] // 16-byte Folded Reload
+; CHECK-NEXT: csel x0, x20, x19, eq
+; CHECK-NEXT: ldp x20, x19, [sp, #32] // 16-byte Folded Reload
+; CHECK-NEXT: ldr x30, [sp], #48 // 8-byte Folded Reload
+; CHECK-NEXT: ret
+entry:
+ %c = icmp eq i64 %x, -9223372036854775808
+ br i1 %c, label %then, label %exit
+then:
+ call void @g()
+ br label %exit
+exit:
+ %d = icmp eq i64 %y, -9223372036854775808
+ %r = select i1 %d, i64 %a, i64 %b
+ ret i64 %r
+}
+
+; A loop containing a call: the constant is hoisted into a callee-saved
+; register.
+define void @br_loop(ptr %p, i64 %n) {
+; CHECK-LABEL: br_loop:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: cmp x1, #1
+; CHECK-NEXT: b.lt .LBB14_6
+; CHECK-NEXT: // %bb.1: // %loop.preheader
+; CHECK-NEXT: stp x30, x21, [sp, #-32]! // 16-byte Folded Spill
+; CHECK-NEXT: stp x20, x19, [sp, #16] // 16-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 32
+; CHECK-NEXT: .cfi_offset w19, -8
+; CHECK-NEXT: .cfi_offset w20, -16
+; CHECK-NEXT: .cfi_offset w21, -24
+; CHECK-NEXT: .cfi_offset w30, -32
+; CHECK-NEXT: mov x19, x1
+; CHECK-NEXT: mov x20, x0
+; CHECK-NEXT: mov x21, #-9223372036854775808 // =0x8000000000000000
+; CHECK-NEXT: b .LBB14_3
+; CHECK-NEXT: .LBB14_2: // %latch
+; CHECK-NEXT: // in Loop: Header=BB14_3 Depth=1
+; CHECK-NEXT: subs x19, x19, #1
+; CHECK-NEXT: b.eq .LBB14_5
+; CHECK-NEXT: .LBB14_3: // %loop
+; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: ldr x8, [x20], #8
+; CHECK-NEXT: cmp x8, x21
+; CHECK-NEXT: b.ne .LBB14_2
+; CHECK-NEXT: // %bb.4: // %call
+; CHECK-NEXT: // in Loop: Header=BB14_3 Depth=1
+; CHECK-NEXT: bl g
+; CHECK-NEXT: b .LBB14_2
+; CHECK-NEXT: .LBB14_5:
+; CHECK-NEXT: ldp x20, x19, [sp, #16] // 16-byte Folded Reload
+; CHECK-NEXT: ldp x30, x21, [sp], #32 // 16-byte Folded Reload
+; CHECK-NEXT: .LBB14_6: // %exit
+; CHECK-NEXT: ret
+entry:
+ %cmp0 = icmp sgt i64 %n, 0
+ br i1 %cmp0, label %loop, label %exit
+loop:
+ %i = phi i64 [ 0, %entry ], [ %inc, %latch ]
+ %addr = getelementptr inbounds i64, ptr %p, i64 %i
+ %v = load i64, ptr %addr
+ %c = icmp eq i64 %v, -9223372036854775808
+ br i1 %c, label %call, label %latch
+call:
+ call void @g()
+ br label %latch
+latch:
+ %inc = add nuw nsw i64 %i, 1
+ %done = icmp eq i64 %inc, %n
+ br i1 %done, label %exit, label %loop
+exit:
+ ret void
+}
+
+; Speculative load hardening needs a flag-setting branch.
+define void @br_eq_int64_min_slh(i64 %x) speculative_load_hardening {
+; CHECK-LABEL: br_eq_int64_min_slh:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: cmp sp, #0
+; CHECK-NEXT: mov x8, #-9223372036854775808 // =0x8000000000000000
+; CHECK-NEXT: csetm x16, ne
+; CHECK-NEXT: cmp x0, x8
+; CHECK-NEXT: b.ne .LBB15_3
+; CHECK-NEXT: // %bb.1:
+; CHECK-NEXT: csel x16, x16, xzr, eq
+; CHECK-NEXT: // %bb.2: // %then
+; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: mov x0, sp
+; CHECK-NEXT: and x0, x0, x16
+; CHECK-NEXT: mov sp, x0
+; CHECK-NEXT: bl g
+; CHECK-NEXT: cmp sp, #0
+; CHECK-NEXT: csetm x16, ne
+; CHECK-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: and x30, x30, x16
+; CHECK-NEXT: csdb
+; CHECK-NEXT: b .LBB15_4
+; CHECK-NEXT: .LBB15_3:
+; CHECK-NEXT: csel x16, x16, xzr, ne
+; CHECK-NEXT: .LBB15_4: // %exit
+; CHECK-NEXT: mov x0, sp
+; CHECK-NEXT: and x0, x0, x16
+; CHECK-NEXT: mov sp, x0
+; CHECK-NEXT: ret
+entry:
+ %c = icmp eq i64 %x, -9223372036854775808
+ br i1 %c, label %then, label %exit
+then:
+ call void @g()
+ br label %exit
+exit:
+ ret void
+}
>From 101c2a104dd7255577bb3c3b6b6d3fbb1a21d4bc Mon Sep 17 00:00:00 2001
From: Andrew Gaul <andrew at gaul.org>
Date: Sat, 12 Sep 2026 00:23:32 -0700
Subject: [PATCH 2/2] [AArch64] Fold a compare-and-branch against the sign bit
into CMP XZR + B.VS/B.VC
x == INT64_MIN feeding a branch materializes the constant:
mov x8, #-9223372036854775808
cmp x0, x8
b.ne .LBB0_2
Zero minus x overflows exactly when x is INT64_MIN, so the same test is
cmp xzr, x0
b.vc .LBB0_2
Two instructions, no scratch register, no materialization, and still a
compare-and-branch that fuses on the cores that fuse CMP with B.cond. A
rewrite to EOR + CBZ/CBNZ would cover any logical immediate but gives
that fusion up: two issue slots against a fused pair plus a MOV that
some cores resolve at rename, as pointed out in review. The sign bit is
the case that dominates: rustc places the niches of an enum that has two
or more dataless variants beside a Vec or String payload at
isize::MAX + 1 upward, so every test for such a variant compares against
0x8000000000000000, and Gecko's TimeDuration keeps INT64_MIN as a
sentinel. Firefox 155's libxul.so contains 13,019 branch compares
against a logical immediate that is not a CMP immediate, 11,262 of them
against the sign bit. The W form handles 0x80000000.
The fold is a pre-RA peephole in AArch64MIPeepholeOpt, keyed on the
B.cond: the nearest NZCV writer above it must be a register-form SUBS
with a dead result, one operand a MOVi64imm or MOVi32imm of the sign
bit, and its flags must reach exactly that branch (examineCFlagsUse,
which also refuses flags live into a successor), since the rewritten
compare sets N, Z and C differently. The SUBS is rewritten in place to
compare XZR against the tested value, the branch condition becomes VS
for EQ and VC for NE, and the MOV is deleted with its last use. Running
after the conditional-compare pass, a compare that became part of a
CCMP chain has no branch consumer and is never a candidate; the branch
stays a B.cond, so speculative load hardening is unaffected. A converted
site costs the same fused slot as the original whether or not the MOV
survives for another consumer, so no use-count condition is needed:
when the constant is reused across a call or in a loop and LLVM had
parked it in a callee-saved register, that register and its spill go
away.
On Apple M4 Max the shapes measured are at parity or better; the
per-site saving is one instruction of fetch and, where the constant was
hoisted, a callee-saved register.
Found via armlint.
Co-Authored-By: Claude Fable 5.1 <noreply at anthropic.com>
Claude-Session: https://claude.ai/code/session_01KZMjy58X6Z2sH8fd4SSRi4
---
.../Target/AArch64/AArch64MIPeepholeOpt.cpp | 89 +++++++++++++++++++
.../CodeGen/AArch64/branch-eq-logical-imm.ll | 73 +++++++--------
2 files changed, 123 insertions(+), 39 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp b/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp
index 554d5938cf2cd..b454fbcb41dc2 100644
--- a/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp
+++ b/llvm/lib/Target/AArch64/AArch64MIPeepholeOpt.cpp
@@ -143,6 +143,7 @@ class AArch64MIPeepholeOptImpl {
bool visitFMOVDr(MachineInstr &MI);
bool visitUBFMXri(MachineInstr &MI);
bool visitCopy(MachineInstr &MI);
+ bool visitBcc(MachineInstr &MI);
};
struct AArch64MIPeepholeOptLegacy : public MachineFunctionPass {
@@ -957,6 +958,91 @@ bool AArch64MIPeepholeOptImpl::visitCopy(MachineInstr &MI) {
return true;
}
+// A compare against the sign bit feeding an equality branch:
+// %c = MOVi64imm 0x8000000000000000
+// %d = SUBSXrr %x, %c, implicit-def $nzcv ; %d dead
+// Bcc EQ|NE, %bb
+// Zero minus x overflows exactly when x is INT64_MIN, so the same test is
+// %d = SUBSXrr $xzr, %x, implicit-def $nzcv
+// Bcc VS|VC, %bb
+// -- two instructions, no scratch register, no materialization, and still a
+// compare-and-branch that fuses on the cores that fuse CMP with B.cond. (A
+// rewrite to EOR + CBZ/CBNZ would cover any logical immediate but gives that
+// up: two issue slots against a fused pair plus a MOV that some cores resolve
+// at rename.) The MOV goes away with its last use. The W form handles
+// 0x80000000. Running after the conditional-compare pass, a compare that
+// became part of a CCMP chain has no branch consumer and is never a candidate.
+bool AArch64MIPeepholeOptImpl::visitBcc(MachineInstr &MI) {
+ auto CC = static_cast<AArch64CC::CondCode>(MI.getOperand(0).getImm());
+ if (CC != AArch64CC::EQ && CC != AArch64CC::NE)
+ return false;
+
+ // The compare the branch reads: the nearest NZCV writer above it.
+ MachineBasicBlock &MBB = *MI.getParent();
+ MachineInstr *Cmp = nullptr;
+ for (MachineBasicBlock::iterator It = MI.getIterator(); It != MBB.begin();) {
+ --It;
+ if (It->modifiesRegister(AArch64::NZCV, TRI)) {
+ Cmp = &*It;
+ break;
+ }
+ }
+ if (!Cmp)
+ return false;
+ unsigned CmpOpc = Cmp->getOpcode();
+ if (CmpOpc != AArch64::SUBSWrr && CmpOpc != AArch64::SUBSXrr)
+ return false;
+ bool Is64 = CmpOpc == AArch64::SUBSXrr;
+ Register Dst = Cmp->getOperand(0).getReg();
+ if (Dst.isVirtual() ? !MRI->use_nodbg_empty(Dst)
+ : (Dst != AArch64::WZR && Dst != AArch64::XZR))
+ return false;
+
+ // One operand materializes the sign bit; the other is the tested value.
+ unsigned ConstIdx = 0;
+ MachineInstr *MovMI = nullptr;
+ uint64_t SignBit = Is64 ? (1ULL << 63) : (1ULL << 31);
+ for (unsigned Idx : {1u, 2u}) {
+ Register R = Cmp->getOperand(Idx).getReg();
+ if (!R.isVirtual())
+ continue;
+ MachineInstr *Def = MRI->getUniqueVRegDef(R);
+ if (!Def ||
+ Def->getOpcode() != (Is64 ? AArch64::MOVi64imm : AArch64::MOVi32imm))
+ continue;
+ uint64_t Imm = Def->getOperand(1).getImm();
+ if (!Is64)
+ Imm &= 0xffffffffu;
+ if (Imm != SignBit)
+ continue;
+ ConstIdx = Idx;
+ MovMI = Def;
+ break;
+ }
+ if (!MovMI)
+ return false;
+ Register ConstReg = Cmp->getOperand(ConstIdx).getReg();
+ Register X = Cmp->getOperand(ConstIdx == 1 ? 2 : 1).getReg();
+ if (!X.isVirtual() || X == ConstReg)
+ return false;
+
+ // The flags must reach exactly this branch: the rewritten compare sets N,
+ // Z and C differently. examineCFlagsUse also refuses flags that are live
+ // into a successor.
+ SmallVector<MachineInstr *, 4> Users;
+ std::optional<UsedNZCV> Used = examineCFlagsUse(*Cmp, *Cmp, *TRI, &Users);
+ if (!Used || Users.size() != 1 || Users[0] != &MI)
+ return false;
+
+ Cmp->getOperand(1).ChangeToRegister(Is64 ? AArch64::XZR : AArch64::WZR,
+ /*isDef=*/false);
+ Cmp->getOperand(2).ChangeToRegister(X, /*isDef=*/false);
+ MI.getOperand(0).setImm(CC == AArch64CC::EQ ? AArch64CC::VS : AArch64CC::VC);
+ if (MRI->use_nodbg_empty(ConstReg))
+ MovMI->eraseFromParent();
+ return true;
+}
+
bool AArch64MIPeepholeOptImpl::run(MachineFunction &MF) {
TII = static_cast<const AArch64InstrInfo *>(MF.getSubtarget().getInstrInfo());
TRI = static_cast<const AArch64RegisterInfo *>(
@@ -1070,6 +1156,9 @@ bool AArch64MIPeepholeOptImpl::run(MachineFunction &MF) {
case AArch64::COPY:
Changed |= visitCopy(MI);
break;
+ case AArch64::Bcc:
+ Changed |= visitBcc(MI);
+ break;
}
}
}
diff --git a/llvm/test/CodeGen/AArch64/branch-eq-logical-imm.ll b/llvm/test/CodeGen/AArch64/branch-eq-logical-imm.ll
index f75d436f19605..f4508fa30ba44 100644
--- a/llvm/test/CodeGen/AArch64/branch-eq-logical-imm.ll
+++ b/llvm/test/CodeGen/AArch64/branch-eq-logical-imm.ll
@@ -3,6 +3,7 @@
; Equality branches against constants that no CMP/CMN immediate encodes.
; The sign bit is the common one: rustc's enum niches and INT64_MIN sentinels.
+; x == INT_MIN is the overflow of 0 - x, so it lowers to cmp xzr, x ; b.vs.
declare void @g()
declare void @h()
@@ -10,9 +11,8 @@ declare void @h()
define void @br_eq_int64_min(i64 %x) {
; CHECK-LABEL: br_eq_int64_min:
; CHECK: // %bb.0: // %entry
-; CHECK-NEXT: mov x8, #-9223372036854775808 // =0x8000000000000000
-; CHECK-NEXT: cmp x0, x8
-; CHECK-NEXT: b.ne .LBB0_2
+; CHECK-NEXT: cmp xzr, x0
+; CHECK-NEXT: b.vc .LBB0_2
; CHECK-NEXT: // %bb.1: // %then
; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
; CHECK-NEXT: .cfi_def_cfa_offset 16
@@ -34,9 +34,8 @@ exit:
define void @br_ne_int64_min(i64 %x) {
; CHECK-LABEL: br_ne_int64_min:
; CHECK: // %bb.0: // %entry
-; CHECK-NEXT: mov x8, #-9223372036854775808 // =0x8000000000000000
-; CHECK-NEXT: cmp x0, x8
-; CHECK-NEXT: b.eq .LBB1_2
+; CHECK-NEXT: cmp xzr, x0
+; CHECK-NEXT: b.vs .LBB1_2
; CHECK-NEXT: // %bb.1: // %then
; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
; CHECK-NEXT: .cfi_def_cfa_offset 16
@@ -58,9 +57,8 @@ exit:
define void @br_eq_i32_signbit(i32 %x) {
; CHECK-LABEL: br_eq_i32_signbit:
; CHECK: // %bb.0: // %entry
-; CHECK-NEXT: mov w8, #-2147483648 // =0x80000000
-; CHECK-NEXT: cmp w0, w8
-; CHECK-NEXT: b.ne .LBB2_2
+; CHECK-NEXT: cmp wzr, w0
+; CHECK-NEXT: b.vc .LBB2_2
; CHECK-NEXT: // %bb.1: // %then
; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
; CHECK-NEXT: .cfi_def_cfa_offset 16
@@ -131,26 +129,26 @@ exit:
define void @br_two_uses(i64 %x, i64 %y) {
; CHECK-LABEL: br_two_uses:
; CHECK: // %bb.0: // %entry
-; CHECK-NEXT: str x30, [sp, #-32]! // 8-byte Folded Spill
-; CHECK-NEXT: stp x20, x19, [sp, #16] // 16-byte Folded Spill
-; CHECK-NEXT: .cfi_def_cfa_offset 32
+; CHECK-NEXT: stp x30, x19, [sp, #-16]! // 16-byte Folded Spill
+; CHECK-NEXT: .cfi_def_cfa_offset 16
; CHECK-NEXT: .cfi_offset w19, -8
-; CHECK-NEXT: .cfi_offset w20, -16
-; CHECK-NEXT: .cfi_offset w30, -32
-; CHECK-NEXT: mov x20, #-9223372036854775808 // =0x8000000000000000
+; CHECK-NEXT: .cfi_offset w30, -16
+; CHECK-NEXT: cmp xzr, x0
; CHECK-NEXT: mov x19, x1
-; CHECK-NEXT: cmp x0, x20
-; CHECK-NEXT: b.ne .LBB5_2
-; CHECK-NEXT: // %bb.1: // %callg
+; CHECK-NEXT: b.vs .LBB5_3
+; CHECK-NEXT: // %bb.1: // %second
+; CHECK-NEXT: cmp xzr, x19
+; CHECK-NEXT: b.vs .LBB5_4
+; CHECK-NEXT: .LBB5_2: // %exit
+; CHECK-NEXT: ldp x30, x19, [sp], #16 // 16-byte Folded Reload
+; CHECK-NEXT: ret
+; CHECK-NEXT: .LBB5_3: // %callg
; CHECK-NEXT: bl g
-; CHECK-NEXT: .LBB5_2: // %second
-; CHECK-NEXT: cmp x19, x20
-; CHECK-NEXT: b.ne .LBB5_4
-; CHECK-NEXT: // %bb.3: // %callh
+; CHECK-NEXT: cmp xzr, x19
+; CHECK-NEXT: b.vc .LBB5_2
+; CHECK-NEXT: .LBB5_4: // %callh
; CHECK-NEXT: bl h
-; CHECK-NEXT: .LBB5_4: // %exit
-; CHECK-NEXT: ldp x20, x19, [sp, #16] // 16-byte Folded Reload
-; CHECK-NEXT: ldr x30, [sp], #32 // 8-byte Folded Reload
+; CHECK-NEXT: ldp x30, x19, [sp], #16 // 16-byte Folded Reload
; CHECK-NEXT: ret
entry:
%cx = icmp eq i64 %x, -9223372036854775808
@@ -385,12 +383,12 @@ define i64 @br_shared_with_select(i64 %x, i64 %y, i64 %a, i64 %b) {
; CHECK-NEXT: .cfi_offset w21, -24
; CHECK-NEXT: .cfi_offset w22, -32
; CHECK-NEXT: .cfi_offset w30, -48
-; CHECK-NEXT: mov x22, #-9223372036854775808 // =0x8000000000000000
; CHECK-NEXT: mov x19, x3
; CHECK-NEXT: mov x20, x2
-; CHECK-NEXT: cmp x0, x22
; CHECK-NEXT: mov x21, x1
-; CHECK-NEXT: b.ne .LBB13_2
+; CHECK-NEXT: cmp xzr, x0
+; CHECK-NEXT: mov x22, #-9223372036854775808 // =0x8000000000000000
+; CHECK-NEXT: b.vc .LBB13_2
; CHECK-NEXT: // %bb.1: // %then
; CHECK-NEXT: bl g
; CHECK-NEXT: .LBB13_2: // %exit
@@ -420,16 +418,14 @@ define void @br_loop(ptr %p, i64 %n) {
; CHECK-NEXT: cmp x1, #1
; CHECK-NEXT: b.lt .LBB14_6
; CHECK-NEXT: // %bb.1: // %loop.preheader
-; CHECK-NEXT: stp x30, x21, [sp, #-32]! // 16-byte Folded Spill
+; CHECK-NEXT: str x30, [sp, #-32]! // 8-byte Folded Spill
; CHECK-NEXT: stp x20, x19, [sp, #16] // 16-byte Folded Spill
; CHECK-NEXT: .cfi_def_cfa_offset 32
; CHECK-NEXT: .cfi_offset w19, -8
; CHECK-NEXT: .cfi_offset w20, -16
-; CHECK-NEXT: .cfi_offset w21, -24
; CHECK-NEXT: .cfi_offset w30, -32
; CHECK-NEXT: mov x19, x1
; CHECK-NEXT: mov x20, x0
-; CHECK-NEXT: mov x21, #-9223372036854775808 // =0x8000000000000000
; CHECK-NEXT: b .LBB14_3
; CHECK-NEXT: .LBB14_2: // %latch
; CHECK-NEXT: // in Loop: Header=BB14_3 Depth=1
@@ -438,15 +434,15 @@ define void @br_loop(ptr %p, i64 %n) {
; CHECK-NEXT: .LBB14_3: // %loop
; CHECK-NEXT: // =>This Inner Loop Header: Depth=1
; CHECK-NEXT: ldr x8, [x20], #8
-; CHECK-NEXT: cmp x8, x21
-; CHECK-NEXT: b.ne .LBB14_2
+; CHECK-NEXT: cmp xzr, x8
+; CHECK-NEXT: b.vc .LBB14_2
; CHECK-NEXT: // %bb.4: // %call
; CHECK-NEXT: // in Loop: Header=BB14_3 Depth=1
; CHECK-NEXT: bl g
; CHECK-NEXT: b .LBB14_2
; CHECK-NEXT: .LBB14_5:
; CHECK-NEXT: ldp x20, x19, [sp, #16] // 16-byte Folded Reload
-; CHECK-NEXT: ldp x30, x21, [sp], #32 // 16-byte Folded Reload
+; CHECK-NEXT: ldr x30, [sp], #32 // 8-byte Folded Reload
; CHECK-NEXT: .LBB14_6: // %exit
; CHECK-NEXT: ret
entry:
@@ -474,12 +470,11 @@ define void @br_eq_int64_min_slh(i64 %x) speculative_load_hardening {
; CHECK-LABEL: br_eq_int64_min_slh:
; CHECK: // %bb.0: // %entry
; CHECK-NEXT: cmp sp, #0
-; CHECK-NEXT: mov x8, #-9223372036854775808 // =0x8000000000000000
; CHECK-NEXT: csetm x16, ne
-; CHECK-NEXT: cmp x0, x8
-; CHECK-NEXT: b.ne .LBB15_3
+; CHECK-NEXT: cmp xzr, x0
+; CHECK-NEXT: b.vc .LBB15_3
; CHECK-NEXT: // %bb.1:
-; CHECK-NEXT: csel x16, x16, xzr, eq
+; CHECK-NEXT: csel x16, x16, xzr, vs
; CHECK-NEXT: // %bb.2: // %then
; CHECK-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
; CHECK-NEXT: .cfi_def_cfa_offset 16
@@ -495,7 +490,7 @@ define void @br_eq_int64_min_slh(i64 %x) speculative_load_hardening {
; CHECK-NEXT: csdb
; CHECK-NEXT: b .LBB15_4
; CHECK-NEXT: .LBB15_3:
-; CHECK-NEXT: csel x16, x16, xzr, ne
+; CHECK-NEXT: csel x16, x16, xzr, vc
; CHECK-NEXT: .LBB15_4: // %exit
; CHECK-NEXT: mov x0, sp
; CHECK-NEXT: and x0, x0, x16
More information about the llvm-commits
mailing list