[llvm] [X86] Emit ADD instead of SHL by 1 when shrinking TEST with a mask (PR #217508)
Simon Pilgrim via llvm-commits
llvm-commits at lists.llvm.org
Wed Aug 26 03:19:03 PDT 2026
https://github.com/RKSimon updated https://github.com/llvm/llvm-project/pull/217508
>From 83743fb17adcd68d5fa752e62c8a7156301a80da Mon Sep 17 00:00:00 2001
From: Andrew Gaul <andrew at gaul.org>
Date: Thu, 20 Aug 2026 09:19:35 -0700
Subject: [PATCH 1/2] [X86] cmp.ll - add test coverage for 63-bit lowmask
compares
Precommit the tests for #217508 with current codegen: the immediate-TEST
shrink turns (and x, 0x7fffffffffffffff) == 0 into shl-by-1 plus a TEST
that the peephole never eliminates (isDefConvertible rejects
LEA-convertible shift amounts 1-3). Also cover a 62-bit mask (shift
amount 2, same problem but outside the planned fix) and an extra use of
the shifted value.
Co-Authored-By: Claude Fable 5 <noreply at anthropic.com>
---
llvm/test/CodeGen/X86/cmp.ll | 106 +++++++++++++++++++++++++++++++++++
1 file changed, 106 insertions(+)
diff --git a/llvm/test/CodeGen/X86/cmp.ll b/llvm/test/CodeGen/X86/cmp.ll
index ed3f0e0f0aa71..cf87891552762 100644
--- a/llvm/test/CodeGen/X86/cmp.ll
+++ b/llvm/test/CodeGen/X86/cmp.ll
@@ -675,6 +675,112 @@ define i32 @lowmask_i64_mask8(i64 %val) {
ret i32 %ret
}
+define i32 @lowmask_i64_mask63(i64 %val) {
+; NO-NDD-LABEL: lowmask_i64_mask63:
+; NO-NDD: # %bb.0:
+; NO-NDD-NEXT: shlq %rdi # encoding: [0x48,0xd1,0xe7]
+; NO-NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NO-NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NO-NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NO-NDD-NEXT: retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask63:
+; NDD: # %bb.0:
+; NDD-NEXT: shlq %rdi # EVEX TO LEGACY Compression encoding: [0x48,0xd1,0xe7]
+; NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NDD-NEXT: retq # encoding: [0xc3]
+ %and = and i64 %val, 9223372036854775807
+ %cmp = icmp eq i64 %and, 0
+ %ret = zext i1 %cmp to i32
+ ret i32 %ret
+}
+
+define i64 @lowmask_i64_mask63_extra_use(i64 %val) nounwind {
+; NO-NDD-LABEL: lowmask_i64_mask63_extra_use:
+; NO-NDD: # %bb.0:
+; NO-NDD-NEXT: leaq (,%rdi,2), %rcx # encoding: [0x48,0x8d,0x0c,0x7d,0x00,0x00,0x00,0x00]
+; NO-NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NO-NDD-NEXT: testq %rcx, %rcx # encoding: [0x48,0x85,0xc9]
+; NO-NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NO-NDD-NEXT: imulq %rdi, %rax # encoding: [0x48,0x0f,0xaf,0xc7]
+; NO-NDD-NEXT: retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask63_extra_use:
+; NDD: # %bb.0:
+; NDD-NEXT: shlq %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0xd1,0xe7]
+; NDD-NEXT: xorl %ecx, %ecx # encoding: [0x31,0xc9]
+; NDD-NEXT: testq %rax, %rax # encoding: [0x48,0x85,0xc0]
+; NDD-NEXT: sete %cl # encoding: [0x0f,0x94,0xc1]
+; NDD-NEXT: imulq %rdi, %rcx, %rax # encoding: [0x62,0xf4,0xfc,0x18,0xaf,0xcf]
+; NDD-NEXT: retq # encoding: [0xc3]
+ %and = and i64 %val, 9223372036854775807
+ %cmp = icmp eq i64 %and, 0
+ %z = zext i1 %cmp to i64
+ %ret = mul i64 %z, %val
+ ret i64 %ret
+}
+
+define void @lowmask_i64_mask63_br(i64 %val) nounwind {
+; NO-NDD-LABEL: lowmask_i64_mask63_br:
+; NO-NDD: # %bb.0: # %entry
+; NO-NDD-NEXT: shlq %rdi # encoding: [0x48,0xd1,0xe7]
+; NO-NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NO-NDD-NEXT: je .LBB34_1 # encoding: [0x74,A]
+; NO-NDD-NEXT: # fixup A - offset: 1, value: .LBB34_1, kind: FK_PCRel_1
+; NO-NDD-NEXT: # %bb.2: # %f
+; NO-NDD-NEXT: retq # encoding: [0xc3]
+; NO-NDD-NEXT: .LBB34_1: # %t
+; NO-NDD-NEXT: movq $1, d64(%rip) # encoding: [0x48,0xc7,0x05,A,A,A,A,0x01,0x00,0x00,0x00]
+; NO-NDD-NEXT: # fixup A - offset: 3, value: d64-4, kind: reloc_riprel_4byte
+; NO-NDD-NEXT: retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask63_br:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: shlq %rdi # EVEX TO LEGACY Compression encoding: [0x48,0xd1,0xe7]
+; NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NDD-NEXT: je .LBB34_1 # encoding: [0x74,A]
+; NDD-NEXT: # fixup A - offset: 1, value: .LBB34_1, kind: FK_PCRel_1
+; NDD-NEXT: # %bb.2: # %f
+; NDD-NEXT: retq # encoding: [0xc3]
+; NDD-NEXT: .LBB34_1: # %t
+; NDD-NEXT: movq $1, d64(%rip) # encoding: [0x48,0xc7,0x05,A,A,A,A,0x01,0x00,0x00,0x00]
+; NDD-NEXT: # fixup A - offset: 3, value: d64-4, kind: reloc_riprel_4byte
+; NDD-NEXT: retq # encoding: [0xc3]
+entry:
+ %and = and i64 %val, 9223372036854775807
+ %cmp = icmp eq i64 %and, 0
+ br i1 %cmp, label %t, label %f
+t:
+ store i64 1, ptr @d64
+ br label %f
+f:
+ ret void
+}
+
+define i32 @lowmask_i64_mask62(i64 %val) {
+; NO-NDD-LABEL: lowmask_i64_mask62:
+; NO-NDD: # %bb.0:
+; NO-NDD-NEXT: shlq $2, %rdi # encoding: [0x48,0xc1,0xe7,0x02]
+; NO-NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NO-NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NO-NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NO-NDD-NEXT: retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask62:
+; NDD: # %bb.0:
+; NDD-NEXT: shlq $2, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0xc1,0xe7,0x02]
+; NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NDD-NEXT: retq # encoding: [0xc3]
+ %and = and i64 %val, 4611686018427387903
+ %cmp = icmp eq i64 %and, 0
+ %ret = zext i1 %cmp to i32
+ ret i32 %ret
+}
+
define i32 @highmask_i32_mask32(i32 %val) {
; CHECK-LABEL: highmask_i32_mask32:
; CHECK: # %bb.0:
>From 52b7667ac2815bdcf7766896474bfaf16076c4b3 Mon Sep 17 00:00:00 2001
From: Andrew Gaul <andrew at gaul.org>
Date: Tue, 18 Aug 2026 00:46:38 -0700
Subject: [PATCH 2/2] [X86] Emit ADD instead of SHL by 1 when shrinking TEST
with a mask
The immediate-TEST shrink rewrites (and x, 0x7fffffffffffffff) == 0
into SHL64ri $1 + TEST64rr, expecting the redundant TEST to be
"subsequently eliminated" (per the comment). For shift amounts 1-3 it
never is: isDefConvertible() rejects those SHLs so that they stay
convertible to LEA, and the dead TEST survives into final binaries.
Emit ADD64rr x, x instead when the shift amount is 1. Doubling is
value- and ZF-identical to the shift at the same encoding length,
executes on more ports, and ADDrr is def-convertible, so the peephole
really does fold the TEST away, leaving add+jcc/setcc instead of
shl+test+jcc/setcc.
The shape is common: it is Rust libstd's panic-counter fast path
((GLOBAL_PANIC_COUNT & ~(1 << 63)) == 0, inlined at every
std::thread::panicking() check -- 249 copies in uutils coreutils) and
LLVM's own is_fpclass zero-class lowering, as the is_fpclass.ll diff
shows.
Found via x86lint.
Co-Authored-By: Claude Fable 5 <noreply at anthropic.com>
---
llvm/lib/Target/X86/X86ISelDAGToDAG.cpp | 16 +++++++++++++++-
llvm/test/CodeGen/X86/cmp.ll | 25 ++++++++++---------------
llvm/test/CodeGen/X86/is_fpclass.ll | 6 ++----
3 files changed, 27 insertions(+), 20 deletions(-)
diff --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index c3fc43c2bc6db..f5bbe87a4de2f 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -6462,7 +6462,21 @@ void X86DAGToDAGISel::Select(SDNode *Node) {
} else if (TrailingZeros == 0 && SavesBytes) {
// If the mask covers the least significant bit, then we can replace
// TEST+AND with a SHL and check eflags.
- // This emits a redundant TEST which is subsequently eliminated.
+ // This emits a redundant TEST which is subsequently eliminated,
+ // except for shift amounts 1 to 3: isDefConvertible() rejects those
+ // SHLs to keep them convertible to LEA, so the TEST would survive.
+ if (LeadingZeros == 1) {
+ // Shift out the top bit by doubling with ADD reg,reg instead: it
+ // is the same length and sets ZF identically, but the peephole
+ // does fold the TEST into it, and it runs on more ports.
+ MachineSDNode *Add = CurDAG->getMachineNode(
+ GET_ND_IF_ENABLED(X86::ADD64rr), dl, MVT::i64, MVT::i32,
+ N0.getOperand(0), N0.getOperand(0));
+ MachineSDNode *Test = CurDAG->getMachineNode(
+ X86::TEST64rr, dl, MVT::i32, SDValue(Add, 0), SDValue(Add, 0));
+ ReplaceNode(Node, Test);
+ return;
+ }
ShiftOpcode = GET_ND_IF_ENABLED(X86::SHL64ri);
ShiftAmt = LeadingZeros;
SubRegIdx = 0;
diff --git a/llvm/test/CodeGen/X86/cmp.ll b/llvm/test/CodeGen/X86/cmp.ll
index cf87891552762..7b10610b06da9 100644
--- a/llvm/test/CodeGen/X86/cmp.ll
+++ b/llvm/test/CodeGen/X86/cmp.ll
@@ -678,17 +678,15 @@ define i32 @lowmask_i64_mask8(i64 %val) {
define i32 @lowmask_i64_mask63(i64 %val) {
; NO-NDD-LABEL: lowmask_i64_mask63:
; NO-NDD: # %bb.0:
-; NO-NDD-NEXT: shlq %rdi # encoding: [0x48,0xd1,0xe7]
; NO-NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NO-NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NO-NDD-NEXT: addq %rdi, %rdi # encoding: [0x48,0x01,0xff]
; NO-NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
; NO-NDD-NEXT: retq # encoding: [0xc3]
;
; NDD-LABEL: lowmask_i64_mask63:
; NDD: # %bb.0:
-; NDD-NEXT: shlq %rdi # EVEX TO LEGACY Compression encoding: [0x48,0xd1,0xe7]
; NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NDD-NEXT: addq %rdi, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x01,0xff]
; NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
; NDD-NEXT: retq # encoding: [0xc3]
%and = and i64 %val, 9223372036854775807
@@ -700,20 +698,19 @@ define i32 @lowmask_i64_mask63(i64 %val) {
define i64 @lowmask_i64_mask63_extra_use(i64 %val) nounwind {
; NO-NDD-LABEL: lowmask_i64_mask63_extra_use:
; NO-NDD: # %bb.0:
-; NO-NDD-NEXT: leaq (,%rdi,2), %rcx # encoding: [0x48,0x8d,0x0c,0x7d,0x00,0x00,0x00,0x00]
; NO-NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
-; NO-NDD-NEXT: testq %rcx, %rcx # encoding: [0x48,0x85,0xc9]
+; NO-NDD-NEXT: movq %rdi, %rcx # encoding: [0x48,0x89,0xf9]
+; NO-NDD-NEXT: addq %rdi, %rcx # encoding: [0x48,0x01,0xf9]
; NO-NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
; NO-NDD-NEXT: imulq %rdi, %rax # encoding: [0x48,0x0f,0xaf,0xc7]
; NO-NDD-NEXT: retq # encoding: [0xc3]
;
; NDD-LABEL: lowmask_i64_mask63_extra_use:
; NDD: # %bb.0:
-; NDD-NEXT: shlq %rdi, %rax # encoding: [0x62,0xf4,0xfc,0x18,0xd1,0xe7]
-; NDD-NEXT: xorl %ecx, %ecx # encoding: [0x31,0xc9]
-; NDD-NEXT: testq %rax, %rax # encoding: [0x48,0x85,0xc0]
-; NDD-NEXT: sete %cl # encoding: [0x0f,0x94,0xc1]
-; NDD-NEXT: imulq %rdi, %rcx, %rax # encoding: [0x62,0xf4,0xfc,0x18,0xaf,0xcf]
+; NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NDD-NEXT: addq %rdi, %rdi, %rcx # encoding: [0x62,0xf4,0xf4,0x18,0x01,0xff]
+; NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NDD-NEXT: imulq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0xaf,0xc7]
; NDD-NEXT: retq # encoding: [0xc3]
%and = and i64 %val, 9223372036854775807
%cmp = icmp eq i64 %and, 0
@@ -725,8 +722,7 @@ define i64 @lowmask_i64_mask63_extra_use(i64 %val) nounwind {
define void @lowmask_i64_mask63_br(i64 %val) nounwind {
; NO-NDD-LABEL: lowmask_i64_mask63_br:
; NO-NDD: # %bb.0: # %entry
-; NO-NDD-NEXT: shlq %rdi # encoding: [0x48,0xd1,0xe7]
-; NO-NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NO-NDD-NEXT: addq %rdi, %rdi # encoding: [0x48,0x01,0xff]
; NO-NDD-NEXT: je .LBB34_1 # encoding: [0x74,A]
; NO-NDD-NEXT: # fixup A - offset: 1, value: .LBB34_1, kind: FK_PCRel_1
; NO-NDD-NEXT: # %bb.2: # %f
@@ -738,8 +734,7 @@ define void @lowmask_i64_mask63_br(i64 %val) nounwind {
;
; NDD-LABEL: lowmask_i64_mask63_br:
; NDD: # %bb.0: # %entry
-; NDD-NEXT: shlq %rdi # EVEX TO LEGACY Compression encoding: [0x48,0xd1,0xe7]
-; NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NDD-NEXT: addq %rdi, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x01,0xff]
; NDD-NEXT: je .LBB34_1 # encoding: [0x74,A]
; NDD-NEXT: # fixup A - offset: 1, value: .LBB34_1, kind: FK_PCRel_1
; NDD-NEXT: # %bb.2: # %f
diff --git a/llvm/test/CodeGen/X86/is_fpclass.ll b/llvm/test/CodeGen/X86/is_fpclass.ll
index 9b9732f433ace..ac9b9afb90ad5 100644
--- a/llvm/test/CodeGen/X86/is_fpclass.ll
+++ b/llvm/test/CodeGen/X86/is_fpclass.ll
@@ -1260,8 +1260,7 @@ define i1 @iszero_d(double %x) {
; X64-LABEL: iszero_d:
; X64: # %bb.0: # %entry
; X64-NEXT: movq %xmm0, %rax
-; X64-NEXT: shlq %rax
-; X64-NEXT: testq %rax, %rax
+; X64-NEXT: addq %rax, %rax
; X64-NEXT: sete %al
; X64-NEXT: retq
entry:
@@ -1397,8 +1396,7 @@ define i1 @iszero_d_strictfp(double %x) strictfp {
; X64-LABEL: iszero_d_strictfp:
; X64: # %bb.0: # %entry
; X64-NEXT: movq %xmm0, %rax
-; X64-NEXT: shlq %rax
-; X64-NEXT: testq %rax, %rax
+; X64-NEXT: addq %rax, %rax
; X64-NEXT: sete %al
; X64-NEXT: retq
entry:
More information about the llvm-commits
mailing list