[llvm] 0d79745 - [X86] Emit ADD instead of SHL by 1 when shrinking TEST with a mask (#217508)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Aug 26 05:17:21 PDT 2026
Author: Andrew Gaul
Date: 2026-08-26T12:17:16Z
New Revision: 0d79745f0a96a3d7adad6447ddbc345e6d0d288a
URL: https://github.com/llvm/llvm-project/commit/0d79745f0a96a3d7adad6447ddbc345e6d0d288a
DIFF: https://github.com/llvm/llvm-project/commit/0d79745f0a96a3d7adad6447ddbc345e6d0d288a.diff
LOG: [X86] Emit ADD instead of SHL by 1 when shrinking TEST with a mask (#217508)
The immediate-TEST shrink rewrites (and x, 0x7fffffffffffffff) == 0 into
SHL64ri $1 + TEST64rr, expecting the redundant TEST to be "subsequently
eliminated" (per the comment). For shift amounts 1-3 it never is:
isDefConvertible() rejects those SHLs so that they stay convertible to
LEA, and the dead TEST survives into final binaries.
Emit ADD64rr x, x instead when the shift amount is 1. Doubling is value-
and ZF-identical to the shift at the same encoding length, executes on
more ports, and ADDrr is def-convertible, so the peephole really does
fold the TEST away, leaving add+jcc/setcc instead of shl+test+jcc/setcc.
The shape is common: it is Rust libstd's panic-counter fast path
(GLOBAL_PANIC_COUNT & ~(1 << 63) == 0, inlined at every
std::thread::panicking() check -- 249 copies in uutils coreutils) and
LLVM's own is_fpclass zero-class lowering, as the is_fpclass.ll diff
shows.
Found via x86lint.
Assisted-by: Claude Code (Claude Fable 5)
---------
Co-authored-by: Claude Fable 5 <noreply at anthropic.com>
Co-authored-by: Simon Pilgrim <llvm-dev at redking.me.uk>
Added:
Modified:
llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
llvm/test/CodeGen/X86/cmp.ll
llvm/test/CodeGen/X86/is_fpclass.ll
Removed:
################################################################################
diff --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index c3fc43c2bc6db..f5bbe87a4de2f 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -6462,7 +6462,21 @@ void X86DAGToDAGISel::Select(SDNode *Node) {
} else if (TrailingZeros == 0 && SavesBytes) {
// If the mask covers the least significant bit, then we can replace
// TEST+AND with a SHL and check eflags.
- // This emits a redundant TEST which is subsequently eliminated.
+ // This emits a redundant TEST which is subsequently eliminated,
+ // except for shift amounts 1 to 3: isDefConvertible() rejects those
+ // SHLs to keep them convertible to LEA, so the TEST would survive.
+ if (LeadingZeros == 1) {
+ // Shift out the top bit by doubling with ADD reg,reg instead: it
+ // is the same length and sets ZF identically, but the peephole
+ // does fold the TEST into it, and it runs on more ports.
+ MachineSDNode *Add = CurDAG->getMachineNode(
+ GET_ND_IF_ENABLED(X86::ADD64rr), dl, MVT::i64, MVT::i32,
+ N0.getOperand(0), N0.getOperand(0));
+ MachineSDNode *Test = CurDAG->getMachineNode(
+ X86::TEST64rr, dl, MVT::i32, SDValue(Add, 0), SDValue(Add, 0));
+ ReplaceNode(Node, Test);
+ return;
+ }
ShiftOpcode = GET_ND_IF_ENABLED(X86::SHL64ri);
ShiftAmt = LeadingZeros;
SubRegIdx = 0;
diff --git a/llvm/test/CodeGen/X86/cmp.ll b/llvm/test/CodeGen/X86/cmp.ll
index ed3f0e0f0aa71..7b10610b06da9 100644
--- a/llvm/test/CodeGen/X86/cmp.ll
+++ b/llvm/test/CodeGen/X86/cmp.ll
@@ -675,6 +675,107 @@ define i32 @lowmask_i64_mask8(i64 %val) {
ret i32 %ret
}
+define i32 @lowmask_i64_mask63(i64 %val) {
+; NO-NDD-LABEL: lowmask_i64_mask63:
+; NO-NDD: # %bb.0:
+; NO-NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NO-NDD-NEXT: addq %rdi, %rdi # encoding: [0x48,0x01,0xff]
+; NO-NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NO-NDD-NEXT: retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask63:
+; NDD: # %bb.0:
+; NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NDD-NEXT: addq %rdi, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x01,0xff]
+; NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NDD-NEXT: retq # encoding: [0xc3]
+ %and = and i64 %val, 9223372036854775807
+ %cmp = icmp eq i64 %and, 0
+ %ret = zext i1 %cmp to i32
+ ret i32 %ret
+}
+
+define i64 @lowmask_i64_mask63_extra_use(i64 %val) nounwind {
+; NO-NDD-LABEL: lowmask_i64_mask63_extra_use:
+; NO-NDD: # %bb.0:
+; NO-NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NO-NDD-NEXT: movq %rdi, %rcx # encoding: [0x48,0x89,0xf9]
+; NO-NDD-NEXT: addq %rdi, %rcx # encoding: [0x48,0x01,0xf9]
+; NO-NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NO-NDD-NEXT: imulq %rdi, %rax # encoding: [0x48,0x0f,0xaf,0xc7]
+; NO-NDD-NEXT: retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask63_extra_use:
+; NDD: # %bb.0:
+; NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NDD-NEXT: addq %rdi, %rdi, %rcx # encoding: [0x62,0xf4,0xf4,0x18,0x01,0xff]
+; NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NDD-NEXT: imulq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0xaf,0xc7]
+; NDD-NEXT: retq # encoding: [0xc3]
+ %and = and i64 %val, 9223372036854775807
+ %cmp = icmp eq i64 %and, 0
+ %z = zext i1 %cmp to i64
+ %ret = mul i64 %z, %val
+ ret i64 %ret
+}
+
+define void @lowmask_i64_mask63_br(i64 %val) nounwind {
+; NO-NDD-LABEL: lowmask_i64_mask63_br:
+; NO-NDD: # %bb.0: # %entry
+; NO-NDD-NEXT: addq %rdi, %rdi # encoding: [0x48,0x01,0xff]
+; NO-NDD-NEXT: je .LBB34_1 # encoding: [0x74,A]
+; NO-NDD-NEXT: # fixup A - offset: 1, value: .LBB34_1, kind: FK_PCRel_1
+; NO-NDD-NEXT: # %bb.2: # %f
+; NO-NDD-NEXT: retq # encoding: [0xc3]
+; NO-NDD-NEXT: .LBB34_1: # %t
+; NO-NDD-NEXT: movq $1, d64(%rip) # encoding: [0x48,0xc7,0x05,A,A,A,A,0x01,0x00,0x00,0x00]
+; NO-NDD-NEXT: # fixup A - offset: 3, value: d64-4, kind: reloc_riprel_4byte
+; NO-NDD-NEXT: retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask63_br:
+; NDD: # %bb.0: # %entry
+; NDD-NEXT: addq %rdi, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x01,0xff]
+; NDD-NEXT: je .LBB34_1 # encoding: [0x74,A]
+; NDD-NEXT: # fixup A - offset: 1, value: .LBB34_1, kind: FK_PCRel_1
+; NDD-NEXT: # %bb.2: # %f
+; NDD-NEXT: retq # encoding: [0xc3]
+; NDD-NEXT: .LBB34_1: # %t
+; NDD-NEXT: movq $1, d64(%rip) # encoding: [0x48,0xc7,0x05,A,A,A,A,0x01,0x00,0x00,0x00]
+; NDD-NEXT: # fixup A - offset: 3, value: d64-4, kind: reloc_riprel_4byte
+; NDD-NEXT: retq # encoding: [0xc3]
+entry:
+ %and = and i64 %val, 9223372036854775807
+ %cmp = icmp eq i64 %and, 0
+ br i1 %cmp, label %t, label %f
+t:
+ store i64 1, ptr @d64
+ br label %f
+f:
+ ret void
+}
+
+define i32 @lowmask_i64_mask62(i64 %val) {
+; NO-NDD-LABEL: lowmask_i64_mask62:
+; NO-NDD: # %bb.0:
+; NO-NDD-NEXT: shlq $2, %rdi # encoding: [0x48,0xc1,0xe7,0x02]
+; NO-NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NO-NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NO-NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NO-NDD-NEXT: retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask62:
+; NDD: # %bb.0:
+; NDD-NEXT: shlq $2, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0xc1,0xe7,0x02]
+; NDD-NEXT: xorl %eax, %eax # encoding: [0x31,0xc0]
+; NDD-NEXT: testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NDD-NEXT: sete %al # encoding: [0x0f,0x94,0xc0]
+; NDD-NEXT: retq # encoding: [0xc3]
+ %and = and i64 %val, 4611686018427387903
+ %cmp = icmp eq i64 %and, 0
+ %ret = zext i1 %cmp to i32
+ ret i32 %ret
+}
+
define i32 @highmask_i32_mask32(i32 %val) {
; CHECK-LABEL: highmask_i32_mask32:
; CHECK: # %bb.0:
diff --git a/llvm/test/CodeGen/X86/is_fpclass.ll b/llvm/test/CodeGen/X86/is_fpclass.ll
index 9b9732f433ace..ac9b9afb90ad5 100644
--- a/llvm/test/CodeGen/X86/is_fpclass.ll
+++ b/llvm/test/CodeGen/X86/is_fpclass.ll
@@ -1260,8 +1260,7 @@ define i1 @iszero_d(double %x) {
; X64-LABEL: iszero_d:
; X64: # %bb.0: # %entry
; X64-NEXT: movq %xmm0, %rax
-; X64-NEXT: shlq %rax
-; X64-NEXT: testq %rax, %rax
+; X64-NEXT: addq %rax, %rax
; X64-NEXT: sete %al
; X64-NEXT: retq
entry:
@@ -1397,8 +1396,7 @@ define i1 @iszero_d_strictfp(double %x) strictfp {
; X64-LABEL: iszero_d_strictfp:
; X64: # %bb.0: # %entry
; X64-NEXT: movq %xmm0, %rax
-; X64-NEXT: shlq %rax
-; X64-NEXT: testq %rax, %rax
+; X64-NEXT: addq %rax, %rax
; X64-NEXT: sete %al
; X64-NEXT: retq
entry:
More information about the llvm-commits
mailing list