[llvm] 0d79745 - [X86] Emit ADD instead of SHL by 1 when shrinking TEST with a mask (#217508)

via llvm-commits llvm-commits at lists.llvm.org
Wed Aug 26 05:17:21 PDT 2026


Author: Andrew Gaul
Date: 2026-08-26T12:17:16Z
New Revision: 0d79745f0a96a3d7adad6447ddbc345e6d0d288a

URL: https://github.com/llvm/llvm-project/commit/0d79745f0a96a3d7adad6447ddbc345e6d0d288a
DIFF: https://github.com/llvm/llvm-project/commit/0d79745f0a96a3d7adad6447ddbc345e6d0d288a.diff

LOG: [X86] Emit ADD instead of SHL by 1 when shrinking TEST with a mask (#217508)

The immediate-TEST shrink rewrites (and x, 0x7fffffffffffffff) == 0 into
SHL64ri $1 + TEST64rr, expecting the redundant TEST to be "subsequently
eliminated" (per the comment). For shift amounts 1-3 it never is:
isDefConvertible() rejects those SHLs so that they stay convertible to
LEA, and the dead TEST survives into final binaries.

Emit ADD64rr x, x instead when the shift amount is 1. Doubling is value-
and ZF-identical to the shift at the same encoding length, executes on
more ports, and ADDrr is def-convertible, so the peephole really does
fold the TEST away, leaving add+jcc/setcc instead of shl+test+jcc/setcc.

The shape is common: it is Rust libstd's panic-counter fast path
(GLOBAL_PANIC_COUNT & ~(1 << 63) == 0, inlined at every
std::thread::panicking() check -- 249 copies in uutils coreutils) and
LLVM's own is_fpclass zero-class lowering, as the is_fpclass.ll diff
shows.

Found via x86lint.

Assisted-by: Claude Code (Claude Fable 5)

---------

Co-authored-by: Claude Fable 5 <noreply at anthropic.com>
Co-authored-by: Simon Pilgrim <llvm-dev at redking.me.uk>

Added: 
    

Modified: 
    llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
    llvm/test/CodeGen/X86/cmp.ll
    llvm/test/CodeGen/X86/is_fpclass.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index c3fc43c2bc6db..f5bbe87a4de2f 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -6462,7 +6462,21 @@ void X86DAGToDAGISel::Select(SDNode *Node) {
         } else if (TrailingZeros == 0 && SavesBytes) {
           // If the mask covers the least significant bit, then we can replace
           // TEST+AND with a SHL and check eflags.
-          // This emits a redundant TEST which is subsequently eliminated.
+          // This emits a redundant TEST which is subsequently eliminated,
+          // except for shift amounts 1 to 3: isDefConvertible() rejects those
+          // SHLs to keep them convertible to LEA, so the TEST would survive.
+          if (LeadingZeros == 1) {
+            // Shift out the top bit by doubling with ADD reg,reg instead: it
+            // is the same length and sets ZF identically, but the peephole
+            // does fold the TEST into it, and it runs on more ports.
+            MachineSDNode *Add = CurDAG->getMachineNode(
+                GET_ND_IF_ENABLED(X86::ADD64rr), dl, MVT::i64, MVT::i32,
+                N0.getOperand(0), N0.getOperand(0));
+            MachineSDNode *Test = CurDAG->getMachineNode(
+                X86::TEST64rr, dl, MVT::i32, SDValue(Add, 0), SDValue(Add, 0));
+            ReplaceNode(Node, Test);
+            return;
+          }
           ShiftOpcode = GET_ND_IF_ENABLED(X86::SHL64ri);
           ShiftAmt = LeadingZeros;
           SubRegIdx = 0;

diff  --git a/llvm/test/CodeGen/X86/cmp.ll b/llvm/test/CodeGen/X86/cmp.ll
index ed3f0e0f0aa71..7b10610b06da9 100644
--- a/llvm/test/CodeGen/X86/cmp.ll
+++ b/llvm/test/CodeGen/X86/cmp.ll
@@ -675,6 +675,107 @@ define i32 @lowmask_i64_mask8(i64 %val) {
   ret i32 %ret
 }
 
+define i32 @lowmask_i64_mask63(i64 %val) {
+; NO-NDD-LABEL: lowmask_i64_mask63:
+; NO-NDD:       # %bb.0:
+; NO-NDD-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
+; NO-NDD-NEXT:    addq %rdi, %rdi # encoding: [0x48,0x01,0xff]
+; NO-NDD-NEXT:    sete %al # encoding: [0x0f,0x94,0xc0]
+; NO-NDD-NEXT:    retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask63:
+; NDD:       # %bb.0:
+; NDD-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
+; NDD-NEXT:    addq %rdi, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x01,0xff]
+; NDD-NEXT:    sete %al # encoding: [0x0f,0x94,0xc0]
+; NDD-NEXT:    retq # encoding: [0xc3]
+  %and = and i64 %val, 9223372036854775807
+  %cmp = icmp eq i64 %and, 0
+  %ret = zext i1 %cmp to i32
+  ret i32 %ret
+}
+
+define i64 @lowmask_i64_mask63_extra_use(i64 %val) nounwind {
+; NO-NDD-LABEL: lowmask_i64_mask63_extra_use:
+; NO-NDD:       # %bb.0:
+; NO-NDD-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
+; NO-NDD-NEXT:    movq %rdi, %rcx # encoding: [0x48,0x89,0xf9]
+; NO-NDD-NEXT:    addq %rdi, %rcx # encoding: [0x48,0x01,0xf9]
+; NO-NDD-NEXT:    sete %al # encoding: [0x0f,0x94,0xc0]
+; NO-NDD-NEXT:    imulq %rdi, %rax # encoding: [0x48,0x0f,0xaf,0xc7]
+; NO-NDD-NEXT:    retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask63_extra_use:
+; NDD:       # %bb.0:
+; NDD-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
+; NDD-NEXT:    addq %rdi, %rdi, %rcx # encoding: [0x62,0xf4,0xf4,0x18,0x01,0xff]
+; NDD-NEXT:    sete %al # encoding: [0x0f,0x94,0xc0]
+; NDD-NEXT:    imulq %rdi, %rax # EVEX TO LEGACY Compression encoding: [0x48,0x0f,0xaf,0xc7]
+; NDD-NEXT:    retq # encoding: [0xc3]
+  %and = and i64 %val, 9223372036854775807
+  %cmp = icmp eq i64 %and, 0
+  %z = zext i1 %cmp to i64
+  %ret = mul i64 %z, %val
+  ret i64 %ret
+}
+
+define void @lowmask_i64_mask63_br(i64 %val) nounwind {
+; NO-NDD-LABEL: lowmask_i64_mask63_br:
+; NO-NDD:       # %bb.0: # %entry
+; NO-NDD-NEXT:    addq %rdi, %rdi # encoding: [0x48,0x01,0xff]
+; NO-NDD-NEXT:    je .LBB34_1 # encoding: [0x74,A]
+; NO-NDD-NEXT:    # fixup A - offset: 1, value: .LBB34_1, kind: FK_PCRel_1
+; NO-NDD-NEXT:  # %bb.2: # %f
+; NO-NDD-NEXT:    retq # encoding: [0xc3]
+; NO-NDD-NEXT:  .LBB34_1: # %t
+; NO-NDD-NEXT:    movq $1, d64(%rip) # encoding: [0x48,0xc7,0x05,A,A,A,A,0x01,0x00,0x00,0x00]
+; NO-NDD-NEXT:    # fixup A - offset: 3, value: d64-4, kind: reloc_riprel_4byte
+; NO-NDD-NEXT:    retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask63_br:
+; NDD:       # %bb.0: # %entry
+; NDD-NEXT:    addq %rdi, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0x01,0xff]
+; NDD-NEXT:    je .LBB34_1 # encoding: [0x74,A]
+; NDD-NEXT:    # fixup A - offset: 1, value: .LBB34_1, kind: FK_PCRel_1
+; NDD-NEXT:  # %bb.2: # %f
+; NDD-NEXT:    retq # encoding: [0xc3]
+; NDD-NEXT:  .LBB34_1: # %t
+; NDD-NEXT:    movq $1, d64(%rip) # encoding: [0x48,0xc7,0x05,A,A,A,A,0x01,0x00,0x00,0x00]
+; NDD-NEXT:    # fixup A - offset: 3, value: d64-4, kind: reloc_riprel_4byte
+; NDD-NEXT:    retq # encoding: [0xc3]
+entry:
+  %and = and i64 %val, 9223372036854775807
+  %cmp = icmp eq i64 %and, 0
+  br i1 %cmp, label %t, label %f
+t:
+  store i64 1, ptr @d64
+  br label %f
+f:
+  ret void
+}
+
+define i32 @lowmask_i64_mask62(i64 %val) {
+; NO-NDD-LABEL: lowmask_i64_mask62:
+; NO-NDD:       # %bb.0:
+; NO-NDD-NEXT:    shlq $2, %rdi # encoding: [0x48,0xc1,0xe7,0x02]
+; NO-NDD-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
+; NO-NDD-NEXT:    testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NO-NDD-NEXT:    sete %al # encoding: [0x0f,0x94,0xc0]
+; NO-NDD-NEXT:    retq # encoding: [0xc3]
+;
+; NDD-LABEL: lowmask_i64_mask62:
+; NDD:       # %bb.0:
+; NDD-NEXT:    shlq $2, %rdi # EVEX TO LEGACY Compression encoding: [0x48,0xc1,0xe7,0x02]
+; NDD-NEXT:    xorl %eax, %eax # encoding: [0x31,0xc0]
+; NDD-NEXT:    testq %rdi, %rdi # encoding: [0x48,0x85,0xff]
+; NDD-NEXT:    sete %al # encoding: [0x0f,0x94,0xc0]
+; NDD-NEXT:    retq # encoding: [0xc3]
+  %and = and i64 %val, 4611686018427387903
+  %cmp = icmp eq i64 %and, 0
+  %ret = zext i1 %cmp to i32
+  ret i32 %ret
+}
+
 define i32 @highmask_i32_mask32(i32 %val) {
 ; CHECK-LABEL: highmask_i32_mask32:
 ; CHECK:       # %bb.0:

diff  --git a/llvm/test/CodeGen/X86/is_fpclass.ll b/llvm/test/CodeGen/X86/is_fpclass.ll
index 9b9732f433ace..ac9b9afb90ad5 100644
--- a/llvm/test/CodeGen/X86/is_fpclass.ll
+++ b/llvm/test/CodeGen/X86/is_fpclass.ll
@@ -1260,8 +1260,7 @@ define i1 @iszero_d(double %x) {
 ; X64-LABEL: iszero_d:
 ; X64:       # %bb.0: # %entry
 ; X64-NEXT:    movq %xmm0, %rax
-; X64-NEXT:    shlq %rax
-; X64-NEXT:    testq %rax, %rax
+; X64-NEXT:    addq %rax, %rax
 ; X64-NEXT:    sete %al
 ; X64-NEXT:    retq
 entry:
@@ -1397,8 +1396,7 @@ define i1 @iszero_d_strictfp(double %x) strictfp {
 ; X64-LABEL: iszero_d_strictfp:
 ; X64:       # %bb.0: # %entry
 ; X64-NEXT:    movq %xmm0, %rax
-; X64-NEXT:    shlq %rax
-; X64-NEXT:    testq %rax, %rax
+; X64-NEXT:    addq %rax, %rax
 ; X64-NEXT:    sete %al
 ; X64-NEXT:    retq
 entry:


        


More information about the llvm-commits mailing list