[llvm] [X86] Select standalone ~(-1 << n) masks as BZHI (PR #226158)

via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 06:21:14 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-x86

Author: Chris Kennelly (ckennelly)

<details>
<summary>Changes</summary>

InstCombine canonicalizes (1 << n) - 1 to xor (shl -1, n), -1. When that mask feeds an AND, X86DAGToDAGISel::matchBitExtract folds the whole expression into BZHI, and the add form is selected as BZHI even on its own because Select enters matchBitExtract from ISD::ADD. The xor form on its own is not: a mask that is returned, stored, or consumed by anything but AND is selected as mov -1; shlx; not.

Enter matchBitExtract from ISD::XOR too under BMI2, so the standalone mask becomes mov -1; bzhi. That is one instruction shorter and avoids the shlx+not dependency chain. BMI1-only targets keep the shift form, since BEXTR would need the count moved into bits 15:8 first.

Assisted-by: Claude Code

---
Full diff: https://github.com/llvm/llvm-project/pull/226158.diff


2 Files Affected:

- (modified) llvm/lib/Target/X86/X86ISelDAGToDAG.cpp (+9-4) 
- (added) llvm/test/CodeGen/X86/bzhi-standalone-mask.ll (+260) 


``````````diff
diff --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index c2de8283f15ff..f272fe63ade4a 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -4146,11 +4146,12 @@ bool X86DAGToDAGISel::foldLoadStoreIntoMemOperand(SDNode *Node) {
 //   c) x &  (-1 >> (32 - y))
 //   d) x << (32 - y) >> (32 - y)
 //   e) (1 << nbits) - 1
+//   f) ~(-1 << nbits)
 bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
-  assert(
-      (Node->getOpcode() == ISD::ADD || Node->getOpcode() == ISD::AND ||
-       Node->getOpcode() == ISD::SRL) &&
-      "Should be either an and-mask, or right-shift after clearing high bits.");
+  assert((Node->getOpcode() == ISD::ADD || Node->getOpcode() == ISD::AND ||
+          Node->getOpcode() == ISD::XOR || Node->getOpcode() == ISD::SRL) &&
+         "Should be either an and-mask, a standalone low-bits mask, or "
+         "right-shift after clearing high bits.");
 
   // BEXTR is BMI instruction, BZHI is BMI2 instruction. We need at least one.
   if (!Subtarget->hasBMI() && !Subtarget->hasBMI2())
@@ -5818,6 +5819,10 @@ void X86DAGToDAGISel::Select(SDNode *Node) {
     [[fallthrough]];
   case ISD::OR:
   case ISD::XOR:
+    // A standalone ~(-1 << n) mask is (-1 & lowmask(n)): mov -1; bzhi beats
+    // mov -1; shlx; not.
+    if (Opcode == ISD::XOR && Subtarget->hasBMI2() && matchBitExtract(Node))
+      return;
     if (tryShrinkShlLogicImm(Node))
       return;
     if (Opcode == ISD::OR && tryMatchBitSelect(Node))
diff --git a/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll b/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll
new file mode 100644
index 0000000000000..950277d4d41d9
--- /dev/null
+++ b/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll
@@ -0,0 +1,260 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=i686-unknown-linux-gnu -mattr=+bmi,+bmi2 | FileCheck %s --check-prefixes=X86
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=-bmi,-bmi2 | FileCheck %s --check-prefixes=X64,X64-NOBMI
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+bmi,-bmi2 | FileCheck %s --check-prefixes=X64,X64-BMI1
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+bmi,+bmi2 | FileCheck %s --check-prefixes=X64,X64-BMI2
+
+; A low-bits mask that is not immediately consumed by an `and`. InstCombine
+; canonicalizes (1 << n) - 1 to ~(-1 << n), and with BMI2 that is
+; bzhi(-1, n): one instruction instead of shlx + not.
+
+define i32 @mask32(i32 %n) nounwind {
+; X86-LABEL: mask32:
+; X86:       # %bb.0:
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    movl $-1, %ecx
+; X86-NEXT:    bzhil %eax, %ecx, %eax
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: mask32:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movl %edi, %ecx
+; X64-NOBMI-NEXT:    movl $-1, %eax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT:    shll %cl, %eax
+; X64-NOBMI-NEXT:    notl %eax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: mask32:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movl %edi, %ecx
+; X64-BMI1-NEXT:    movl $-1, %eax
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT:    shll %cl, %eax
+; X64-BMI1-NEXT:    notl %eax
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: mask32:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movl $-1, %eax
+; X64-BMI2-NEXT:    bzhil %edi, %eax, %eax
+; X64-BMI2-NEXT:    retq
+  %shl = shl i32 -1, %n
+  %mask = xor i32 %shl, -1
+  ret i32 %mask
+}
+
+define i32 @mask32_indexzext(i8 zeroext %n) nounwind {
+; X86-LABEL: mask32_indexzext:
+; X86:       # %bb.0:
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    movl $-1, %ecx
+; X86-NEXT:    bzhil %eax, %ecx, %eax
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: mask32_indexzext:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movl %edi, %ecx
+; X64-NOBMI-NEXT:    movl $-1, %eax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT:    shll %cl, %eax
+; X64-NOBMI-NEXT:    notl %eax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: mask32_indexzext:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movl %edi, %ecx
+; X64-BMI1-NEXT:    movl $-1, %eax
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT:    shll %cl, %eax
+; X64-BMI1-NEXT:    notl %eax
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: mask32_indexzext:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movl $-1, %eax
+; X64-BMI2-NEXT:    bzhil %edi, %eax, %eax
+; X64-BMI2-NEXT:    retq
+  %conv = zext i8 %n to i32
+  %shl = shl i32 -1, %conv
+  %mask = xor i32 %shl, -1
+  ret i32 %mask
+}
+
+define i64 @mask64(i64 %n) nounwind {
+; X86-LABEL: mask64:
+; X86:       # %bb.0:
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT:    movl $-1, %edx
+; X86-NEXT:    shlxl %ecx, %edx, %eax
+; X86-NEXT:    testb $32, %cl
+; X86-NEXT:    je .LBB2_2
+; X86-NEXT:  # %bb.1:
+; X86-NEXT:    movl %eax, %edx
+; X86-NEXT:    xorl %eax, %eax
+; X86-NEXT:  .LBB2_2:
+; X86-NEXT:    notl %eax
+; X86-NEXT:    notl %edx
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: mask64:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movq %rdi, %rcx
+; X64-NOBMI-NEXT:    movq $-1, %rax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $rcx
+; X64-NOBMI-NEXT:    shlq %cl, %rax
+; X64-NOBMI-NEXT:    notq %rax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: mask64:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movq %rdi, %rcx
+; X64-BMI1-NEXT:    movq $-1, %rax
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $rcx
+; X64-BMI1-NEXT:    shlq %cl, %rax
+; X64-BMI1-NEXT:    notq %rax
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: mask64:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movq $-1, %rax
+; X64-BMI2-NEXT:    bzhiq %rdi, %rax, %rax
+; X64-BMI2-NEXT:    retq
+  %shl = shl i64 -1, %n
+  %mask = xor i64 %shl, -1
+  ret i64 %mask
+}
+
+define i32 @mask64_32_trunc(i64 %n) nounwind {
+; X86-LABEL: mask64_32_trunc:
+; X86:       # %bb.0:
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT:    xorl %eax, %eax
+; X86-NEXT:    testb $32, %cl
+; X86-NEXT:    jne .LBB3_2
+; X86-NEXT:  # %bb.1:
+; X86-NEXT:    movl $-1, %eax
+; X86-NEXT:    shlxl %ecx, %eax, %eax
+; X86-NEXT:  .LBB3_2:
+; X86-NEXT:    notl %eax
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: mask64_32_trunc:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movq %rdi, %rcx
+; X64-NOBMI-NEXT:    movq $-1, %rax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $rcx
+; X64-NOBMI-NEXT:    shlq %cl, %rax
+; X64-NOBMI-NEXT:    notl %eax
+; X64-NOBMI-NEXT:    # kill: def $eax killed $eax killed $rax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: mask64_32_trunc:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movq %rdi, %rcx
+; X64-BMI1-NEXT:    movq $-1, %rax
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $rcx
+; X64-BMI1-NEXT:    shlq %cl, %rax
+; X64-BMI1-NEXT:    notl %eax
+; X64-BMI1-NEXT:    # kill: def $eax killed $eax killed $rax
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: mask64_32_trunc:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movl $-1, %eax
+; X64-BMI2-NEXT:    bzhil %edi, %eax, %eax
+; X64-BMI2-NEXT:    retq
+  %shl = shl i64 -1, %n
+  %trunc = trunc i64 %shl to i32
+  %mask = xor i32 %trunc, -1
+  ret i32 %mask
+}
+
+; The mask is both applied and escapes: with BMI2 both uses become bzhi.
+define i32 @mask32_and_and_escape(i32 %x, i32 %n, ptr %escape) nounwind {
+; X86-LABEL: mask32_and_and_escape:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT:    movl $-1, %edx
+; X86-NEXT:    bzhil %ecx, %edx, %edx
+; X86-NEXT:    movl %edx, (%eax)
+; X86-NEXT:    bzhil %ecx, {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: mask32_and_and_escape:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movl %esi, %ecx
+; X64-NOBMI-NEXT:    movl $-1, %eax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT:    shll %cl, %eax
+; X64-NOBMI-NEXT:    notl %eax
+; X64-NOBMI-NEXT:    movl %eax, (%rdx)
+; X64-NOBMI-NEXT:    andl %edi, %eax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: mask32_and_and_escape:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movl %esi, %ecx
+; X64-BMI1-NEXT:    movl $-1, %esi
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT:    shll %cl, %esi
+; X64-BMI1-NEXT:    andnl %edi, %esi, %eax
+; X64-BMI1-NEXT:    notl %esi
+; X64-BMI1-NEXT:    movl %esi, (%rdx)
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: mask32_and_and_escape:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movl $-1, %eax
+; X64-BMI2-NEXT:    bzhil %esi, %eax, %eax
+; X64-BMI2-NEXT:    movl %eax, (%rdx)
+; X64-BMI2-NEXT:    bzhil %esi, %edi, %eax
+; X64-BMI2-NEXT:    retq
+  %shl = shl i32 -1, %n
+  %mask = xor i32 %shl, -1
+  store i32 %mask, ptr %escape
+  %and = and i32 %x, %mask
+  ret i32 %and
+}
+
+; Not a low-bits mask: the xor constant is not all-ones.
+define i32 @not_mask32(i32 %n) nounwind {
+; X86-LABEL: not_mask32:
+; X86:       # %bb.0:
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    movl $-1, %ecx
+; X86-NEXT:    shlxl %eax, %ecx, %eax
+; X86-NEXT:    xorl $255, %eax
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: not_mask32:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movl %edi, %ecx
+; X64-NOBMI-NEXT:    movl $-1, %eax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT:    shll %cl, %eax
+; X64-NOBMI-NEXT:    xorl $255, %eax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: not_mask32:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movl %edi, %ecx
+; X64-BMI1-NEXT:    movl $-1, %eax
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT:    shll %cl, %eax
+; X64-BMI1-NEXT:    xorl $255, %eax
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: not_mask32:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movl $-1, %eax
+; X64-BMI2-NEXT:    shlxl %edi, %eax, %eax
+; X64-BMI2-NEXT:    xorl $255, %eax
+; X64-BMI2-NEXT:    retq
+  %shl = shl i32 -1, %n
+  %mask = xor i32 %shl, 255
+  ret i32 %mask
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; X64: {{.*}}

``````````

</details>


https://github.com/llvm/llvm-project/pull/226158


More information about the llvm-commits mailing list