[llvm] [X86] Select standalone ~(-1 << n) masks as BZHI (PR #226158)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 06:21:14 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-x86
Author: Chris Kennelly (ckennelly)
<details>
<summary>Changes</summary>
InstCombine canonicalizes (1 << n) - 1 to xor (shl -1, n), -1. When that mask feeds an AND, X86DAGToDAGISel::matchBitExtract folds the whole expression into BZHI, and the add form is selected as BZHI even on its own because Select enters matchBitExtract from ISD::ADD. The xor form on its own is not: a mask that is returned, stored, or consumed by anything but AND is selected as mov -1; shlx; not.
Enter matchBitExtract from ISD::XOR too under BMI2, so the standalone mask becomes mov -1; bzhi. That is one instruction shorter and avoids the shlx+not dependency chain. BMI1-only targets keep the shift form, since BEXTR would need the count moved into bits 15:8 first.
Assisted-by: Claude Code
---
Full diff: https://github.com/llvm/llvm-project/pull/226158.diff
2 Files Affected:
- (modified) llvm/lib/Target/X86/X86ISelDAGToDAG.cpp (+9-4)
- (added) llvm/test/CodeGen/X86/bzhi-standalone-mask.ll (+260)
``````````diff
diff --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index c2de8283f15ff..f272fe63ade4a 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -4146,11 +4146,12 @@ bool X86DAGToDAGISel::foldLoadStoreIntoMemOperand(SDNode *Node) {
// c) x & (-1 >> (32 - y))
// d) x << (32 - y) >> (32 - y)
// e) (1 << nbits) - 1
+// f) ~(-1 << nbits)
bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
- assert(
- (Node->getOpcode() == ISD::ADD || Node->getOpcode() == ISD::AND ||
- Node->getOpcode() == ISD::SRL) &&
- "Should be either an and-mask, or right-shift after clearing high bits.");
+ assert((Node->getOpcode() == ISD::ADD || Node->getOpcode() == ISD::AND ||
+ Node->getOpcode() == ISD::XOR || Node->getOpcode() == ISD::SRL) &&
+ "Should be either an and-mask, a standalone low-bits mask, or "
+ "right-shift after clearing high bits.");
// BEXTR is BMI instruction, BZHI is BMI2 instruction. We need at least one.
if (!Subtarget->hasBMI() && !Subtarget->hasBMI2())
@@ -5818,6 +5819,10 @@ void X86DAGToDAGISel::Select(SDNode *Node) {
[[fallthrough]];
case ISD::OR:
case ISD::XOR:
+ // A standalone ~(-1 << n) mask is (-1 & lowmask(n)): mov -1; bzhi beats
+ // mov -1; shlx; not.
+ if (Opcode == ISD::XOR && Subtarget->hasBMI2() && matchBitExtract(Node))
+ return;
if (tryShrinkShlLogicImm(Node))
return;
if (Opcode == ISD::OR && tryMatchBitSelect(Node))
diff --git a/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll b/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll
new file mode 100644
index 0000000000000..950277d4d41d9
--- /dev/null
+++ b/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll
@@ -0,0 +1,260 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=i686-unknown-linux-gnu -mattr=+bmi,+bmi2 | FileCheck %s --check-prefixes=X86
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=-bmi,-bmi2 | FileCheck %s --check-prefixes=X64,X64-NOBMI
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+bmi,-bmi2 | FileCheck %s --check-prefixes=X64,X64-BMI1
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+bmi,+bmi2 | FileCheck %s --check-prefixes=X64,X64-BMI2
+
+; A low-bits mask that is not immediately consumed by an `and`. InstCombine
+; canonicalizes (1 << n) - 1 to ~(-1 << n), and with BMI2 that is
+; bzhi(-1, n): one instruction instead of shlx + not.
+
+define i32 @mask32(i32 %n) nounwind {
+; X86-LABEL: mask32:
+; X86: # %bb.0:
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT: movl $-1, %ecx
+; X86-NEXT: bzhil %eax, %ecx, %eax
+; X86-NEXT: retl
+;
+; X64-NOBMI-LABEL: mask32:
+; X64-NOBMI: # %bb.0:
+; X64-NOBMI-NEXT: movl %edi, %ecx
+; X64-NOBMI-NEXT: movl $-1, %eax
+; X64-NOBMI-NEXT: # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT: shll %cl, %eax
+; X64-NOBMI-NEXT: notl %eax
+; X64-NOBMI-NEXT: retq
+;
+; X64-BMI1-LABEL: mask32:
+; X64-BMI1: # %bb.0:
+; X64-BMI1-NEXT: movl %edi, %ecx
+; X64-BMI1-NEXT: movl $-1, %eax
+; X64-BMI1-NEXT: # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT: shll %cl, %eax
+; X64-BMI1-NEXT: notl %eax
+; X64-BMI1-NEXT: retq
+;
+; X64-BMI2-LABEL: mask32:
+; X64-BMI2: # %bb.0:
+; X64-BMI2-NEXT: movl $-1, %eax
+; X64-BMI2-NEXT: bzhil %edi, %eax, %eax
+; X64-BMI2-NEXT: retq
+ %shl = shl i32 -1, %n
+ %mask = xor i32 %shl, -1
+ ret i32 %mask
+}
+
+define i32 @mask32_indexzext(i8 zeroext %n) nounwind {
+; X86-LABEL: mask32_indexzext:
+; X86: # %bb.0:
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT: movl $-1, %ecx
+; X86-NEXT: bzhil %eax, %ecx, %eax
+; X86-NEXT: retl
+;
+; X64-NOBMI-LABEL: mask32_indexzext:
+; X64-NOBMI: # %bb.0:
+; X64-NOBMI-NEXT: movl %edi, %ecx
+; X64-NOBMI-NEXT: movl $-1, %eax
+; X64-NOBMI-NEXT: # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT: shll %cl, %eax
+; X64-NOBMI-NEXT: notl %eax
+; X64-NOBMI-NEXT: retq
+;
+; X64-BMI1-LABEL: mask32_indexzext:
+; X64-BMI1: # %bb.0:
+; X64-BMI1-NEXT: movl %edi, %ecx
+; X64-BMI1-NEXT: movl $-1, %eax
+; X64-BMI1-NEXT: # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT: shll %cl, %eax
+; X64-BMI1-NEXT: notl %eax
+; X64-BMI1-NEXT: retq
+;
+; X64-BMI2-LABEL: mask32_indexzext:
+; X64-BMI2: # %bb.0:
+; X64-BMI2-NEXT: movl $-1, %eax
+; X64-BMI2-NEXT: bzhil %edi, %eax, %eax
+; X64-BMI2-NEXT: retq
+ %conv = zext i8 %n to i32
+ %shl = shl i32 -1, %conv
+ %mask = xor i32 %shl, -1
+ ret i32 %mask
+}
+
+define i64 @mask64(i64 %n) nounwind {
+; X86-LABEL: mask64:
+; X86: # %bb.0:
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT: movl $-1, %edx
+; X86-NEXT: shlxl %ecx, %edx, %eax
+; X86-NEXT: testb $32, %cl
+; X86-NEXT: je .LBB2_2
+; X86-NEXT: # %bb.1:
+; X86-NEXT: movl %eax, %edx
+; X86-NEXT: xorl %eax, %eax
+; X86-NEXT: .LBB2_2:
+; X86-NEXT: notl %eax
+; X86-NEXT: notl %edx
+; X86-NEXT: retl
+;
+; X64-NOBMI-LABEL: mask64:
+; X64-NOBMI: # %bb.0:
+; X64-NOBMI-NEXT: movq %rdi, %rcx
+; X64-NOBMI-NEXT: movq $-1, %rax
+; X64-NOBMI-NEXT: # kill: def $cl killed $cl killed $rcx
+; X64-NOBMI-NEXT: shlq %cl, %rax
+; X64-NOBMI-NEXT: notq %rax
+; X64-NOBMI-NEXT: retq
+;
+; X64-BMI1-LABEL: mask64:
+; X64-BMI1: # %bb.0:
+; X64-BMI1-NEXT: movq %rdi, %rcx
+; X64-BMI1-NEXT: movq $-1, %rax
+; X64-BMI1-NEXT: # kill: def $cl killed $cl killed $rcx
+; X64-BMI1-NEXT: shlq %cl, %rax
+; X64-BMI1-NEXT: notq %rax
+; X64-BMI1-NEXT: retq
+;
+; X64-BMI2-LABEL: mask64:
+; X64-BMI2: # %bb.0:
+; X64-BMI2-NEXT: movq $-1, %rax
+; X64-BMI2-NEXT: bzhiq %rdi, %rax, %rax
+; X64-BMI2-NEXT: retq
+ %shl = shl i64 -1, %n
+ %mask = xor i64 %shl, -1
+ ret i64 %mask
+}
+
+define i32 @mask64_32_trunc(i64 %n) nounwind {
+; X86-LABEL: mask64_32_trunc:
+; X86: # %bb.0:
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT: xorl %eax, %eax
+; X86-NEXT: testb $32, %cl
+; X86-NEXT: jne .LBB3_2
+; X86-NEXT: # %bb.1:
+; X86-NEXT: movl $-1, %eax
+; X86-NEXT: shlxl %ecx, %eax, %eax
+; X86-NEXT: .LBB3_2:
+; X86-NEXT: notl %eax
+; X86-NEXT: retl
+;
+; X64-NOBMI-LABEL: mask64_32_trunc:
+; X64-NOBMI: # %bb.0:
+; X64-NOBMI-NEXT: movq %rdi, %rcx
+; X64-NOBMI-NEXT: movq $-1, %rax
+; X64-NOBMI-NEXT: # kill: def $cl killed $cl killed $rcx
+; X64-NOBMI-NEXT: shlq %cl, %rax
+; X64-NOBMI-NEXT: notl %eax
+; X64-NOBMI-NEXT: # kill: def $eax killed $eax killed $rax
+; X64-NOBMI-NEXT: retq
+;
+; X64-BMI1-LABEL: mask64_32_trunc:
+; X64-BMI1: # %bb.0:
+; X64-BMI1-NEXT: movq %rdi, %rcx
+; X64-BMI1-NEXT: movq $-1, %rax
+; X64-BMI1-NEXT: # kill: def $cl killed $cl killed $rcx
+; X64-BMI1-NEXT: shlq %cl, %rax
+; X64-BMI1-NEXT: notl %eax
+; X64-BMI1-NEXT: # kill: def $eax killed $eax killed $rax
+; X64-BMI1-NEXT: retq
+;
+; X64-BMI2-LABEL: mask64_32_trunc:
+; X64-BMI2: # %bb.0:
+; X64-BMI2-NEXT: movl $-1, %eax
+; X64-BMI2-NEXT: bzhil %edi, %eax, %eax
+; X64-BMI2-NEXT: retq
+ %shl = shl i64 -1, %n
+ %trunc = trunc i64 %shl to i32
+ %mask = xor i32 %trunc, -1
+ ret i32 %mask
+}
+
+; The mask is both applied and escapes: with BMI2 both uses become bzhi.
+define i32 @mask32_and_and_escape(i32 %x, i32 %n, ptr %escape) nounwind {
+; X86-LABEL: mask32_and_and_escape:
+; X86: # %bb.0:
+; X86-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT: movl $-1, %edx
+; X86-NEXT: bzhil %ecx, %edx, %edx
+; X86-NEXT: movl %edx, (%eax)
+; X86-NEXT: bzhil %ecx, {{[0-9]+}}(%esp), %eax
+; X86-NEXT: retl
+;
+; X64-NOBMI-LABEL: mask32_and_and_escape:
+; X64-NOBMI: # %bb.0:
+; X64-NOBMI-NEXT: movl %esi, %ecx
+; X64-NOBMI-NEXT: movl $-1, %eax
+; X64-NOBMI-NEXT: # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT: shll %cl, %eax
+; X64-NOBMI-NEXT: notl %eax
+; X64-NOBMI-NEXT: movl %eax, (%rdx)
+; X64-NOBMI-NEXT: andl %edi, %eax
+; X64-NOBMI-NEXT: retq
+;
+; X64-BMI1-LABEL: mask32_and_and_escape:
+; X64-BMI1: # %bb.0:
+; X64-BMI1-NEXT: movl %esi, %ecx
+; X64-BMI1-NEXT: movl $-1, %esi
+; X64-BMI1-NEXT: # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT: shll %cl, %esi
+; X64-BMI1-NEXT: andnl %edi, %esi, %eax
+; X64-BMI1-NEXT: notl %esi
+; X64-BMI1-NEXT: movl %esi, (%rdx)
+; X64-BMI1-NEXT: retq
+;
+; X64-BMI2-LABEL: mask32_and_and_escape:
+; X64-BMI2: # %bb.0:
+; X64-BMI2-NEXT: movl $-1, %eax
+; X64-BMI2-NEXT: bzhil %esi, %eax, %eax
+; X64-BMI2-NEXT: movl %eax, (%rdx)
+; X64-BMI2-NEXT: bzhil %esi, %edi, %eax
+; X64-BMI2-NEXT: retq
+ %shl = shl i32 -1, %n
+ %mask = xor i32 %shl, -1
+ store i32 %mask, ptr %escape
+ %and = and i32 %x, %mask
+ ret i32 %and
+}
+
+; Not a low-bits mask: the xor constant is not all-ones.
+define i32 @not_mask32(i32 %n) nounwind {
+; X86-LABEL: not_mask32:
+; X86: # %bb.0:
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT: movl $-1, %ecx
+; X86-NEXT: shlxl %eax, %ecx, %eax
+; X86-NEXT: xorl $255, %eax
+; X86-NEXT: retl
+;
+; X64-NOBMI-LABEL: not_mask32:
+; X64-NOBMI: # %bb.0:
+; X64-NOBMI-NEXT: movl %edi, %ecx
+; X64-NOBMI-NEXT: movl $-1, %eax
+; X64-NOBMI-NEXT: # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT: shll %cl, %eax
+; X64-NOBMI-NEXT: xorl $255, %eax
+; X64-NOBMI-NEXT: retq
+;
+; X64-BMI1-LABEL: not_mask32:
+; X64-BMI1: # %bb.0:
+; X64-BMI1-NEXT: movl %edi, %ecx
+; X64-BMI1-NEXT: movl $-1, %eax
+; X64-BMI1-NEXT: # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT: shll %cl, %eax
+; X64-BMI1-NEXT: xorl $255, %eax
+; X64-BMI1-NEXT: retq
+;
+; X64-BMI2-LABEL: not_mask32:
+; X64-BMI2: # %bb.0:
+; X64-BMI2-NEXT: movl $-1, %eax
+; X64-BMI2-NEXT: shlxl %edi, %eax, %eax
+; X64-BMI2-NEXT: xorl $255, %eax
+; X64-BMI2-NEXT: retq
+ %shl = shl i32 -1, %n
+ %mask = xor i32 %shl, 255
+ ret i32 %mask
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; X64: {{.*}}
``````````
</details>
https://github.com/llvm/llvm-project/pull/226158
More information about the llvm-commits
mailing list