[llvm] [X86] Select standalone ~(-1 << n) masks as BZHI (PR #226158)

Chris Kennelly via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 07:30:06 PDT 2026


https://github.com/ckennelly updated https://github.com/llvm/llvm-project/pull/226158

>From 799f6796bddfbea01fa508e87ac51f3ba8fc38c3 Mon Sep 17 00:00:00 2001
From: Chris Kennelly <ckennelly at ckennelly.com>
Date: Thu, 24 Sep 2026 14:27:52 +0000
Subject: [PATCH 1/2] [X86] Add tests for standalone ~(-1 << n) masks (NFC)

InstCombine canonicalizes (1 << n) - 1 to xor (shl -1, n), -1. When that
mask feeds an AND, X86DAGToDAGISel::matchBitExtract folds the whole
expression into BZHI, but a mask that is returned, stored, or consumed
by anything but AND is selected as mov -1; shlx; not on BMI2 targets.

Add tests for the standalone i32 and i64 masks, a zero-extended i8
count, a truncated i64 mask, a mask that is both applied and escapes,
and a non-all-ones xor that must not be treated as a mask, at the
current codegen.

Assisted-by: Claude Code
---
 llvm/test/CodeGen/X86/bzhi-standalone-mask.ll | 269 ++++++++++++++++++
 1 file changed, 269 insertions(+)
 create mode 100644 llvm/test/CodeGen/X86/bzhi-standalone-mask.ll

diff --git a/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll b/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll
new file mode 100644
index 0000000000000..111db474433a0
--- /dev/null
+++ b/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll
@@ -0,0 +1,269 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=i686-unknown-linux-gnu -mattr=+bmi,+bmi2 | FileCheck %s --check-prefixes=X86
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=-bmi,-bmi2 | FileCheck %s --check-prefixes=X64,X64-NOBMI
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+bmi,-bmi2 | FileCheck %s --check-prefixes=X64,X64-BMI1
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+bmi,+bmi2 | FileCheck %s --check-prefixes=X64,X64-BMI2
+
+; A low-bits mask that is not immediately consumed by an `and`. InstCombine
+; canonicalizes (1 << n) - 1 to ~(-1 << n); with BMI2 this can be
+; bzhi(-1, n): one instruction instead of shlx + not.
+
+define i32 @mask32(i32 %n) nounwind {
+; X86-LABEL: mask32:
+; X86:       # %bb.0:
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    movl $-1, %ecx
+; X86-NEXT:    shlxl %eax, %ecx, %eax
+; X86-NEXT:    notl %eax
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: mask32:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movl %edi, %ecx
+; X64-NOBMI-NEXT:    movl $-1, %eax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT:    shll %cl, %eax
+; X64-NOBMI-NEXT:    notl %eax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: mask32:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movl %edi, %ecx
+; X64-BMI1-NEXT:    movl $-1, %eax
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT:    shll %cl, %eax
+; X64-BMI1-NEXT:    notl %eax
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: mask32:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movl $-1, %eax
+; X64-BMI2-NEXT:    shlxl %edi, %eax, %eax
+; X64-BMI2-NEXT:    notl %eax
+; X64-BMI2-NEXT:    retq
+  %shl = shl i32 -1, %n
+  %mask = xor i32 %shl, -1
+  ret i32 %mask
+}
+
+define i32 @mask32_indexzext(i8 zeroext %n) nounwind {
+; X86-LABEL: mask32_indexzext:
+; X86:       # %bb.0:
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    movl $-1, %ecx
+; X86-NEXT:    shlxl %eax, %ecx, %eax
+; X86-NEXT:    notl %eax
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: mask32_indexzext:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movl %edi, %ecx
+; X64-NOBMI-NEXT:    movl $-1, %eax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT:    shll %cl, %eax
+; X64-NOBMI-NEXT:    notl %eax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: mask32_indexzext:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movl %edi, %ecx
+; X64-BMI1-NEXT:    movl $-1, %eax
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT:    shll %cl, %eax
+; X64-BMI1-NEXT:    notl %eax
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: mask32_indexzext:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movl $-1, %eax
+; X64-BMI2-NEXT:    shlxl %edi, %eax, %eax
+; X64-BMI2-NEXT:    notl %eax
+; X64-BMI2-NEXT:    retq
+  %conv = zext i8 %n to i32
+  %shl = shl i32 -1, %conv
+  %mask = xor i32 %shl, -1
+  ret i32 %mask
+}
+
+define i64 @mask64(i64 %n) nounwind {
+; X86-LABEL: mask64:
+; X86:       # %bb.0:
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT:    movl $-1, %edx
+; X86-NEXT:    shlxl %ecx, %edx, %eax
+; X86-NEXT:    testb $32, %cl
+; X86-NEXT:    je .LBB2_2
+; X86-NEXT:  # %bb.1:
+; X86-NEXT:    movl %eax, %edx
+; X86-NEXT:    xorl %eax, %eax
+; X86-NEXT:  .LBB2_2:
+; X86-NEXT:    notl %eax
+; X86-NEXT:    notl %edx
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: mask64:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movq %rdi, %rcx
+; X64-NOBMI-NEXT:    movq $-1, %rax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $rcx
+; X64-NOBMI-NEXT:    shlq %cl, %rax
+; X64-NOBMI-NEXT:    notq %rax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: mask64:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movq %rdi, %rcx
+; X64-BMI1-NEXT:    movq $-1, %rax
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $rcx
+; X64-BMI1-NEXT:    shlq %cl, %rax
+; X64-BMI1-NEXT:    notq %rax
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: mask64:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movq $-1, %rax
+; X64-BMI2-NEXT:    shlxq %rdi, %rax, %rax
+; X64-BMI2-NEXT:    notq %rax
+; X64-BMI2-NEXT:    retq
+  %shl = shl i64 -1, %n
+  %mask = xor i64 %shl, -1
+  ret i64 %mask
+}
+
+define i32 @mask64_32_trunc(i64 %n) nounwind {
+; X86-LABEL: mask64_32_trunc:
+; X86:       # %bb.0:
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT:    xorl %eax, %eax
+; X86-NEXT:    testb $32, %cl
+; X86-NEXT:    jne .LBB3_2
+; X86-NEXT:  # %bb.1:
+; X86-NEXT:    movl $-1, %eax
+; X86-NEXT:    shlxl %ecx, %eax, %eax
+; X86-NEXT:  .LBB3_2:
+; X86-NEXT:    notl %eax
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: mask64_32_trunc:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movq %rdi, %rcx
+; X64-NOBMI-NEXT:    movq $-1, %rax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $rcx
+; X64-NOBMI-NEXT:    shlq %cl, %rax
+; X64-NOBMI-NEXT:    notl %eax
+; X64-NOBMI-NEXT:    # kill: def $eax killed $eax killed $rax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: mask64_32_trunc:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movq %rdi, %rcx
+; X64-BMI1-NEXT:    movq $-1, %rax
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $rcx
+; X64-BMI1-NEXT:    shlq %cl, %rax
+; X64-BMI1-NEXT:    notl %eax
+; X64-BMI1-NEXT:    # kill: def $eax killed $eax killed $rax
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: mask64_32_trunc:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movq $-1, %rax
+; X64-BMI2-NEXT:    shlxq %rdi, %rax, %rax
+; X64-BMI2-NEXT:    notl %eax
+; X64-BMI2-NEXT:    # kill: def $eax killed $eax killed $rax
+; X64-BMI2-NEXT:    retq
+  %shl = shl i64 -1, %n
+  %trunc = trunc i64 %shl to i32
+  %mask = xor i32 %trunc, -1
+  ret i32 %mask
+}
+
+; The mask is both applied and escapes: with BMI2 both uses become bzhi.
+define i32 @mask32_and_and_escape(i32 %x, i32 %n, ptr %escape) nounwind {
+; X86-LABEL: mask32_and_and_escape:
+; X86:       # %bb.0:
+; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT:    movl $-1, %edx
+; X86-NEXT:    shlxl %ecx, %edx, %edx
+; X86-NEXT:    notl %edx
+; X86-NEXT:    movl %edx, (%eax)
+; X86-NEXT:    bzhil %ecx, {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: mask32_and_and_escape:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movl %esi, %ecx
+; X64-NOBMI-NEXT:    movl $-1, %eax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT:    shll %cl, %eax
+; X64-NOBMI-NEXT:    notl %eax
+; X64-NOBMI-NEXT:    movl %eax, (%rdx)
+; X64-NOBMI-NEXT:    andl %edi, %eax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: mask32_and_and_escape:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movl %esi, %ecx
+; X64-BMI1-NEXT:    movl $-1, %esi
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT:    shll %cl, %esi
+; X64-BMI1-NEXT:    andnl %edi, %esi, %eax
+; X64-BMI1-NEXT:    notl %esi
+; X64-BMI1-NEXT:    movl %esi, (%rdx)
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: mask32_and_and_escape:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movl $-1, %eax
+; X64-BMI2-NEXT:    shlxl %esi, %eax, %eax
+; X64-BMI2-NEXT:    notl %eax
+; X64-BMI2-NEXT:    movl %eax, (%rdx)
+; X64-BMI2-NEXT:    bzhil %esi, %edi, %eax
+; X64-BMI2-NEXT:    retq
+  %shl = shl i32 -1, %n
+  %mask = xor i32 %shl, -1
+  store i32 %mask, ptr %escape
+  %and = and i32 %x, %mask
+  ret i32 %and
+}
+
+; Not a low-bits mask: the xor constant is not all-ones.
+define i32 @not_mask32(i32 %n) nounwind {
+; X86-LABEL: not_mask32:
+; X86:       # %bb.0:
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    movl $-1, %ecx
+; X86-NEXT:    shlxl %eax, %ecx, %eax
+; X86-NEXT:    xorl $255, %eax
+; X86-NEXT:    retl
+;
+; X64-NOBMI-LABEL: not_mask32:
+; X64-NOBMI:       # %bb.0:
+; X64-NOBMI-NEXT:    movl %edi, %ecx
+; X64-NOBMI-NEXT:    movl $-1, %eax
+; X64-NOBMI-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI-NEXT:    shll %cl, %eax
+; X64-NOBMI-NEXT:    xorl $255, %eax
+; X64-NOBMI-NEXT:    retq
+;
+; X64-BMI1-LABEL: not_mask32:
+; X64-BMI1:       # %bb.0:
+; X64-BMI1-NEXT:    movl %edi, %ecx
+; X64-BMI1-NEXT:    movl $-1, %eax
+; X64-BMI1-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-BMI1-NEXT:    shll %cl, %eax
+; X64-BMI1-NEXT:    xorl $255, %eax
+; X64-BMI1-NEXT:    retq
+;
+; X64-BMI2-LABEL: not_mask32:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movl $-1, %eax
+; X64-BMI2-NEXT:    shlxl %edi, %eax, %eax
+; X64-BMI2-NEXT:    xorl $255, %eax
+; X64-BMI2-NEXT:    retq
+  %shl = shl i32 -1, %n
+  %mask = xor i32 %shl, 255
+  ret i32 %mask
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; X64: {{.*}}

>From 8266e994a8f4a48198d6000de0b2f38e3e6e8daa Mon Sep 17 00:00:00 2001
From: Chris Kennelly <ckennelly at ckennelly.com>
Date: Wed, 23 Sep 2026 23:04:56 +0000
Subject: [PATCH 2/2] [X86] Select standalone ~(-1 << n) masks as BZHI

InstCombine canonicalizes (1 << n) - 1 to xor (shl -1, n), -1. When that
mask feeds an AND, X86DAGToDAGISel::matchBitExtract folds the whole
expression into BZHI, and the add form is selected as BZHI even on its
own because Select enters matchBitExtract from ISD::ADD. The xor form on
its own is not: a mask that is returned, stored, or consumed by anything
but AND is selected as mov -1; shlx; not.

Enter matchBitExtract from ISD::XOR too under BMI2, so the standalone
mask becomes mov -1; bzhi. That is one instruction shorter and avoids
the shlx+not dependency chain. BMI1-only targets keep the shift form,
since BEXTR would need the count moved into bits 15:8 first.

Assisted-by: Claude Code
---
 llvm/lib/Target/X86/X86ISelDAGToDAG.cpp       | 13 ++++++---
 llvm/test/CodeGen/X86/bzhi-standalone-mask.ll | 27 +++++++------------
 2 files changed, 18 insertions(+), 22 deletions(-)

diff --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index c2de8283f15ff..f272fe63ade4a 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -4146,11 +4146,12 @@ bool X86DAGToDAGISel::foldLoadStoreIntoMemOperand(SDNode *Node) {
 //   c) x &  (-1 >> (32 - y))
 //   d) x << (32 - y) >> (32 - y)
 //   e) (1 << nbits) - 1
+//   f) ~(-1 << nbits)
 bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
-  assert(
-      (Node->getOpcode() == ISD::ADD || Node->getOpcode() == ISD::AND ||
-       Node->getOpcode() == ISD::SRL) &&
-      "Should be either an and-mask, or right-shift after clearing high bits.");
+  assert((Node->getOpcode() == ISD::ADD || Node->getOpcode() == ISD::AND ||
+          Node->getOpcode() == ISD::XOR || Node->getOpcode() == ISD::SRL) &&
+         "Should be either an and-mask, a standalone low-bits mask, or "
+         "right-shift after clearing high bits.");
 
   // BEXTR is BMI instruction, BZHI is BMI2 instruction. We need at least one.
   if (!Subtarget->hasBMI() && !Subtarget->hasBMI2())
@@ -5818,6 +5819,10 @@ void X86DAGToDAGISel::Select(SDNode *Node) {
     [[fallthrough]];
   case ISD::OR:
   case ISD::XOR:
+    // A standalone ~(-1 << n) mask is (-1 & lowmask(n)): mov -1; bzhi beats
+    // mov -1; shlx; not.
+    if (Opcode == ISD::XOR && Subtarget->hasBMI2() && matchBitExtract(Node))
+      return;
     if (tryShrinkShlLogicImm(Node))
       return;
     if (Opcode == ISD::OR && tryMatchBitSelect(Node))
diff --git a/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll b/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll
index 111db474433a0..449aa689709b9 100644
--- a/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll
+++ b/llvm/test/CodeGen/X86/bzhi-standalone-mask.ll
@@ -13,8 +13,7 @@ define i32 @mask32(i32 %n) nounwind {
 ; X86:       # %bb.0:
 ; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
 ; X86-NEXT:    movl $-1, %ecx
-; X86-NEXT:    shlxl %eax, %ecx, %eax
-; X86-NEXT:    notl %eax
+; X86-NEXT:    bzhil %eax, %ecx, %eax
 ; X86-NEXT:    retl
 ;
 ; X64-NOBMI-LABEL: mask32:
@@ -38,8 +37,7 @@ define i32 @mask32(i32 %n) nounwind {
 ; X64-BMI2-LABEL: mask32:
 ; X64-BMI2:       # %bb.0:
 ; X64-BMI2-NEXT:    movl $-1, %eax
-; X64-BMI2-NEXT:    shlxl %edi, %eax, %eax
-; X64-BMI2-NEXT:    notl %eax
+; X64-BMI2-NEXT:    bzhil %edi, %eax, %eax
 ; X64-BMI2-NEXT:    retq
   %shl = shl i32 -1, %n
   %mask = xor i32 %shl, -1
@@ -51,8 +49,7 @@ define i32 @mask32_indexzext(i8 zeroext %n) nounwind {
 ; X86:       # %bb.0:
 ; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
 ; X86-NEXT:    movl $-1, %ecx
-; X86-NEXT:    shlxl %eax, %ecx, %eax
-; X86-NEXT:    notl %eax
+; X86-NEXT:    bzhil %eax, %ecx, %eax
 ; X86-NEXT:    retl
 ;
 ; X64-NOBMI-LABEL: mask32_indexzext:
@@ -76,8 +73,7 @@ define i32 @mask32_indexzext(i8 zeroext %n) nounwind {
 ; X64-BMI2-LABEL: mask32_indexzext:
 ; X64-BMI2:       # %bb.0:
 ; X64-BMI2-NEXT:    movl $-1, %eax
-; X64-BMI2-NEXT:    shlxl %edi, %eax, %eax
-; X64-BMI2-NEXT:    notl %eax
+; X64-BMI2-NEXT:    bzhil %edi, %eax, %eax
 ; X64-BMI2-NEXT:    retq
   %conv = zext i8 %n to i32
   %shl = shl i32 -1, %conv
@@ -122,8 +118,7 @@ define i64 @mask64(i64 %n) nounwind {
 ; X64-BMI2-LABEL: mask64:
 ; X64-BMI2:       # %bb.0:
 ; X64-BMI2-NEXT:    movq $-1, %rax
-; X64-BMI2-NEXT:    shlxq %rdi, %rax, %rax
-; X64-BMI2-NEXT:    notq %rax
+; X64-BMI2-NEXT:    bzhiq %rdi, %rax, %rax
 ; X64-BMI2-NEXT:    retq
   %shl = shl i64 -1, %n
   %mask = xor i64 %shl, -1
@@ -166,10 +161,8 @@ define i32 @mask64_32_trunc(i64 %n) nounwind {
 ;
 ; X64-BMI2-LABEL: mask64_32_trunc:
 ; X64-BMI2:       # %bb.0:
-; X64-BMI2-NEXT:    movq $-1, %rax
-; X64-BMI2-NEXT:    shlxq %rdi, %rax, %rax
-; X64-BMI2-NEXT:    notl %eax
-; X64-BMI2-NEXT:    # kill: def $eax killed $eax killed $rax
+; X64-BMI2-NEXT:    movl $-1, %eax
+; X64-BMI2-NEXT:    bzhil %edi, %eax, %eax
 ; X64-BMI2-NEXT:    retq
   %shl = shl i64 -1, %n
   %trunc = trunc i64 %shl to i32
@@ -184,8 +177,7 @@ define i32 @mask32_and_and_escape(i32 %x, i32 %n, ptr %escape) nounwind {
 ; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
 ; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
 ; X86-NEXT:    movl $-1, %edx
-; X86-NEXT:    shlxl %ecx, %edx, %edx
-; X86-NEXT:    notl %edx
+; X86-NEXT:    bzhil %ecx, %edx, %edx
 ; X86-NEXT:    movl %edx, (%eax)
 ; X86-NEXT:    bzhil %ecx, {{[0-9]+}}(%esp), %eax
 ; X86-NEXT:    retl
@@ -215,8 +207,7 @@ define i32 @mask32_and_and_escape(i32 %x, i32 %n, ptr %escape) nounwind {
 ; X64-BMI2-LABEL: mask32_and_and_escape:
 ; X64-BMI2:       # %bb.0:
 ; X64-BMI2-NEXT:    movl $-1, %eax
-; X64-BMI2-NEXT:    shlxl %esi, %eax, %eax
-; X64-BMI2-NEXT:    notl %eax
+; X64-BMI2-NEXT:    bzhil %esi, %eax, %eax
 ; X64-BMI2-NEXT:    movl %eax, (%rdx)
 ; X64-BMI2-NEXT:    bzhil %esi, %edi, %eax
 ; X64-BMI2-NEXT:    retq



More information about the llvm-commits mailing list