[llvm] [X86] Fix failure to lower i8/i16 bzhi pattern to bzhi (PR #226401)

via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 01:57:41 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-x86

Author: Divyansh Yadav (schizophrenicmaniac)

<details>
<summary>Changes</summary>

This PR fixes an issue where `DAGToDAGISel` failed to lower bit extraction patterns to the `bzhi` instruction for 8-bit and 16-bit integer types (`i8` and `i16`), even though it already handled them correctly for `i32` and `i64`.

**Root Cause**:
For `i8` and `i16`, the `DAGCombiner` transforms shift combinations by promoting them to `i32` shifts and inserting an explicit `AND` mask (e.g., `AND 255` or `AND 65535`) to simulate the narrower type. This caused the pattern matching in `X86ISelDAGToDAG::matchBitExtract` to fail because it encountered the `AND` node instead of the expected `SHL` node. Additionally, the existing logic wasn't equipped to insert/extract subregisters to use the natively 32-bit `bzhi` instruction for narrower types.

**Fix**:
1. Updated `matchPatternD` in `X86ISelDAGToDAG.cpp` to look past the `AND 255` and `AND 65535` masks that the `DAGCombiner` inserts for promoted `i8`/`i16` shifts.
2. Updated `peekThroughOneUseTruncation` to assert that the source size is strictly greater than the destination size, allowing for truncations involving `i8` and `i16`.
3. Adjusted the `matchBitExtract` logic to support `i8` and `i16` values by dynamically wrapping the operations with `INSERT_SUBREG` to place them into an `i32` register for `BZHI32`, and extracting the result back out using `EXTRACT_SUBREG`.
4. Updated `clear-highbits.ll` using `update_llc_test_checks.py` to assert that we are now properly generating the `bzhi` instruction for `i8` and `i16` types.

Fixes #<!-- -->226376.


---
Full diff: https://github.com/llvm/llvm-project/pull/226401.diff


2 Files Affected:

- (modified) llvm/lib/Target/X86/X86ISelDAGToDAG.cpp (+62-11) 
- (modified) llvm/test/CodeGen/X86/clear-highbits.ll (+140-81) 


``````````diff
diff --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index c2de8283f15ff..75dd37d52a2e6 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -4158,12 +4158,17 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
 
   MVT NVT = Node->getSimpleValueType(0);
 
-  // Only supported for 32 and 64 bits.
-  if (NVT != MVT::i32 && NVT != MVT::i64)
+  // Only supported for 8, 16, 32 and 64 bits.
+  if (NVT != MVT::i8 && NVT != MVT::i16 && NVT != MVT::i32 && NVT != MVT::i64)
     return false;
 
   SDValue NBits;
   bool NegateNBits;
+  // The bitwidth of the value that the matched pattern operates on. This can
+  // be narrower than the node's type when the pattern was widened during type
+  // legalization, e.g. an i16 pattern that is visible as an i32 node with an
+  // explicit 0xFFFF mask.
+  unsigned MatchedBitwidth = NVT.getSizeInBits();
 
   // If we have BMI2's BZHI, we are ok with muti-use patterns.
   // Else, if we only have BMI1's BEXTR, we require one-use.
@@ -4187,9 +4192,9 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
 
   auto peekThroughOneUseTruncation = [checkOneUse](SDValue V) {
     if (V->getOpcode() == ISD::TRUNCATE && checkOneUse(V)) {
-      assert(V.getSimpleValueType() == MVT::i32 &&
-             V.getOperand(0).getSimpleValueType() == MVT::i64 &&
-             "Expected i64 -> i32 truncation");
+      assert(V.getSimpleValueType().getSizeInBits() <
+                 V.getOperand(0).getSimpleValueType().getSizeInBits() &&
+             "Expected truncation");
       V = V.getOperand(0);
     }
     return V;
@@ -4245,8 +4250,10 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
 
   // Try to match potentially-truncated shift amount as `(bitwidth - y)`,
   // or leave the shift amount as-is, but then we'll have to negate it.
-  auto canonicalizeShiftAmt = [&NBits, &NegateNBits](SDValue ShiftAmt,
-                                                     unsigned Bitwidth) {
+  auto canonicalizeShiftAmt = [&NBits, &NegateNBits,
+                               &MatchedBitwidth](SDValue ShiftAmt,
+                                                 unsigned Bitwidth) {
+    MatchedBitwidth = Bitwidth;
     NBits = ShiftAmt;
     NegateNBits = true;
     // Skip over a truncate of the shift amount, if any.
@@ -4300,9 +4307,20 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
     if (Node->getOpcode() != ISD::SRL)
       return false;
     SDValue N0 = Node->getOperand(0);
+    unsigned Bitwidth = N0.getSimpleValueType().getSizeInBits();
+    if (N0->getOpcode() == ISD::AND) {
+      if (auto *C = dyn_cast<ConstantSDNode>(N0->getOperand(1))) {
+        if (C->getZExtValue() == 65535) {
+          Bitwidth = 16;
+          N0 = N0->getOperand(0);
+        } else if (C->getZExtValue() == 255) {
+          Bitwidth = 8;
+          N0 = N0->getOperand(0);
+        }
+      }
+    }
     if (N0->getOpcode() != ISD::SHL)
       return false;
-    unsigned Bitwidth = N0.getSimpleValueType().getSizeInBits();
     SDValue N1 = Node->getOperand(1);
     SDValue N01 = N0->getOperand(1);
     // Both of the shifts must be by the exact same value.
@@ -4379,10 +4397,24 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
     insertDAGNode(*CurDAG, SDValue(Node, 0), NBits);
   }
 
+  MVT OVT = NVT;
+  if (NVT == MVT::i8 || NVT == MVT::i16) {
+    NVT = MVT::i32;
+    SDValue ImplDef = SDValue(
+        CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i32), 0);
+    SDValue SubRegIdx = CurDAG->getTargetConstant(
+        OVT == MVT::i8 ? X86::sub_8bit : X86::sub_16bit, DL, MVT::i32);
+    X = SDValue(CurDAG->getMachineNode(TargetOpcode::INSERT_SUBREG, DL,
+                                       MVT::i32, ImplDef, X, SubRegIdx),
+                0);
+    insertDAGNode(*CurDAG, SDValue(Node, 0), X);
+  }
+
   // We might have matched the amount of high bits to be cleared,
   // but we want the amount of low bits to be kept, so negate it then.
   if (NegateNBits) {
-    SDValue BitWidthC = CurDAG->getConstant(NVT.getSizeInBits(), DL, MVT::i32);
+    SDValue BitWidthC =
+        CurDAG->getConstant(MatchedBitwidth, DL, MVT::i32);
     insertDAGNode(*CurDAG, SDValue(Node, 0), BitWidthC);
 
     NBits = CurDAG->getNode(ISD::SUB, DL, MVT::i32, BitWidthC, NBits);
@@ -4398,8 +4430,17 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
     }
 
     SDValue Extract = CurDAG->getNode(X86ISD::BZHI, DL, NVT, X, NBits);
+    if (OVT != NVT) {
+      SelectCode(Extract.getNode());
+      SDValue SubRegIdx = CurDAG->getTargetConstant(
+          OVT == MVT::i8 ? X86::sub_8bit : X86::sub_16bit, DL, MVT::i32);
+      Extract = SDValue(CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
+                                               OVT, Extract, SubRegIdx),
+                        0);
+    }
     ReplaceNode(Node, Extract.getNode());
-    SelectCode(Extract.getNode());
+    if (OVT == NVT)
+      SelectCode(Extract.getNode());
     return true;
   }
 
@@ -4464,8 +4505,18 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
     Extract = CurDAG->getNode(ISD::TRUNCATE, DL, NVT, Extract);
   }
 
+  if (OVT != NVT) {
+    SelectCode(Extract.getNode());
+    SDValue SubRegIdx = CurDAG->getTargetConstant(
+        OVT == MVT::i8 ? X86::sub_8bit : X86::sub_16bit, DL, MVT::i32);
+    Extract = SDValue(CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
+                                             OVT, Extract, SubRegIdx),
+                      0);
+  }
+
   ReplaceNode(Node, Extract.getNode());
-  SelectCode(Extract.getNode());
+  if (OVT == NVT)
+    SelectCode(Extract.getNode());
 
   return true;
 }
diff --git a/llvm/test/CodeGen/X86/clear-highbits.ll b/llvm/test/CodeGen/X86/clear-highbits.ll
index 755b1094234fd..81aa23282383a 100644
--- a/llvm/test/CodeGen/X86/clear-highbits.ll
+++ b/llvm/test/CodeGen/X86/clear-highbits.ll
@@ -22,46 +22,84 @@
 ; ---------------------------------------------------------------------------- ;
 
 define i8 @clear_highbits8_c0(i8 %val, i8 %numhighbits) nounwind {
-; X86-LABEL: clear_highbits8_c0:
-; X86:       # %bb.0:
-; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
-; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    shlb %cl, %al
-; X86-NEXT:    shrb %cl, %al
-; X86-NEXT:    retl
-;
-; X64-LABEL: clear_highbits8_c0:
-; X64:       # %bb.0:
-; X64-NEXT:    movl %esi, %ecx
-; X64-NEXT:    movl %edi, %eax
-; X64-NEXT:    shlb %cl, %al
-; X64-NEXT:    # kill: def $cl killed $cl killed $ecx
-; X64-NEXT:    shrb %cl, %al
-; X64-NEXT:    # kill: def $al killed $al killed $eax
-; X64-NEXT:    retq
+; X86-NOBMI2-LABEL: clear_highbits8_c0:
+; X86-NOBMI2:       # %bb.0:
+; X86-NOBMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NOBMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-NOBMI2-NEXT:    shlb %cl, %al
+; X86-NOBMI2-NEXT:    shrb %cl, %al
+; X86-NOBMI2-NEXT:    retl
+;
+; X86-BMI2-LABEL: clear_highbits8_c0:
+; X86-BMI2:       # %bb.0:
+; X86-BMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-BMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-BMI2-NEXT:    movl $8, %edx
+; X86-BMI2-NEXT:    subl %eax, %edx
+; X86-BMI2-NEXT:    bzhil %edx, %ecx, %eax
+; X86-BMI2-NEXT:    # kill: def $al killed $al killed $eax
+; X86-BMI2-NEXT:    retl
+;
+; X64-NOBMI2-LABEL: clear_highbits8_c0:
+; X64-NOBMI2:       # %bb.0:
+; X64-NOBMI2-NEXT:    movl %esi, %ecx
+; X64-NOBMI2-NEXT:    movl %edi, %eax
+; X64-NOBMI2-NEXT:    shlb %cl, %al
+; X64-NOBMI2-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI2-NEXT:    shrb %cl, %al
+; X64-NOBMI2-NEXT:    # kill: def $al killed $al killed $eax
+; X64-NOBMI2-NEXT:    retq
+;
+; X64-BMI2-LABEL: clear_highbits8_c0:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movl $8, %eax
+; X64-BMI2-NEXT:    subl %esi, %eax
+; X64-BMI2-NEXT:    bzhil %eax, %edi, %eax
+; X64-BMI2-NEXT:    # kill: def $al killed $al killed $eax
+; X64-BMI2-NEXT:    retq
   %mask = lshr i8 -1, %numhighbits
   %masked = and i8 %mask, %val
   ret i8 %masked
 }
 
 define i8 @clear_highbits8_c2_load(ptr %w, i8 %numhighbits) nounwind {
-; X86-LABEL: clear_highbits8_c2_load:
-; X86:       # %bb.0:
-; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    movzbl (%eax), %eax
-; X86-NEXT:    shlb %cl, %al
-; X86-NEXT:    shrb %cl, %al
-; X86-NEXT:    retl
-;
-; X64-LABEL: clear_highbits8_c2_load:
-; X64:       # %bb.0:
-; X64-NEXT:    movl %esi, %ecx
-; X64-NEXT:    movzbl (%rdi), %eax
-; X64-NEXT:    shlb %cl, %al
-; X64-NEXT:    # kill: def $cl killed $cl killed $ecx
-; X64-NEXT:    shrb %cl, %al
-; X64-NEXT:    retq
+; X86-NOBMI2-LABEL: clear_highbits8_c2_load:
+; X86-NOBMI2:       # %bb.0:
+; X86-NOBMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NOBMI2-NEXT:    movl {{[0-9]+}}(%esp), %eax
+; X86-NOBMI2-NEXT:    movzbl (%eax), %eax
+; X86-NOBMI2-NEXT:    shlb %cl, %al
+; X86-NOBMI2-NEXT:    shrb %cl, %al
+; X86-NOBMI2-NEXT:    retl
+;
+; X86-BMI2-LABEL: clear_highbits8_c2_load:
+; X86-BMI2:       # %bb.0:
+; X86-BMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-BMI2-NEXT:    movl {{[0-9]+}}(%esp), %ecx
+; X86-BMI2-NEXT:    movzbl (%ecx), %ecx
+; X86-BMI2-NEXT:    movl $8, %edx
+; X86-BMI2-NEXT:    subl %eax, %edx
+; X86-BMI2-NEXT:    bzhil %edx, %ecx, %eax
+; X86-BMI2-NEXT:    # kill: def $al killed $al killed $eax
+; X86-BMI2-NEXT:    retl
+;
+; X64-NOBMI2-LABEL: clear_highbits8_c2_load:
+; X64-NOBMI2:       # %bb.0:
+; X64-NOBMI2-NEXT:    movl %esi, %ecx
+; X64-NOBMI2-NEXT:    movzbl (%rdi), %eax
+; X64-NOBMI2-NEXT:    shlb %cl, %al
+; X64-NOBMI2-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI2-NEXT:    shrb %cl, %al
+; X64-NOBMI2-NEXT:    retq
+;
+; X64-BMI2-LABEL: clear_highbits8_c2_load:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movzbl (%rdi), %eax
+; X64-BMI2-NEXT:    movl $8, %ecx
+; X64-BMI2-NEXT:    subl %esi, %ecx
+; X64-BMI2-NEXT:    bzhil %ecx, %eax, %eax
+; X64-BMI2-NEXT:    # kill: def $al killed $al killed $eax
+; X64-BMI2-NEXT:    retq
   %val = load i8, ptr %w
   %mask = lshr i8 -1, %numhighbits
   %masked = and i8 %mask, %val
@@ -69,23 +107,41 @@ define i8 @clear_highbits8_c2_load(ptr %w, i8 %numhighbits) nounwind {
 }
 
 define i8 @clear_highbits8_c4_commutative(i8 %val, i8 %numhighbits) nounwind {
-; X86-LABEL: clear_highbits8_c4_commutative:
-; X86:       # %bb.0:
-; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
-; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    shlb %cl, %al
-; X86-NEXT:    shrb %cl, %al
-; X86-NEXT:    retl
-;
-; X64-LABEL: clear_highbits8_c4_commutative:
-; X64:       # %bb.0:
-; X64-NEXT:    movl %esi, %ecx
-; X64-NEXT:    movl %edi, %eax
-; X64-NEXT:    shlb %cl, %al
-; X64-NEXT:    # kill: def $cl killed $cl killed $ecx
-; X64-NEXT:    shrb %cl, %al
-; X64-NEXT:    # kill: def $al killed $al killed $eax
-; X64-NEXT:    retq
+; X86-NOBMI2-LABEL: clear_highbits8_c4_commutative:
+; X86-NOBMI2:       # %bb.0:
+; X86-NOBMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NOBMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-NOBMI2-NEXT:    shlb %cl, %al
+; X86-NOBMI2-NEXT:    shrb %cl, %al
+; X86-NOBMI2-NEXT:    retl
+;
+; X86-BMI2-LABEL: clear_highbits8_c4_commutative:
+; X86-BMI2:       # %bb.0:
+; X86-BMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-BMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-BMI2-NEXT:    movl $8, %edx
+; X86-BMI2-NEXT:    subl %eax, %edx
+; X86-BMI2-NEXT:    bzhil %edx, %ecx, %eax
+; X86-BMI2-NEXT:    # kill: def $al killed $al killed $eax
+; X86-BMI2-NEXT:    retl
+;
+; X64-NOBMI2-LABEL: clear_highbits8_c4_commutative:
+; X64-NOBMI2:       # %bb.0:
+; X64-NOBMI2-NEXT:    movl %esi, %ecx
+; X64-NOBMI2-NEXT:    movl %edi, %eax
+; X64-NOBMI2-NEXT:    shlb %cl, %al
+; X64-NOBMI2-NEXT:    # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI2-NEXT:    shrb %cl, %al
+; X64-NOBMI2-NEXT:    # kill: def $al killed $al killed $eax
+; X64-NOBMI2-NEXT:    retq
+;
+; X64-BMI2-LABEL: clear_highbits8_c4_commutative:
+; X64-BMI2:       # %bb.0:
+; X64-BMI2-NEXT:    movl $8, %eax
+; X64-BMI2-NEXT:    subl %esi, %eax
+; X64-BMI2-NEXT:    bzhil %eax, %edi, %eax
+; X64-BMI2-NEXT:    # kill: def $al killed $al killed $eax
+; X64-BMI2-NEXT:    retq
   %mask = lshr i8 -1, %numhighbits
   %masked = and i8 %val, %mask ; swapped order
   ret i8 %masked
@@ -109,9 +165,9 @@ define i16 @clear_highbits16_c0(i16 %val, i16 %numhighbits) nounwind {
 ; X86-BMI2-LABEL: clear_highbits16_c0:
 ; X86-BMI2:       # %bb.0:
 ; X86-BMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
-; X86-BMI2-NEXT:    shlxl %eax, {{[0-9]+}}(%esp), %ecx
-; X86-BMI2-NEXT:    movzwl %cx, %ecx
-; X86-BMI2-NEXT:    shrxl %eax, %ecx, %eax
+; X86-BMI2-NEXT:    movl $16, %ecx
+; X86-BMI2-NEXT:    subl %eax, %ecx
+; X86-BMI2-NEXT:    bzhil %ecx, {{[0-9]+}}(%esp), %eax
 ; X86-BMI2-NEXT:    # kill: def $ax killed $ax killed $eax
 ; X86-BMI2-NEXT:    retl
 ;
@@ -127,9 +183,9 @@ define i16 @clear_highbits16_c0(i16 %val, i16 %numhighbits) nounwind {
 ;
 ; X64-BMI2-LABEL: clear_highbits16_c0:
 ; X64-BMI2:       # %bb.0:
-; X64-BMI2-NEXT:    shlxl %esi, %edi, %eax
-; X64-BMI2-NEXT:    movzwl %ax, %eax
-; X64-BMI2-NEXT:    shrxl %esi, %eax, %eax
+; X64-BMI2-NEXT:    movl $16, %eax
+; X64-BMI2-NEXT:    subl %esi, %eax
+; X64-BMI2-NEXT:    bzhil %eax, %edi, %eax
 ; X64-BMI2-NEXT:    # kill: def $ax killed $ax killed $eax
 ; X64-BMI2-NEXT:    retq
   %mask = lshr i16 -1, %numhighbits
@@ -151,9 +207,9 @@ define i16 @clear_highbits16_c1_indexzext(i16 %val, i8 %numhighbits) nounwind {
 ; X86-BMI2-LABEL: clear_highbits16_c1_indexzext:
 ; X86-BMI2:       # %bb.0:
 ; X86-BMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
-; X86-BMI2-NEXT:    shlxl %eax, {{[0-9]+}}(%esp), %ecx
-; X86-BMI2-NEXT:    movzwl %cx, %ecx
-; X86-BMI2-NEXT:    shrxl %eax, %ecx, %eax
+; X86-BMI2-NEXT:    movl $16, %ecx
+; X86-BMI2-NEXT:    subl %eax, %ecx
+; X86-BMI2-NEXT:    bzhil %ecx, {{[0-9]+}}(%esp), %eax
 ; X86-BMI2-NEXT:    # kill: def $ax killed $ax killed $eax
 ; X86-BMI2-NEXT:    retl
 ;
@@ -169,9 +225,9 @@ define i16 @clear_highbits16_c1_indexzext(i16 %val, i8 %numhighbits) nounwind {
 ;
 ; X64-BMI2-LABEL: clear_highbits16_c1_indexzext:
 ; X64-BMI2:       # %bb.0:
-; X64-BMI2-NEXT:    shlxl %esi, %edi, %eax
-; X64-BMI2-NEXT:    movzwl %ax, %eax
-; X64-BMI2-NEXT:    shrxl %esi, %eax, %eax
+; X64-BMI2-NEXT:    movl $16, %eax
+; X64-BMI2-NEXT:    subl %esi, %eax
+; X64-BMI2-NEXT:    bzhil %eax, %edi, %eax
 ; X64-BMI2-NEXT:    # kill: def $ax killed $ax killed $eax
 ; X64-BMI2-NEXT:    retq
   %sh_prom = zext i8 %numhighbits to i16
@@ -197,9 +253,9 @@ define i16 @clear_highbits16_c2_load(ptr %w, i16 %numhighbits) nounwind {
 ; X86-BMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
 ; X86-BMI2-NEXT:    movl {{[0-9]+}}(%esp), %ecx
 ; X86-BMI2-NEXT:    movzwl (%ecx), %ecx
-; X86-BMI2-NEXT:    shlxl %eax, %ecx, %ecx
-; X86-BMI2-NEXT:    movzwl %cx, %ecx
-; X86-BMI2-NEXT:    shrxl %eax, %ecx, %eax
+; X86-BMI2-NEXT:    movl $16, %edx
+; X86-BMI2-NEXT:    subl %eax, %edx
+; X86-BMI2-NEXT:    bzhil %edx, %ecx, %eax
 ; X86-BMI2-NEXT:    # kill: def $ax killed $ax killed $eax
 ; X86-BMI2-NEXT:    retl
 ;
@@ -217,9 +273,9 @@ define i16 @clear_highbits16_c2_load(ptr %w, i16 %numhighbits) nounwind {
 ; X64-BMI2-LABEL: clear_highbits16_c2_load:
 ; X64-BMI2:       # %bb.0:
 ; X64-BMI2-NEXT:    movzwl (%rdi), %eax
-; X64-BMI2-NEXT:    shlxl %esi, %eax, %eax
-; X64-BMI2-NEXT:    movzwl %ax, %eax
-; X64-BMI2-NEXT:    shrxl %esi, %eax, %eax
+; X64-BMI2-NEXT:    movl $16, %ecx
+; X64-BMI2-NEXT:    subl %esi, %ecx
+; X64-BMI2-NEXT:    bzhil %ecx, %eax, %eax
 ; X64-BMI2-NEXT:    # kill: def $ax killed $ax killed $eax
 ; X64-BMI2-NEXT:    retq
   %val = load i16, ptr %w
@@ -245,9 +301,9 @@ define i16 @clear_highbits16_c3_load_indexzext(ptr %w, i8 %numhighbits) nounwind
 ; X86-BMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
 ; X86-BMI2-NEXT:    movl {{[0-9]+}}(%esp), %ecx
 ; X86-BMI2-NEXT:    movzwl (%ecx), %ecx
-; X86-BMI2-NEXT:    shlxl %eax, %ecx, %ecx
-; X86-BMI2-NEXT:    movzwl %cx, %ecx
-; X86-BMI2-NEXT:    shrxl %eax, %ecx, %eax
+; X86-BMI2-NEXT:    movl $16, %edx
+; X86-BMI2-NEXT:    subl %eax, %edx
+; X86-BMI2-NEXT:    bzhil %edx, %ecx, %eax
 ; X86-BMI2-NEXT:    # kill: def $ax killed $ax killed $eax
 ; X86-BMI2-NEXT:    retl
 ;
@@ -265,9 +321,9 @@ define i16 @clear_highbits16_c3_load_indexzext(ptr %w, i8 %numhighbits) nounwind
 ; X64-BMI2-LABEL: clear_highbits16_c3_load_indexzext:
 ; X64-BMI2:       # %bb.0:
 ; X64-BMI2-NEXT:    movzwl (%rdi), %eax
-; X64-BMI2-NEXT:    shlxl %esi, %eax, %eax
-; X64-BMI2-NEXT:    movzwl %ax, %eax
-; X64-BMI2-NEXT:    shrxl %esi, %eax, %eax
+; X64-BMI2-NEXT:    movl $16, %ecx
+; X64-BMI2-NEXT:    subl %esi, %ecx
+; X64-BMI2-NEXT:    bzhil %ecx, %eax, %eax
 ; X64-BMI2-NEXT:    # kill: def $ax killed $ax killed $eax
 ; X64-BMI2-NEXT:    retq
   %val = load i16, ptr %w
@@ -291,9 +347,9 @@ define i16 @clear_highbits16_c4_commutative(i16 %val, i16 %numhighbits) nounwind
 ; X86-BMI2-LABEL: clear_highbits16_c4_commutative:
 ; X86-BMI2:       # %bb.0:
 ; X86-BMI2-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
-; X86-BMI2-NEXT:    shlxl %eax, {{[0-9]+}}(%esp), %ecx
-; X86-BMI2-NEXT:    movzwl %cx, %ecx
-; X86-BMI2-NEXT:    shrxl %eax, %ecx, %eax
+; X86-BMI2-NEXT:    movl $16, %ecx
+; X86-BMI2-NEXT:    subl %eax, %ecx
+; X86-BMI2-NEXT:    bzhil %ecx, {{[0-9]+}}(%esp), %eax
 ; X86-BMI2-NEXT:    # kill: def $ax killed $ax killed $eax
 ; X86-BMI2-NEXT:    retl
 ;
@@ -309,9 +365,9 @@ define i16 @clear_highbits16_c4_commutative(i16 %val, i16 %numhighbits) nounwind
 ;
 ; X64-BMI2-LABEL: clear_highbits16_c4_commutative:
 ; X64-BMI2:       # %bb.0:
-; X64-BMI2-NEXT:    shlxl %esi, %edi, %eax
-; X64-BMI2-NEXT:    movzwl %ax, %eax
-; X64-BMI2-NEXT:    shrxl %esi, %eax, %eax
+; X64-BMI2-NEXT:    movl $16, %eax
+; X64-BMI2-NEXT:    subl %esi, %eax
+; X64-BMI2-NEXT:    bzhil %eax, %edi, %eax
 ; X64-BMI2-NEXT:    # kill: def $ax killed $ax killed $eax
 ; X64-BMI2-NEXT:    retq
   %mask = lshr i16 -1, %numhighbits
@@ -1392,3 +1448,6 @@ define i32 @clear_highbits32_48_extrause(i32 %val, i32 %numlowbits, ptr %escape)
   %masked = and i32 %mask, %val
   ret i32 %masked
 }
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; X64: {{.*}}
+; X86: {{.*}}

``````````

</details>


https://github.com/llvm/llvm-project/pull/226401


More information about the llvm-commits mailing list