[llvm] [X86] Fix failure to lower i8/i16 bzhi pattern to bzhi (PR #226401)
Divyansh Yadav via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 25 01:56:28 PDT 2026
https://github.com/schizophrenicmaniac created https://github.com/llvm/llvm-project/pull/226401
This PR fixes an issue where `DAGToDAGISel` failed to lower bit extraction patterns to the `bzhi` instruction for 8-bit and 16-bit integer types (`i8` and `i16`), even though it already handled them correctly for `i32` and `i64`.
**Root Cause**:
For `i8` and `i16`, the `DAGCombiner` transforms shift combinations by promoting them to `i32` shifts and inserting an explicit `AND` mask (e.g., `AND 255` or `AND 65535`) to simulate the narrower type. This caused the pattern matching in `X86ISelDAGToDAG::matchBitExtract` to fail because it encountered the `AND` node instead of the expected `SHL` node. Additionally, the existing logic wasn't equipped to insert/extract subregisters to use the natively 32-bit `bzhi` instruction for narrower types.
**Fix**:
1. Updated `matchPatternD` in `X86ISelDAGToDAG.cpp` to look past the `AND 255` and `AND 65535` masks that the `DAGCombiner` inserts for promoted `i8`/`i16` shifts.
2. Updated `peekThroughOneUseTruncation` to assert that the source size is strictly greater than the destination size, allowing for truncations involving `i8` and `i16`.
3. Adjusted the `matchBitExtract` logic to support `i8` and `i16` values by dynamically wrapping the operations with `INSERT_SUBREG` to place them into an `i32` register for `BZHI32`, and extracting the result back out using `EXTRACT_SUBREG`.
4. Updated `clear-highbits.ll` using `update_llc_test_checks.py` to assert that we are now properly generating the `bzhi` instruction for `i8` and `i16` types.
Fixes #226376.
>From e99da6ebccb6b4f58b531d24e9b83c642a3fbb68 Mon Sep 17 00:00:00 2001
From: Divyansh <anshmcs at gmail.com>
Date: Fri, 25 Sep 2026 14:23:55 +0530
Subject: [PATCH] [X86] Fix failure to lower i8/i16 bzhi pattern to bzhi
---
llvm/lib/Target/X86/X86ISelDAGToDAG.cpp | 73 ++++++--
llvm/test/CodeGen/X86/clear-highbits.ll | 221 +++++++++++++++---------
2 files changed, 202 insertions(+), 92 deletions(-)
diff --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index c2de8283f15ff8..75dd37d52a2e6e 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -4158,12 +4158,17 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
MVT NVT = Node->getSimpleValueType(0);
- // Only supported for 32 and 64 bits.
- if (NVT != MVT::i32 && NVT != MVT::i64)
+ // Only supported for 8, 16, 32 and 64 bits.
+ if (NVT != MVT::i8 && NVT != MVT::i16 && NVT != MVT::i32 && NVT != MVT::i64)
return false;
SDValue NBits;
bool NegateNBits;
+ // The bitwidth of the value that the matched pattern operates on. This can
+ // be narrower than the node's type when the pattern was widened during type
+ // legalization, e.g. an i16 pattern that is visible as an i32 node with an
+ // explicit 0xFFFF mask.
+ unsigned MatchedBitwidth = NVT.getSizeInBits();
// If we have BMI2's BZHI, we are ok with muti-use patterns.
// Else, if we only have BMI1's BEXTR, we require one-use.
@@ -4187,9 +4192,9 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
auto peekThroughOneUseTruncation = [checkOneUse](SDValue V) {
if (V->getOpcode() == ISD::TRUNCATE && checkOneUse(V)) {
- assert(V.getSimpleValueType() == MVT::i32 &&
- V.getOperand(0).getSimpleValueType() == MVT::i64 &&
- "Expected i64 -> i32 truncation");
+ assert(V.getSimpleValueType().getSizeInBits() <
+ V.getOperand(0).getSimpleValueType().getSizeInBits() &&
+ "Expected truncation");
V = V.getOperand(0);
}
return V;
@@ -4245,8 +4250,10 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
// Try to match potentially-truncated shift amount as `(bitwidth - y)`,
// or leave the shift amount as-is, but then we'll have to negate it.
- auto canonicalizeShiftAmt = [&NBits, &NegateNBits](SDValue ShiftAmt,
- unsigned Bitwidth) {
+ auto canonicalizeShiftAmt = [&NBits, &NegateNBits,
+ &MatchedBitwidth](SDValue ShiftAmt,
+ unsigned Bitwidth) {
+ MatchedBitwidth = Bitwidth;
NBits = ShiftAmt;
NegateNBits = true;
// Skip over a truncate of the shift amount, if any.
@@ -4300,9 +4307,20 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
if (Node->getOpcode() != ISD::SRL)
return false;
SDValue N0 = Node->getOperand(0);
+ unsigned Bitwidth = N0.getSimpleValueType().getSizeInBits();
+ if (N0->getOpcode() == ISD::AND) {
+ if (auto *C = dyn_cast<ConstantSDNode>(N0->getOperand(1))) {
+ if (C->getZExtValue() == 65535) {
+ Bitwidth = 16;
+ N0 = N0->getOperand(0);
+ } else if (C->getZExtValue() == 255) {
+ Bitwidth = 8;
+ N0 = N0->getOperand(0);
+ }
+ }
+ }
if (N0->getOpcode() != ISD::SHL)
return false;
- unsigned Bitwidth = N0.getSimpleValueType().getSizeInBits();
SDValue N1 = Node->getOperand(1);
SDValue N01 = N0->getOperand(1);
// Both of the shifts must be by the exact same value.
@@ -4379,10 +4397,24 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
insertDAGNode(*CurDAG, SDValue(Node, 0), NBits);
}
+ MVT OVT = NVT;
+ if (NVT == MVT::i8 || NVT == MVT::i16) {
+ NVT = MVT::i32;
+ SDValue ImplDef = SDValue(
+ CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i32), 0);
+ SDValue SubRegIdx = CurDAG->getTargetConstant(
+ OVT == MVT::i8 ? X86::sub_8bit : X86::sub_16bit, DL, MVT::i32);
+ X = SDValue(CurDAG->getMachineNode(TargetOpcode::INSERT_SUBREG, DL,
+ MVT::i32, ImplDef, X, SubRegIdx),
+ 0);
+ insertDAGNode(*CurDAG, SDValue(Node, 0), X);
+ }
+
// We might have matched the amount of high bits to be cleared,
// but we want the amount of low bits to be kept, so negate it then.
if (NegateNBits) {
- SDValue BitWidthC = CurDAG->getConstant(NVT.getSizeInBits(), DL, MVT::i32);
+ SDValue BitWidthC =
+ CurDAG->getConstant(MatchedBitwidth, DL, MVT::i32);
insertDAGNode(*CurDAG, SDValue(Node, 0), BitWidthC);
NBits = CurDAG->getNode(ISD::SUB, DL, MVT::i32, BitWidthC, NBits);
@@ -4398,8 +4430,17 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
}
SDValue Extract = CurDAG->getNode(X86ISD::BZHI, DL, NVT, X, NBits);
+ if (OVT != NVT) {
+ SelectCode(Extract.getNode());
+ SDValue SubRegIdx = CurDAG->getTargetConstant(
+ OVT == MVT::i8 ? X86::sub_8bit : X86::sub_16bit, DL, MVT::i32);
+ Extract = SDValue(CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
+ OVT, Extract, SubRegIdx),
+ 0);
+ }
ReplaceNode(Node, Extract.getNode());
- SelectCode(Extract.getNode());
+ if (OVT == NVT)
+ SelectCode(Extract.getNode());
return true;
}
@@ -4464,8 +4505,18 @@ bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
Extract = CurDAG->getNode(ISD::TRUNCATE, DL, NVT, Extract);
}
+ if (OVT != NVT) {
+ SelectCode(Extract.getNode());
+ SDValue SubRegIdx = CurDAG->getTargetConstant(
+ OVT == MVT::i8 ? X86::sub_8bit : X86::sub_16bit, DL, MVT::i32);
+ Extract = SDValue(CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
+ OVT, Extract, SubRegIdx),
+ 0);
+ }
+
ReplaceNode(Node, Extract.getNode());
- SelectCode(Extract.getNode());
+ if (OVT == NVT)
+ SelectCode(Extract.getNode());
return true;
}
diff --git a/llvm/test/CodeGen/X86/clear-highbits.ll b/llvm/test/CodeGen/X86/clear-highbits.ll
index 755b1094234fd8..81aa23282383ab 100644
--- a/llvm/test/CodeGen/X86/clear-highbits.ll
+++ b/llvm/test/CodeGen/X86/clear-highbits.ll
@@ -22,46 +22,84 @@
; ---------------------------------------------------------------------------- ;
define i8 @clear_highbits8_c0(i8 %val, i8 %numhighbits) nounwind {
-; X86-LABEL: clear_highbits8_c0:
-; X86: # %bb.0:
-; X86-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X86-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X86-NEXT: shlb %cl, %al
-; X86-NEXT: shrb %cl, %al
-; X86-NEXT: retl
-;
-; X64-LABEL: clear_highbits8_c0:
-; X64: # %bb.0:
-; X64-NEXT: movl %esi, %ecx
-; X64-NEXT: movl %edi, %eax
-; X64-NEXT: shlb %cl, %al
-; X64-NEXT: # kill: def $cl killed $cl killed $ecx
-; X64-NEXT: shrb %cl, %al
-; X64-NEXT: # kill: def $al killed $al killed $eax
-; X64-NEXT: retq
+; X86-NOBMI2-LABEL: clear_highbits8_c0:
+; X86-NOBMI2: # %bb.0:
+; X86-NOBMI2-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NOBMI2-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X86-NOBMI2-NEXT: shlb %cl, %al
+; X86-NOBMI2-NEXT: shrb %cl, %al
+; X86-NOBMI2-NEXT: retl
+;
+; X86-BMI2-LABEL: clear_highbits8_c0:
+; X86-BMI2: # %bb.0:
+; X86-BMI2-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X86-BMI2-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-BMI2-NEXT: movl $8, %edx
+; X86-BMI2-NEXT: subl %eax, %edx
+; X86-BMI2-NEXT: bzhil %edx, %ecx, %eax
+; X86-BMI2-NEXT: # kill: def $al killed $al killed $eax
+; X86-BMI2-NEXT: retl
+;
+; X64-NOBMI2-LABEL: clear_highbits8_c0:
+; X64-NOBMI2: # %bb.0:
+; X64-NOBMI2-NEXT: movl %esi, %ecx
+; X64-NOBMI2-NEXT: movl %edi, %eax
+; X64-NOBMI2-NEXT: shlb %cl, %al
+; X64-NOBMI2-NEXT: # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI2-NEXT: shrb %cl, %al
+; X64-NOBMI2-NEXT: # kill: def $al killed $al killed $eax
+; X64-NOBMI2-NEXT: retq
+;
+; X64-BMI2-LABEL: clear_highbits8_c0:
+; X64-BMI2: # %bb.0:
+; X64-BMI2-NEXT: movl $8, %eax
+; X64-BMI2-NEXT: subl %esi, %eax
+; X64-BMI2-NEXT: bzhil %eax, %edi, %eax
+; X64-BMI2-NEXT: # kill: def $al killed $al killed $eax
+; X64-BMI2-NEXT: retq
%mask = lshr i8 -1, %numhighbits
%masked = and i8 %mask, %val
ret i8 %masked
}
define i8 @clear_highbits8_c2_load(ptr %w, i8 %numhighbits) nounwind {
-; X86-LABEL: clear_highbits8_c2_load:
-; X86: # %bb.0:
-; X86-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X86-NEXT: movl {{[0-9]+}}(%esp), %eax
-; X86-NEXT: movzbl (%eax), %eax
-; X86-NEXT: shlb %cl, %al
-; X86-NEXT: shrb %cl, %al
-; X86-NEXT: retl
-;
-; X64-LABEL: clear_highbits8_c2_load:
-; X64: # %bb.0:
-; X64-NEXT: movl %esi, %ecx
-; X64-NEXT: movzbl (%rdi), %eax
-; X64-NEXT: shlb %cl, %al
-; X64-NEXT: # kill: def $cl killed $cl killed $ecx
-; X64-NEXT: shrb %cl, %al
-; X64-NEXT: retq
+; X86-NOBMI2-LABEL: clear_highbits8_c2_load:
+; X86-NOBMI2: # %bb.0:
+; X86-NOBMI2-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NOBMI2-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X86-NOBMI2-NEXT: movzbl (%eax), %eax
+; X86-NOBMI2-NEXT: shlb %cl, %al
+; X86-NOBMI2-NEXT: shrb %cl, %al
+; X86-NOBMI2-NEXT: retl
+;
+; X86-BMI2-LABEL: clear_highbits8_c2_load:
+; X86-BMI2: # %bb.0:
+; X86-BMI2-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X86-BMI2-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; X86-BMI2-NEXT: movzbl (%ecx), %ecx
+; X86-BMI2-NEXT: movl $8, %edx
+; X86-BMI2-NEXT: subl %eax, %edx
+; X86-BMI2-NEXT: bzhil %edx, %ecx, %eax
+; X86-BMI2-NEXT: # kill: def $al killed $al killed $eax
+; X86-BMI2-NEXT: retl
+;
+; X64-NOBMI2-LABEL: clear_highbits8_c2_load:
+; X64-NOBMI2: # %bb.0:
+; X64-NOBMI2-NEXT: movl %esi, %ecx
+; X64-NOBMI2-NEXT: movzbl (%rdi), %eax
+; X64-NOBMI2-NEXT: shlb %cl, %al
+; X64-NOBMI2-NEXT: # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI2-NEXT: shrb %cl, %al
+; X64-NOBMI2-NEXT: retq
+;
+; X64-BMI2-LABEL: clear_highbits8_c2_load:
+; X64-BMI2: # %bb.0:
+; X64-BMI2-NEXT: movzbl (%rdi), %eax
+; X64-BMI2-NEXT: movl $8, %ecx
+; X64-BMI2-NEXT: subl %esi, %ecx
+; X64-BMI2-NEXT: bzhil %ecx, %eax, %eax
+; X64-BMI2-NEXT: # kill: def $al killed $al killed $eax
+; X64-BMI2-NEXT: retq
%val = load i8, ptr %w
%mask = lshr i8 -1, %numhighbits
%masked = and i8 %mask, %val
@@ -69,23 +107,41 @@ define i8 @clear_highbits8_c2_load(ptr %w, i8 %numhighbits) nounwind {
}
define i8 @clear_highbits8_c4_commutative(i8 %val, i8 %numhighbits) nounwind {
-; X86-LABEL: clear_highbits8_c4_commutative:
-; X86: # %bb.0:
-; X86-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
-; X86-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X86-NEXT: shlb %cl, %al
-; X86-NEXT: shrb %cl, %al
-; X86-NEXT: retl
-;
-; X64-LABEL: clear_highbits8_c4_commutative:
-; X64: # %bb.0:
-; X64-NEXT: movl %esi, %ecx
-; X64-NEXT: movl %edi, %eax
-; X64-NEXT: shlb %cl, %al
-; X64-NEXT: # kill: def $cl killed $cl killed $ecx
-; X64-NEXT: shrb %cl, %al
-; X64-NEXT: # kill: def $al killed $al killed $eax
-; X64-NEXT: retq
+; X86-NOBMI2-LABEL: clear_highbits8_c4_commutative:
+; X86-NOBMI2: # %bb.0:
+; X86-NOBMI2-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NOBMI2-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X86-NOBMI2-NEXT: shlb %cl, %al
+; X86-NOBMI2-NEXT: shrb %cl, %al
+; X86-NOBMI2-NEXT: retl
+;
+; X86-BMI2-LABEL: clear_highbits8_c4_commutative:
+; X86-BMI2: # %bb.0:
+; X86-BMI2-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X86-BMI2-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-BMI2-NEXT: movl $8, %edx
+; X86-BMI2-NEXT: subl %eax, %edx
+; X86-BMI2-NEXT: bzhil %edx, %ecx, %eax
+; X86-BMI2-NEXT: # kill: def $al killed $al killed $eax
+; X86-BMI2-NEXT: retl
+;
+; X64-NOBMI2-LABEL: clear_highbits8_c4_commutative:
+; X64-NOBMI2: # %bb.0:
+; X64-NOBMI2-NEXT: movl %esi, %ecx
+; X64-NOBMI2-NEXT: movl %edi, %eax
+; X64-NOBMI2-NEXT: shlb %cl, %al
+; X64-NOBMI2-NEXT: # kill: def $cl killed $cl killed $ecx
+; X64-NOBMI2-NEXT: shrb %cl, %al
+; X64-NOBMI2-NEXT: # kill: def $al killed $al killed $eax
+; X64-NOBMI2-NEXT: retq
+;
+; X64-BMI2-LABEL: clear_highbits8_c4_commutative:
+; X64-BMI2: # %bb.0:
+; X64-BMI2-NEXT: movl $8, %eax
+; X64-BMI2-NEXT: subl %esi, %eax
+; X64-BMI2-NEXT: bzhil %eax, %edi, %eax
+; X64-BMI2-NEXT: # kill: def $al killed $al killed $eax
+; X64-BMI2-NEXT: retq
%mask = lshr i8 -1, %numhighbits
%masked = and i8 %val, %mask ; swapped order
ret i8 %masked
@@ -109,9 +165,9 @@ define i16 @clear_highbits16_c0(i16 %val, i16 %numhighbits) nounwind {
; X86-BMI2-LABEL: clear_highbits16_c0:
; X86-BMI2: # %bb.0:
; X86-BMI2-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X86-BMI2-NEXT: shlxl %eax, {{[0-9]+}}(%esp), %ecx
-; X86-BMI2-NEXT: movzwl %cx, %ecx
-; X86-BMI2-NEXT: shrxl %eax, %ecx, %eax
+; X86-BMI2-NEXT: movl $16, %ecx
+; X86-BMI2-NEXT: subl %eax, %ecx
+; X86-BMI2-NEXT: bzhil %ecx, {{[0-9]+}}(%esp), %eax
; X86-BMI2-NEXT: # kill: def $ax killed $ax killed $eax
; X86-BMI2-NEXT: retl
;
@@ -127,9 +183,9 @@ define i16 @clear_highbits16_c0(i16 %val, i16 %numhighbits) nounwind {
;
; X64-BMI2-LABEL: clear_highbits16_c0:
; X64-BMI2: # %bb.0:
-; X64-BMI2-NEXT: shlxl %esi, %edi, %eax
-; X64-BMI2-NEXT: movzwl %ax, %eax
-; X64-BMI2-NEXT: shrxl %esi, %eax, %eax
+; X64-BMI2-NEXT: movl $16, %eax
+; X64-BMI2-NEXT: subl %esi, %eax
+; X64-BMI2-NEXT: bzhil %eax, %edi, %eax
; X64-BMI2-NEXT: # kill: def $ax killed $ax killed $eax
; X64-BMI2-NEXT: retq
%mask = lshr i16 -1, %numhighbits
@@ -151,9 +207,9 @@ define i16 @clear_highbits16_c1_indexzext(i16 %val, i8 %numhighbits) nounwind {
; X86-BMI2-LABEL: clear_highbits16_c1_indexzext:
; X86-BMI2: # %bb.0:
; X86-BMI2-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X86-BMI2-NEXT: shlxl %eax, {{[0-9]+}}(%esp), %ecx
-; X86-BMI2-NEXT: movzwl %cx, %ecx
-; X86-BMI2-NEXT: shrxl %eax, %ecx, %eax
+; X86-BMI2-NEXT: movl $16, %ecx
+; X86-BMI2-NEXT: subl %eax, %ecx
+; X86-BMI2-NEXT: bzhil %ecx, {{[0-9]+}}(%esp), %eax
; X86-BMI2-NEXT: # kill: def $ax killed $ax killed $eax
; X86-BMI2-NEXT: retl
;
@@ -169,9 +225,9 @@ define i16 @clear_highbits16_c1_indexzext(i16 %val, i8 %numhighbits) nounwind {
;
; X64-BMI2-LABEL: clear_highbits16_c1_indexzext:
; X64-BMI2: # %bb.0:
-; X64-BMI2-NEXT: shlxl %esi, %edi, %eax
-; X64-BMI2-NEXT: movzwl %ax, %eax
-; X64-BMI2-NEXT: shrxl %esi, %eax, %eax
+; X64-BMI2-NEXT: movl $16, %eax
+; X64-BMI2-NEXT: subl %esi, %eax
+; X64-BMI2-NEXT: bzhil %eax, %edi, %eax
; X64-BMI2-NEXT: # kill: def $ax killed $ax killed $eax
; X64-BMI2-NEXT: retq
%sh_prom = zext i8 %numhighbits to i16
@@ -197,9 +253,9 @@ define i16 @clear_highbits16_c2_load(ptr %w, i16 %numhighbits) nounwind {
; X86-BMI2-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X86-BMI2-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X86-BMI2-NEXT: movzwl (%ecx), %ecx
-; X86-BMI2-NEXT: shlxl %eax, %ecx, %ecx
-; X86-BMI2-NEXT: movzwl %cx, %ecx
-; X86-BMI2-NEXT: shrxl %eax, %ecx, %eax
+; X86-BMI2-NEXT: movl $16, %edx
+; X86-BMI2-NEXT: subl %eax, %edx
+; X86-BMI2-NEXT: bzhil %edx, %ecx, %eax
; X86-BMI2-NEXT: # kill: def $ax killed $ax killed $eax
; X86-BMI2-NEXT: retl
;
@@ -217,9 +273,9 @@ define i16 @clear_highbits16_c2_load(ptr %w, i16 %numhighbits) nounwind {
; X64-BMI2-LABEL: clear_highbits16_c2_load:
; X64-BMI2: # %bb.0:
; X64-BMI2-NEXT: movzwl (%rdi), %eax
-; X64-BMI2-NEXT: shlxl %esi, %eax, %eax
-; X64-BMI2-NEXT: movzwl %ax, %eax
-; X64-BMI2-NEXT: shrxl %esi, %eax, %eax
+; X64-BMI2-NEXT: movl $16, %ecx
+; X64-BMI2-NEXT: subl %esi, %ecx
+; X64-BMI2-NEXT: bzhil %ecx, %eax, %eax
; X64-BMI2-NEXT: # kill: def $ax killed $ax killed $eax
; X64-BMI2-NEXT: retq
%val = load i16, ptr %w
@@ -245,9 +301,9 @@ define i16 @clear_highbits16_c3_load_indexzext(ptr %w, i8 %numhighbits) nounwind
; X86-BMI2-NEXT: movzbl {{[0-9]+}}(%esp), %eax
; X86-BMI2-NEXT: movl {{[0-9]+}}(%esp), %ecx
; X86-BMI2-NEXT: movzwl (%ecx), %ecx
-; X86-BMI2-NEXT: shlxl %eax, %ecx, %ecx
-; X86-BMI2-NEXT: movzwl %cx, %ecx
-; X86-BMI2-NEXT: shrxl %eax, %ecx, %eax
+; X86-BMI2-NEXT: movl $16, %edx
+; X86-BMI2-NEXT: subl %eax, %edx
+; X86-BMI2-NEXT: bzhil %edx, %ecx, %eax
; X86-BMI2-NEXT: # kill: def $ax killed $ax killed $eax
; X86-BMI2-NEXT: retl
;
@@ -265,9 +321,9 @@ define i16 @clear_highbits16_c3_load_indexzext(ptr %w, i8 %numhighbits) nounwind
; X64-BMI2-LABEL: clear_highbits16_c3_load_indexzext:
; X64-BMI2: # %bb.0:
; X64-BMI2-NEXT: movzwl (%rdi), %eax
-; X64-BMI2-NEXT: shlxl %esi, %eax, %eax
-; X64-BMI2-NEXT: movzwl %ax, %eax
-; X64-BMI2-NEXT: shrxl %esi, %eax, %eax
+; X64-BMI2-NEXT: movl $16, %ecx
+; X64-BMI2-NEXT: subl %esi, %ecx
+; X64-BMI2-NEXT: bzhil %ecx, %eax, %eax
; X64-BMI2-NEXT: # kill: def $ax killed $ax killed $eax
; X64-BMI2-NEXT: retq
%val = load i16, ptr %w
@@ -291,9 +347,9 @@ define i16 @clear_highbits16_c4_commutative(i16 %val, i16 %numhighbits) nounwind
; X86-BMI2-LABEL: clear_highbits16_c4_commutative:
; X86-BMI2: # %bb.0:
; X86-BMI2-NEXT: movzbl {{[0-9]+}}(%esp), %eax
-; X86-BMI2-NEXT: shlxl %eax, {{[0-9]+}}(%esp), %ecx
-; X86-BMI2-NEXT: movzwl %cx, %ecx
-; X86-BMI2-NEXT: shrxl %eax, %ecx, %eax
+; X86-BMI2-NEXT: movl $16, %ecx
+; X86-BMI2-NEXT: subl %eax, %ecx
+; X86-BMI2-NEXT: bzhil %ecx, {{[0-9]+}}(%esp), %eax
; X86-BMI2-NEXT: # kill: def $ax killed $ax killed $eax
; X86-BMI2-NEXT: retl
;
@@ -309,9 +365,9 @@ define i16 @clear_highbits16_c4_commutative(i16 %val, i16 %numhighbits) nounwind
;
; X64-BMI2-LABEL: clear_highbits16_c4_commutative:
; X64-BMI2: # %bb.0:
-; X64-BMI2-NEXT: shlxl %esi, %edi, %eax
-; X64-BMI2-NEXT: movzwl %ax, %eax
-; X64-BMI2-NEXT: shrxl %esi, %eax, %eax
+; X64-BMI2-NEXT: movl $16, %eax
+; X64-BMI2-NEXT: subl %esi, %eax
+; X64-BMI2-NEXT: bzhil %eax, %edi, %eax
; X64-BMI2-NEXT: # kill: def $ax killed $ax killed $eax
; X64-BMI2-NEXT: retq
%mask = lshr i16 -1, %numhighbits
@@ -1392,3 +1448,6 @@ define i32 @clear_highbits32_48_extrause(i32 %val, i32 %numlowbits, ptr %escape)
%masked = and i32 %mask, %val
ret i32 %masked
}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; X64: {{.*}}
+; X86: {{.*}}
More information about the llvm-commits
mailing list