[llvm] c6c2ad5 - [DAG] Fold any-extend(and(trunc(x), C)) -> and(x, C) (#200052)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Jun 3 08:48:46 PDT 2026
Author: Aayush Shrivastava
Date: 2026-06-03T17:48:40+02:00
New Revision: c6c2ad52586c28e74a724570985a48dc0022d250
URL: https://github.com/llvm/llvm-project/commit/c6c2ad52586c28e74a724570985a48dc0022d250
DIFF: https://github.com/llvm/llvm-project/commit/c6c2ad52586c28e74a724570985a48dc0022d250.diff
LOG: [DAG] Fold any-extend(and(trunc(x), C)) -> and(x, C) (#200052)
Fixes #195575
Fix a missed optimization in `DAGCombiner::visitANY_EXTEND` where the
pattern `any-extend(and(trunc(x), C))` was not being folded into `and(x,
C)` on X86, causing a redundant `movzbl` instruction to be emitted after
a small-mask AND.
Added:
llvm/test/CodeGen/X86/aext-and-trunc-avx512.ll
llvm/test/CodeGen/X86/aext-and-trunc.ll
Modified:
llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
llvm/test/CodeGen/AArch64/bitfield-insert.ll
llvm/test/CodeGen/AArch64/bswap-known-bits.ll
llvm/test/CodeGen/AArch64/sshl_sat.ll
llvm/test/CodeGen/AArch64/ushl_sat.ll
llvm/test/CodeGen/X86/and-with-overflow.ll
llvm/test/CodeGen/X86/combine-srem.ll
llvm/test/CodeGen/X86/llvm.frexp.ll
llvm/test/CodeGen/X86/masked_load.ll
llvm/test/CodeGen/X86/masked_store.ll
llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-innerouter.ll
llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-interleavedbits.ll
llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-interleavedbytehalves.ll
llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-lowhigh.ll
Removed:
################################################################################
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index c0f99d9e03c60..530ba5f49824e 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -15891,16 +15891,35 @@ SDValue DAGCombiner::visitZERO_EXTEND(SDNode *N) {
// Fold (zext (and (trunc x), cst)) -> (and x, cst),
// if either of the casts is not free.
+ // Also handles (zext (and (bitcast (extract_subvector vNi1, 0)) cst))
+ // by treating the bitcast+extract as equivalent to a truncate of the
+ // wider bitcast, e.g. on AVX512DQ where v8i1 extract replaces truncate.
if (N0.getOpcode() == ISD::AND &&
- N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
- N0.getOperand(1).getOpcode() == ISD::Constant &&
- (!TLI.isTruncateFree(N0.getOperand(0).getOperand(0), N0.getValueType()) ||
- !TLI.isZExtFree(N0.getValueType(), VT))) {
- SDValue X = N0.getOperand(0).getOperand(0);
- X = DAG.getAnyExtOrTrunc(X, SDLoc(X), VT);
- APInt Mask = N0.getConstantOperandAPInt(1).zext(VT.getSizeInBits());
- return DAG.getNode(ISD::AND, DL, VT,
- X, DAG.getConstant(Mask, DL, VT));
+ N0.getOperand(1).getOpcode() == ISD::Constant) {
+ SDValue AndSrc = N0.getOperand(0);
+ SDValue X;
+ if (AndSrc.getOpcode() == ISD::TRUNCATE) {
+ X = AndSrc.getOperand(0);
+ } else if (AndSrc.getOpcode() == ISD::BITCAST &&
+ AndSrc.getOperand(0).getOpcode() == ISD::EXTRACT_SUBVECTOR &&
+ AndSrc.getOperand(0).getConstantOperandVal(1) == 0) {
+ // (bitcast (extract_subvector vNi1, 0) -> iK) is equivalent to
+ // (truncate (bitcast vNi1 -> iN) -> iK); use the wider vNi1 as X.
+ SDValue Src = AndSrc.getOperand(0).getOperand(0);
+ EVT SrcVT = Src.getValueType();
+ if (SrcVT.isFixedLengthVectorOf(MVT::i1)) {
+ EVT WideIntVT =
+ EVT::getIntegerVT(*DAG.getContext(), SrcVT.getSizeInBits());
+ if (TLI.isTypeLegal(WideIntVT))
+ X = DAG.getBitcast(WideIntVT, Src);
+ }
+ }
+ if (X && (!TLI.isTruncateFree(X, N0.getValueType()) ||
+ !TLI.isZExtFree(N0.getValueType(), VT))) {
+ X = DAG.getAnyExtOrTrunc(X, SDLoc(X), VT);
+ APInt Mask = N0.getConstantOperandAPInt(1).zext(VT.getSizeInBits());
+ return DAG.getNode(ISD::AND, DL, VT, X, DAG.getConstant(Mask, DL, VT));
+ }
}
// Try to simplify (zext (load x)).
@@ -16144,15 +16163,36 @@ SDValue DAGCombiner::visitANY_EXTEND(SDNode *N) {
return DAG.getAnyExtOrTrunc(N0.getOperand(0), DL, VT);
// Fold (aext (and (trunc x), cst)) -> (and x, cst)
- // if the trunc is not free.
+ // if either of the casts is not free, and sign-extending the narrow type is
+ // not cheaper than zero-extending it (which would indicate the target prefers
+ // to keep operations at the narrower width).
+ // Also handles (aext (and (bitcast (extract_subvector vNi1, 0)) cst))
+ // which arises on AVX512DQ where v8i1 extract replaces truncate.
if (N0.getOpcode() == ISD::AND &&
- N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
- N0.getOperand(1).getOpcode() == ISD::Constant &&
- !TLI.isTruncateFree(N0.getOperand(0).getOperand(0), N0.getValueType())) {
- SDValue X = DAG.getAnyExtOrTrunc(N0.getOperand(0).getOperand(0), DL, VT);
- SDValue Y = DAG.getNode(ISD::ANY_EXTEND, DL, VT, N0.getOperand(1));
- assert(isa<ConstantSDNode>(Y) && "Expected constant to be folded!");
- return DAG.getNode(ISD::AND, DL, VT, X, Y);
+ N0.getOperand(1).getOpcode() == ISD::Constant) {
+ SDValue AndSrc = N0.getOperand(0);
+ SDValue X;
+ if (AndSrc.getOpcode() == ISD::TRUNCATE) {
+ X = AndSrc.getOperand(0);
+ } else if (AndSrc.getOpcode() == ISD::BITCAST &&
+ AndSrc.getOperand(0).getOpcode() == ISD::EXTRACT_SUBVECTOR &&
+ AndSrc.getOperand(0).getConstantOperandVal(1) == 0) {
+ SDValue Src = AndSrc.getOperand(0).getOperand(0);
+ EVT SrcVT = Src.getValueType();
+ if (SrcVT.isFixedLengthVectorOf(MVT::i1)) {
+ EVT WideIntVT =
+ EVT::getIntegerVT(*DAG.getContext(), SrcVT.getSizeInBits());
+ if (TLI.isTypeLegal(WideIntVT))
+ X = DAG.getBitcast(WideIntVT, Src);
+ }
+ }
+ if (X && (!TLI.isTruncateFree(X, N0.getValueType()) ||
+ (!TLI.isZExtFree(N0.getValueType(), VT) &&
+ !TLI.isSExtCheaperThanZExt(N0.getValueType(), VT)))) {
+ X = DAG.getAnyExtOrTrunc(X, DL, VT);
+ APInt Mask = N0.getConstantOperandAPInt(1).zext(VT.getSizeInBits());
+ return DAG.getNode(ISD::AND, DL, VT, X, DAG.getConstant(Mask, DL, VT));
+ }
}
// fold (aext (load x)) -> (aext (truncate (extload x)))
diff --git a/llvm/test/CodeGen/AArch64/bitfield-insert.ll b/llvm/test/CodeGen/AArch64/bitfield-insert.ll
index eefb862c5313c..9402aa367ef2b 100644
--- a/llvm/test/CodeGen/AArch64/bitfield-insert.ll
+++ b/llvm/test/CodeGen/AArch64/bitfield-insert.ll
@@ -739,7 +739,7 @@ define i32 @orr_not_bfxil_test2_i32(i32 %0) {
define i16 @implicit_trunc_of_imm(ptr %p, i16 %a, i16 %b) {
; CHECK-LABEL: implicit_trunc_of_imm:
; CHECK: // %bb.0: // %entry
-; CHECK-NEXT: and w8, w1, #0xffffe000
+; CHECK-NEXT: and w8, w1, #0xe000
; CHECK-NEXT: mov x9, x0
; CHECK-NEXT: mov w10, w8
; CHECK-NEXT: mov w0, w8
diff --git a/llvm/test/CodeGen/AArch64/bswap-known-bits.ll b/llvm/test/CodeGen/AArch64/bswap-known-bits.ll
index 7e881198ced02..b42cf9db7bd01 100644
--- a/llvm/test/CodeGen/AArch64/bswap-known-bits.ll
+++ b/llvm/test/CodeGen/AArch64/bswap-known-bits.ll
@@ -131,7 +131,7 @@ define i16 @bswap_src_and_lo_i16(i16 %x) {
define i16 @bswap_src_and_hi_i16(i16 %x) {
; CHECK-LABEL: bswap_src_and_hi_i16:
; CHECK: ; %bb.0:
-; CHECK-NEXT: and w8, w0, #0xffffff00
+; CHECK-NEXT: and w8, w0, #0xff00
; CHECK-NEXT: rev16 w0, w8
; CHECK-NEXT: ret
%m = and i16 %x, 65280
diff --git a/llvm/test/CodeGen/AArch64/sshl_sat.ll b/llvm/test/CodeGen/AArch64/sshl_sat.ll
index be2b3e763733b..e68b1f090949f 100644
--- a/llvm/test/CodeGen/AArch64/sshl_sat.ll
+++ b/llvm/test/CodeGen/AArch64/sshl_sat.ll
@@ -131,7 +131,7 @@ define void @combine_shlsat_vector() nounwind {
define i16 @combine_shlsat_to_shl(i16 %x) nounwind {
; CHECK-LABEL: combine_shlsat_to_shl:
; CHECK: // %bb.0:
-; CHECK-NEXT: and w0, w0, #0xfffffffc
+; CHECK-NEXT: and w0, w0, #0xfffc
; CHECK-NEXT: ret
%x2 = ashr i16 %x, 2
%tmp = call i16 @llvm.sshl.sat.i16(i16 %x2, i16 2)
diff --git a/llvm/test/CodeGen/AArch64/ushl_sat.ll b/llvm/test/CodeGen/AArch64/ushl_sat.ll
index 870f80545f999..ac28a3dfbd168 100644
--- a/llvm/test/CodeGen/AArch64/ushl_sat.ll
+++ b/llvm/test/CodeGen/AArch64/ushl_sat.ll
@@ -117,7 +117,7 @@ define void @combine_shlsat_vector() nounwind {
define i16 @combine_shlsat_to_shl(i16 %x) nounwind {
; CHECK-LABEL: combine_shlsat_to_shl:
; CHECK: // %bb.0:
-; CHECK-NEXT: and w0, w0, #0xfffffffc
+; CHECK-NEXT: and w0, w0, #0xfffc
; CHECK-NEXT: ret
%x2 = lshr i16 %x, 2
%tmp = call i16 @llvm.ushl.sat.i16(i16 %x2, i16 2)
diff --git a/llvm/test/CodeGen/X86/aext-and-trunc-avx512.ll b/llvm/test/CodeGen/X86/aext-and-trunc-avx512.ll
new file mode 100644
index 0000000000000..78f525e4637bc
--- /dev/null
+++ b/llvm/test/CodeGen/X86/aext-and-trunc-avx512.ll
@@ -0,0 +1,41 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+avx512f,+avx512bw,+popcnt | FileCheck %s --check-prefix=BW
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mattr=+avx512f,+avx512bw,+avx512dq,+popcnt | FileCheck %s --check-prefix=DQ
+
+; Test (zext (and (trunc x) C)) -> (and x C) fold with AVX512 mask registers.
+; Ensures "andb $7, %al; movzbl %al, %eax" is folded to "andl $7, %eax".
+; Without AVX512DQ: bitcast v16i1->i16 + truncate i16->i8 (TRUNCATE path).
+; With AVX512DQ: extract_subvector v16i1->v8i1 + bitcast v8i1->i8, which
+; visitBITCAST canonicalises to the same truncate form before the fold fires.
+
+define i8 @ctpop_aext_i3_v3i1(ptr %p) {
+; BW-LABEL: ctpop_aext_i3_v3i1:
+; BW: # %bb.0:
+; BW-NEXT: vmovd {{.*#+}} xmm0 = mem[0],zero,zero,zero
+; BW-NEXT: vpbroadcastb {{.*#+}} xmm1 = [61,61,61,61,61,61,61,61,61,61,61,61,61,61,61,61]
+; BW-NEXT: vpcmpneqb %zmm1, %zmm0, %k0
+; BW-NEXT: kmovd %k0, %eax
+; BW-NEXT: andl $7, %eax
+; BW-NEXT: popcntl %eax, %eax
+; BW-NEXT: # kill: def $al killed $al killed $eax
+; BW-NEXT: vzeroupper
+; BW-NEXT: retq
+;
+; DQ-LABEL: ctpop_aext_i3_v3i1:
+; DQ: # %bb.0:
+; DQ-NEXT: vmovd {{.*#+}} xmm0 = mem[0],zero,zero,zero
+; DQ-NEXT: vpbroadcastb {{.*#+}} xmm1 = [61,61,61,61,61,61,61,61,61,61,61,61,61,61,61,61]
+; DQ-NEXT: vpcmpneqb %zmm1, %zmm0, %k0
+; DQ-NEXT: kmovd %k0, %eax
+; DQ-NEXT: andl $7, %eax
+; DQ-NEXT: popcntl %eax, %eax
+; DQ-NEXT: # kill: def $al killed $al killed $eax
+; DQ-NEXT: vzeroupper
+; DQ-NEXT: retq
+ %v = load <3 x i8>, ptr %p
+ %cmp = icmp ne <3 x i8> %v, splat (i8 61)
+ %bc = bitcast <3 x i1> %cmp to i3
+ %ct = call i3 @llvm.ctpop.i3(i3 %bc)
+ %ext = zext i3 %ct to i8
+ ret i8 %ext
+}
diff --git a/llvm/test/CodeGen/X86/aext-and-trunc.ll b/llvm/test/CodeGen/X86/aext-and-trunc.ll
new file mode 100644
index 0000000000000..5b01be813ed7a
--- /dev/null
+++ b/llvm/test/CodeGen/X86/aext-and-trunc.ll
@@ -0,0 +1,47 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu | FileCheck %s --check-prefix=X64
+; RUN: llc < %s -mtriple=i686-unknown-linux-gnu | FileCheck %s --check-prefix=X86
+
+; Regression tests for the (zext (and (trunc x) C)) -> (and x C) fold in
+; visitZERO_EXTEND (and symmetrically in visitANY_EXTEND). On X86,
+; isTruncateFree(i32->i8) is true but isZExtFree(i8->i32) is false, so the
+; fold fires to avoid redundant narrow-AND + movzx sequences.
+
+; Basic (zext (and (trunc i32) C)) -> (and i32 C) pattern.
+define i32 @zext_and_trunc_i32(i32 %x) {
+; X64-LABEL: zext_and_trunc_i32:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: andl $7, %eax
+; X64-NEXT: retq
+;
+; X86-LABEL: zext_and_trunc_i32:
+; X86: # %bb.0:
+; X86-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X86-NEXT: andl $7, %eax
+; X86-NEXT: retl
+ %trunc = trunc i32 %x to i8
+ %and = and i8 %trunc, 7
+ %zext = zext i8 %and to i32
+ ret i32 %zext
+}
+
+; (zext (and (trunc i64) C)) -> (and i32 C) on x86-64.
+define i32 @zext_and_trunc_i64(i64 %x) {
+; X64-LABEL: zext_and_trunc_i64:
+; X64: # %bb.0:
+; X64-NEXT: movq %rdi, %rax
+; X64-NEXT: andl $15, %eax
+; X64-NEXT: # kill: def $eax killed $eax killed $rax
+; X64-NEXT: retq
+;
+; X86-LABEL: zext_and_trunc_i64:
+; X86: # %bb.0:
+; X86-NEXT: movl {{[0-9]+}}(%esp), %eax
+; X86-NEXT: andl $15, %eax
+; X86-NEXT: retl
+ %trunc = trunc i64 %x to i8
+ %and = and i8 %trunc, 15
+ %zext = zext i8 %and to i32
+ ret i32 %zext
+}
diff --git a/llvm/test/CodeGen/X86/and-with-overflow.ll b/llvm/test/CodeGen/X86/and-with-overflow.ll
index a63f6cc6ea7e2..1fef4165f0591 100644
--- a/llvm/test/CodeGen/X86/and-with-overflow.ll
+++ b/llvm/test/CodeGen/X86/and-with-overflow.ll
@@ -21,8 +21,8 @@ define i8 @and_i8_ri(i8 zeroext %0, i8 zeroext %1) {
; X64-LABEL: and_i8_ri:
; X64: # %bb.0:
; X64-NEXT: movl %edi, %eax
-; X64-NEXT: andb $-17, %al
-; X64-NEXT: movzbl %al, %eax
+; X64-NEXT: andl $-17, %eax
+; X64-NEXT: testb $-17, %dil
; X64-NEXT: cmovel %edi, %eax
; X64-NEXT: # kill: def $al killed $al killed $eax
; X64-NEXT: retq
diff --git a/llvm/test/CodeGen/X86/combine-srem.ll b/llvm/test/CodeGen/X86/combine-srem.ll
index 0ca79adf5392e..82ac096353f06 100644
--- a/llvm/test/CodeGen/X86/combine-srem.ll
+++ b/llvm/test/CodeGen/X86/combine-srem.ll
@@ -498,7 +498,7 @@ define i16 @combine_i16_srem_pow2(i16 %x) {
; CHECK-NEXT: leal 15(%rax), %ecx
; CHECK-NEXT: testw %ax, %ax
; CHECK-NEXT: cmovnsl %edi, %ecx
-; CHECK-NEXT: andl $-16, %ecx
+; CHECK-NEXT: andl $65520, %ecx # imm = 0xFFF0
; CHECK-NEXT: subl %ecx, %eax
; CHECK-NEXT: # kill: def $ax killed $ax killed $rax
; CHECK-NEXT: retq
@@ -513,7 +513,7 @@ define i16 @combine_i16_srem_negpow2(i16 %x) {
; CHECK-NEXT: leal 255(%rax), %ecx
; CHECK-NEXT: testw %ax, %ax
; CHECK-NEXT: cmovnsl %edi, %ecx
-; CHECK-NEXT: andl $-256, %ecx
+; CHECK-NEXT: andl $65280, %ecx # imm = 0xFF00
; CHECK-NEXT: subl %ecx, %eax
; CHECK-NEXT: # kill: def $ax killed $ax killed $rax
; CHECK-NEXT: retq
diff --git a/llvm/test/CodeGen/X86/llvm.frexp.ll b/llvm/test/CodeGen/X86/llvm.frexp.ll
index e3a1b1b83b2e3..9112c176eaf27 100644
--- a/llvm/test/CodeGen/X86/llvm.frexp.ll
+++ b/llvm/test/CodeGen/X86/llvm.frexp.ll
@@ -25,7 +25,7 @@ define { half, i32 } @test_frexp_f16_i32(half %a) nounwind {
; X64-NEXT: cmpl $1024, %esi # imm = 0x400
; X64-NEXT: cmovael %eax, %edi
; X64-NEXT: addl $-14, %edi
-; X64-NEXT: andl $-31745, %ecx # imm = 0x83FF
+; X64-NEXT: andl $33791, %ecx # imm = 0x83FF
; X64-NEXT: orl $14336, %ecx # imm = 0x3800
; X64-NEXT: addl $-31744, %esi # imm = 0x8400
; X64-NEXT: movzwl %si, %esi
@@ -76,7 +76,7 @@ define half @test_frexp_f16_i32_only_use_fract(half %a) nounwind {
; X64-NEXT: andl $32767, %edx # imm = 0x7FFF
; X64-NEXT: cmpl $1024, %edx # imm = 0x400
; X64-NEXT: cmovael %ecx, %eax
-; X64-NEXT: andl $-31745, %eax # imm = 0x83FF
+; X64-NEXT: andl $33791, %eax # imm = 0x83FF
; X64-NEXT: orl $14336, %eax # imm = 0x3800
; X64-NEXT: addl $-31744, %edx # imm = 0x8400
; X64-NEXT: movzwl %dx, %edx
diff --git a/llvm/test/CodeGen/X86/masked_load.ll b/llvm/test/CodeGen/X86/masked_load.ll
index d755c8e34e60b..88b51499518be 100644
--- a/llvm/test/CodeGen/X86/masked_load.ll
+++ b/llvm/test/CodeGen/X86/masked_load.ll
@@ -701,15 +701,15 @@ define <8 x double> @load_v8f64_i8(i8 %trigger, ptr %addr, <8 x double> %dst) {
; AVX1-LABEL: load_v8f64_i8:
; AVX1: ## %bb.0:
; AVX1-NEXT: movl %edi, %eax
-; AVX1-NEXT: shrb %al
+; AVX1-NEXT: shrb $2, %al
; AVX1-NEXT: andb $1, %al
; AVX1-NEXT: movl %edi, %ecx
+; AVX1-NEXT: shrb %cl
; AVX1-NEXT: andb $1, %cl
-; AVX1-NEXT: vmovd %ecx, %xmm2
-; AVX1-NEXT: vpinsrb $2, %eax, %xmm2, %xmm2
-; AVX1-NEXT: movl %edi, %eax
-; AVX1-NEXT: shrb $2, %al
-; AVX1-NEXT: andb $1, %al
+; AVX1-NEXT: movl %edi, %edx
+; AVX1-NEXT: andl $1, %edx
+; AVX1-NEXT: vmovd %edx, %xmm2
+; AVX1-NEXT: vpinsrb $2, %ecx, %xmm2, %xmm2
; AVX1-NEXT: vpinsrb $4, %eax, %xmm2, %xmm2
; AVX1-NEXT: movl %edi, %eax
; AVX1-NEXT: shrb $3, %al
@@ -750,15 +750,15 @@ define <8 x double> @load_v8f64_i8(i8 %trigger, ptr %addr, <8 x double> %dst) {
; AVX2-LABEL: load_v8f64_i8:
; AVX2: ## %bb.0:
; AVX2-NEXT: movl %edi, %eax
-; AVX2-NEXT: shrb %al
+; AVX2-NEXT: shrb $2, %al
; AVX2-NEXT: andb $1, %al
; AVX2-NEXT: movl %edi, %ecx
+; AVX2-NEXT: shrb %cl
; AVX2-NEXT: andb $1, %cl
-; AVX2-NEXT: vmovd %ecx, %xmm2
-; AVX2-NEXT: vpinsrb $2, %eax, %xmm2, %xmm2
-; AVX2-NEXT: movl %edi, %eax
-; AVX2-NEXT: shrb $2, %al
-; AVX2-NEXT: andb $1, %al
+; AVX2-NEXT: movl %edi, %edx
+; AVX2-NEXT: andl $1, %edx
+; AVX2-NEXT: vmovd %edx, %xmm2
+; AVX2-NEXT: vpinsrb $2, %ecx, %xmm2, %xmm2
; AVX2-NEXT: vpinsrb $4, %eax, %xmm2, %xmm2
; AVX2-NEXT: movl %edi, %eax
; AVX2-NEXT: shrb $3, %al
@@ -1205,7 +1205,7 @@ define <2 x float> @load_v2f32_i2(i2 %trigger, ptr %addr, <2 x float> %dst) {
; AVX1OR2-NEXT: movl %edi, %eax
; AVX1OR2-NEXT: andb $2, %al
; AVX1OR2-NEXT: shrb %al
-; AVX1OR2-NEXT: andb $1, %dil
+; AVX1OR2-NEXT: andl $1, %edi
; AVX1OR2-NEXT: vmovd %edi, %xmm1
; AVX1OR2-NEXT: vpinsrb $8, %eax, %xmm1, %xmm1
; AVX1OR2-NEXT: vinsertps {{.*#+}} xmm1 = xmm1[0,2],zero,zero
diff --git a/llvm/test/CodeGen/X86/masked_store.ll b/llvm/test/CodeGen/X86/masked_store.ll
index 5882f3924473e..c0897c290bc05 100644
--- a/llvm/test/CodeGen/X86/masked_store.ll
+++ b/llvm/test/CodeGen/X86/masked_store.ll
@@ -466,7 +466,7 @@ define void @store_v2f32_i2(i2 %trigger, ptr %addr, <2 x float> %val) nounwind {
; AVX1OR2-NEXT: movl %edi, %eax
; AVX1OR2-NEXT: andb $2, %al
; AVX1OR2-NEXT: shrb %al
-; AVX1OR2-NEXT: andb $1, %dil
+; AVX1OR2-NEXT: andl $1, %edi
; AVX1OR2-NEXT: vmovd %edi, %xmm1
; AVX1OR2-NEXT: vpinsrb $8, %eax, %xmm1, %xmm1
; AVX1OR2-NEXT: vinsertps {{.*#+}} xmm1 = xmm1[0,2],zero,zero
diff --git a/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-innerouter.ll b/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-innerouter.ll
index 9a8719f9a64fa..7b53f2b22b08c 100644
--- a/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-innerouter.ll
+++ b/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-innerouter.ll
@@ -39,7 +39,7 @@ define i16 @out16_constmask(i16 %x, i16 %y) {
; CHECK-NOBMI-NEXT: # kill: def $esi killed $esi def $rsi
; CHECK-NOBMI-NEXT: # kill: def $edi killed $edi def $rdi
; CHECK-NOBMI-NEXT: andl $4080, %edi # imm = 0xFF0
-; CHECK-NOBMI-NEXT: andl $-4081, %esi # imm = 0xF00F
+; CHECK-NOBMI-NEXT: andl $61455, %esi # imm = 0xF00F
; CHECK-NOBMI-NEXT: leal (%rsi,%rdi), %eax
; CHECK-NOBMI-NEXT: # kill: def $ax killed $ax killed $eax
; CHECK-NOBMI-NEXT: retq
@@ -49,7 +49,7 @@ define i16 @out16_constmask(i16 %x, i16 %y) {
; CHECK-BMI-NEXT: # kill: def $esi killed $esi def $rsi
; CHECK-BMI-NEXT: # kill: def $edi killed $edi def $rdi
; CHECK-BMI-NEXT: andl $4080, %edi # imm = 0xFF0
-; CHECK-BMI-NEXT: andl $-4081, %esi # imm = 0xF00F
+; CHECK-BMI-NEXT: andl $61455, %esi # imm = 0xF00F
; CHECK-BMI-NEXT: leal (%rsi,%rdi), %eax
; CHECK-BMI-NEXT: # kill: def $ax killed $ax killed $eax
; CHECK-BMI-NEXT: retq
diff --git a/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-interleavedbits.ll b/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-interleavedbits.ll
index c4c4e5ed1fdde..c58f110fe9e64 100644
--- a/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-interleavedbits.ll
+++ b/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-interleavedbits.ll
@@ -39,7 +39,7 @@ define i16 @out16_constmask(i16 %x, i16 %y) {
; CHECK-NOBMI-NEXT: # kill: def $esi killed $esi def $rsi
; CHECK-NOBMI-NEXT: # kill: def $edi killed $edi def $rdi
; CHECK-NOBMI-NEXT: andl $21845, %edi # imm = 0x5555
-; CHECK-NOBMI-NEXT: andl $-21846, %esi # imm = 0xAAAA
+; CHECK-NOBMI-NEXT: andl $43690, %esi # imm = 0xAAAA
; CHECK-NOBMI-NEXT: leal (%rsi,%rdi), %eax
; CHECK-NOBMI-NEXT: # kill: def $ax killed $ax killed $eax
; CHECK-NOBMI-NEXT: retq
@@ -49,7 +49,7 @@ define i16 @out16_constmask(i16 %x, i16 %y) {
; CHECK-BMI-NEXT: # kill: def $esi killed $esi def $rsi
; CHECK-BMI-NEXT: # kill: def $edi killed $edi def $rdi
; CHECK-BMI-NEXT: andl $21845, %edi # imm = 0x5555
-; CHECK-BMI-NEXT: andl $-21846, %esi # imm = 0xAAAA
+; CHECK-BMI-NEXT: andl $43690, %esi # imm = 0xAAAA
; CHECK-BMI-NEXT: leal (%rsi,%rdi), %eax
; CHECK-BMI-NEXT: # kill: def $ax killed $ax killed $eax
; CHECK-BMI-NEXT: retq
diff --git a/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-interleavedbytehalves.ll b/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-interleavedbytehalves.ll
index 2ea74f3942387..f5dbc51fe518f 100644
--- a/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-interleavedbytehalves.ll
+++ b/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-interleavedbytehalves.ll
@@ -39,7 +39,7 @@ define i16 @out16_constmask(i16 %x, i16 %y) {
; CHECK-NOBMI-NEXT: # kill: def $esi killed $esi def $rsi
; CHECK-NOBMI-NEXT: # kill: def $edi killed $edi def $rdi
; CHECK-NOBMI-NEXT: andl $3855, %edi # imm = 0xF0F
-; CHECK-NOBMI-NEXT: andl $-3856, %esi # imm = 0xF0F0
+; CHECK-NOBMI-NEXT: andl $61680, %esi # imm = 0xF0F0
; CHECK-NOBMI-NEXT: leal (%rsi,%rdi), %eax
; CHECK-NOBMI-NEXT: # kill: def $ax killed $ax killed $eax
; CHECK-NOBMI-NEXT: retq
@@ -49,7 +49,7 @@ define i16 @out16_constmask(i16 %x, i16 %y) {
; CHECK-BMI-NEXT: # kill: def $esi killed $esi def $rsi
; CHECK-BMI-NEXT: # kill: def $edi killed $edi def $rdi
; CHECK-BMI-NEXT: andl $3855, %edi # imm = 0xF0F
-; CHECK-BMI-NEXT: andl $-3856, %esi # imm = 0xF0F0
+; CHECK-BMI-NEXT: andl $61680, %esi # imm = 0xF0F0
; CHECK-BMI-NEXT: leal (%rsi,%rdi), %eax
; CHECK-BMI-NEXT: # kill: def $ax killed $ax killed $eax
; CHECK-BMI-NEXT: retq
diff --git a/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-lowhigh.ll b/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-lowhigh.ll
index eb6accd3e623b..9b85a21dd23bc 100644
--- a/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-lowhigh.ll
+++ b/llvm/test/CodeGen/X86/unfold-masked-merge-scalar-constmask-lowhigh.ll
@@ -37,7 +37,7 @@ define i16 @out16_constmask(i16 %x, i16 %y) {
; CHECK-NOBMI-LABEL: out16_constmask:
; CHECK-NOBMI: # %bb.0:
; CHECK-NOBMI-NEXT: movzbl %dil, %eax
-; CHECK-NOBMI-NEXT: andl $-256, %esi
+; CHECK-NOBMI-NEXT: andl $65280, %esi # imm = 0xFF00
; CHECK-NOBMI-NEXT: orl %esi, %eax
; CHECK-NOBMI-NEXT: # kill: def $ax killed $ax killed $eax
; CHECK-NOBMI-NEXT: retq
@@ -45,7 +45,7 @@ define i16 @out16_constmask(i16 %x, i16 %y) {
; CHECK-BMI-LABEL: out16_constmask:
; CHECK-BMI: # %bb.0:
; CHECK-BMI-NEXT: movzbl %dil, %eax
-; CHECK-BMI-NEXT: andl $-256, %esi
+; CHECK-BMI-NEXT: andl $65280, %esi # imm = 0xFF00
; CHECK-BMI-NEXT: orl %esi, %eax
; CHECK-BMI-NEXT: # kill: def $ax killed $ax killed $eax
; CHECK-BMI-NEXT: retq
More information about the llvm-commits
mailing list