[llvm] [DAG][X86] Bitfield insertion can combine into SHL + SHRD (PR #220182)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 15 01:01:38 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-x86
Author: Pratyay Pande (pratyaypande)
<details>
<summary>Changes</summary>
This change combines the following pattern:
```llvm
define dso_local noundef i64 @<!-- -->updateTop10Bits(unsigned long, unsigned long)(i64 noundef %A, i64 noundef %B) local_unnamed_addr {
entry:
%and = and i64 %A, 18014398509481983
%shl = shl i64 %B, 54
%or = or disjoint i64 %shl, %and
ret i64 %or
}
```
Into an `shrd` usage instead of `bhziq`.
This change is intended to fix issue #<!-- -->112488
---
Patch is 62.50 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/220182.diff
3 Files Affected:
- (modified) llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp (+42)
- (modified) llvm/lib/Target/X86/X86ISelLowering.cpp (+36-1)
- (added) llvm/test/CodeGen/X86/insert-bitfield.ll (+1614)
``````````diff
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index a829d7a34d5a6..6f62442690542 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -8844,6 +8844,44 @@ static SDValue visitORCommutative(SelectionDAG &DAG, SDValue N0, SDValue N1,
return SDValue();
}
+// Fold an OR with a masked destination and a left-shifted
+// source into a shift + double-precision shift (SHRD):
+static SDValue combineOrOnSHLToFSHR(SDNode *N, SDLoc &DL, SelectionDAG &DAG) {
+ EVT VT = N->getValueType(0);
+
+ APInt Mask, ShiftAmount;
+ SDValue X, Y;
+
+ // Check for the following pattern:
+ // (or (and X, HighBitsMask(C)), (srl Y, C))
+ // Do not combine if there are multi-use AND and OR.
+ // It does not result in more performant code.
+ if (!sd_match(N, m_Or(m_OneUse(m_And(m_Value(X), m_ConstInt(Mask))),
+ m_OneUse(m_Shl(m_Value(Y), m_ConstInt(ShiftAmount))))))
+ return SDValue();
+
+ // Max bit-width of operands
+ uint64_t MaxMaskBitWidth = VT.getScalarSizeInBits();
+
+ // Check for Mask and ShiftAmount
+ //
+ // (shl Y, ShiftAmount) fills the top (MaxMaskBitWidth - ShiftAmount) bits,
+ // so X must keep exactly the low ShiftAmount.
+ APInt ExpectedMask =
+ APInt::getLowBitsSet(MaxMaskBitWidth, ShiftAmount.getZExtValue());
+
+ if (!((ShiftAmount.getZExtValue() > 0) &&
+ (ShiftAmount.getZExtValue() < MaxMaskBitWidth) &&
+ (Mask == ExpectedMask)))
+ return SDValue();
+
+ uint64_t InvShAmt = MaxMaskBitWidth - ShiftAmount.getZExtValue();
+ SDValue ShAConst = DAG.getShiftAmountConstant(InvShAmt, VT, DL);
+ SDValue SHLVal = DAG.getNode(ISD::SHL, DL, VT, X, ShAConst);
+
+ return DAG.getNode(ISD::FSHR, DL, VT, Y, SHLVal, ShAConst);
+}
+
SDValue DAGCombiner::visitOR(SDNode *N) {
SDValue N0 = N->getOperand(0);
SDValue N1 = N->getOperand(1);
@@ -9043,6 +9081,10 @@ SDValue DAGCombiner::visitOR(SDNode *N) {
if (SDValue Load = MatchLoadCombine(N))
return Load;
+ if (Level == CombineLevel::BeforeLegalizeTypes)
+ if (SDValue Res = combineOrOnSHLToFSHR(N, DL, DAG))
+ return Res;
+
// Simplify the operands using demanded-bits information.
if (SimplifyDemandedBits(SDValue(N, 0)))
return SDValue(N, 0);
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 775fd7642f040..4d2b4ca836e4e 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -32343,8 +32343,43 @@ static SDValue LowerFunnelShift(SDValue Op, const X86Subtarget &Subtarget,
return DAG.getZExtOrTrunc(Res, DL, VT);
}
- if (VT == MVT::i8 || ExpandFunnel)
+ // If expanding the funnel shift is required OR the value type is unsupported
+ if (VT == MVT::i8 || ExpandFunnel) {
+ // DAG Combiner would combine an OR on a masked value and a shifted value
+ // into a shift followed by a funnel shift. The following fold reverses that
+ // operation in case `ExpandFunnel` is set, i.e., if we are not optimizing
+ // for size and if funnel-shift is slow. The fold applies if one operand is
+ // the result of an ISD::SHL or the node is MVT::i8, for which X86 does not
+ // have a funnel-shift instruction.
+ if (Op1.getOpcode() == ISD::SHL && isa<ConstantSDNode>(Amt.getNode())) {
+
+ auto *C = dyn_cast<ConstantSDNode>(Amt.getNode());
+ SDValue SHLOperandShiftAmount = Op1->getOperand(1);
+ uint64_t InvMaskWidth = C->getAPIntValue().urem(EltSizeInBits);
+
+ if (ConstantSDNode *EC =
+ dyn_cast<ConstantSDNode>(SHLOperandShiftAmount.getNode())) {
+ const APInt &ExpectedShiftAmount = EC->getAPIntValue();
+
+ // Check if the shift amounts match.
+ if (ExpectedShiftAmount != InvMaskWidth)
+ return SDValue();
+ }
+
+ uint64_t ShiftAmount = EltSizeInBits - InvMaskWidth;
+ SDValue SHLOperand = Op1.getOperand(0);
+
+ APInt Mask = APInt::getLowBitsSet(EltSizeInBits, ShiftAmount);
+
+ SDValue MaskBitNum = DAG.getShiftAmountConstant(
+ ShiftAmount, SHLOperand.getValueType(), DL);
+ SDValue MaskNode = DAG.getConstant(Mask, DL, VT);
+ SDValue AndMask = DAG.getNode(ISD::AND, DL, VT, SHLOperand, MaskNode);
+ SDValue SHL = DAG.getNode(ISD::SHL, DL, VT, Op0, MaskBitNum);
+ return DAG.getNode(ISD::OR, DL, VT, AndMask, SHL);
+ }
return SDValue();
+ }
// i16 needs to modulo the shift amount, but i32/i64 have implicit modulo.
if (VT == MVT::i16) {
diff --git a/llvm/test/CodeGen/X86/insert-bitfield.ll b/llvm/test/CodeGen/X86/insert-bitfield.ll
new file mode 100644
index 0000000000000..ddb0dc85d89a6
--- /dev/null
+++ b/llvm/test/CodeGen/X86/insert-bitfield.ll
@@ -0,0 +1,1614 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-- | FileCheck %s --check-prefixes=X64
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+slow-shld | FileCheck %s --check-prefixes=X64-SLOW
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s --check-prefixes=X64-AVX2
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+avx512f,+avx512bw,+avx512vl | FileCheck %s --check-prefixes=X64-AVX512
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+avx512vbmi2,+avx512vl | FileCheck %s --check-prefixes=X64-AVX512-BMI2
+
+define i64 @insert_10_i64(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_10_i64:
+; X64: # %bb.0:
+; X64-NEXT: movq %rdi, %rax
+; X64-NEXT: shlq $10, %rax
+; X64-NEXT: shrdq $10, %rsi, %rax
+; X64-NEXT: retq
+;
+; X64-SLOW-LABEL: insert_10_i64:
+; X64-SLOW: # %bb.0:
+; X64-SLOW-NEXT: movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT: andq %rdi, %rax
+; X64-SLOW-NEXT: shlq $54, %rsi
+; X64-SLOW-NEXT: orq %rsi, %rax
+; X64-SLOW-NEXT: retq
+;
+; X64-AVX2-LABEL: insert_10_i64:
+; X64-AVX2: # %bb.0:
+; X64-AVX2-NEXT: movq %rdi, %rax
+; X64-AVX2-NEXT: shlq $10, %rax
+; X64-AVX2-NEXT: shrdq $10, %rsi, %rax
+; X64-AVX2-NEXT: retq
+;
+; X64-AVX512-LABEL: insert_10_i64:
+; X64-AVX512: # %bb.0:
+; X64-AVX512-NEXT: movq %rdi, %rax
+; X64-AVX512-NEXT: shlq $10, %rax
+; X64-AVX512-NEXT: shrdq $10, %rsi, %rax
+; X64-AVX512-NEXT: retq
+;
+; X64-AVX512-BMI2-LABEL: insert_10_i64:
+; X64-AVX512-BMI2: # %bb.0:
+; X64-AVX512-BMI2-NEXT: movq %rdi, %rax
+; X64-AVX512-BMI2-NEXT: shlq $10, %rax
+; X64-AVX512-BMI2-NEXT: shrdq $10, %rsi, %rax
+; X64-AVX512-BMI2-NEXT: retq
+ %and = and i64 %a, 18014398509481983
+ %shl = shl i64 %b, 54
+ %or = or i64 %shl, %and
+ ret i64 %or
+}
+
+define i64 @insert_10_i64_disjoint(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_10_i64_disjoint:
+; X64: # %bb.0:
+; X64-NEXT: movq %rdi, %rax
+; X64-NEXT: shlq $10, %rax
+; X64-NEXT: shrdq $10, %rsi, %rax
+; X64-NEXT: retq
+;
+; X64-SLOW-LABEL: insert_10_i64_disjoint:
+; X64-SLOW: # %bb.0:
+; X64-SLOW-NEXT: movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT: andq %rdi, %rax
+; X64-SLOW-NEXT: shlq $54, %rsi
+; X64-SLOW-NEXT: orq %rsi, %rax
+; X64-SLOW-NEXT: retq
+;
+; X64-AVX2-LABEL: insert_10_i64_disjoint:
+; X64-AVX2: # %bb.0:
+; X64-AVX2-NEXT: movq %rdi, %rax
+; X64-AVX2-NEXT: shlq $10, %rax
+; X64-AVX2-NEXT: shrdq $10, %rsi, %rax
+; X64-AVX2-NEXT: retq
+;
+; X64-AVX512-LABEL: insert_10_i64_disjoint:
+; X64-AVX512: # %bb.0:
+; X64-AVX512-NEXT: movq %rdi, %rax
+; X64-AVX512-NEXT: shlq $10, %rax
+; X64-AVX512-NEXT: shrdq $10, %rsi, %rax
+; X64-AVX512-NEXT: retq
+;
+; X64-AVX512-BMI2-LABEL: insert_10_i64_disjoint:
+; X64-AVX512-BMI2: # %bb.0:
+; X64-AVX512-BMI2-NEXT: movq %rdi, %rax
+; X64-AVX512-BMI2-NEXT: shlq $10, %rax
+; X64-AVX512-BMI2-NEXT: shrdq $10, %rsi, %rax
+; X64-AVX512-BMI2-NEXT: retq
+ %and = and i64 %a, 18014398509481983
+ %shl = shl i64 %b, 54
+ %or = or disjoint i64 %shl, %and
+ ret i64 %or
+}
+
+; Commuted operands.
+define i64 @insert_10_i64_commute(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_10_i64_commute:
+; X64: # %bb.0:
+; X64-NEXT: movq %rdi, %rax
+; X64-NEXT: shlq $10, %rax
+; X64-NEXT: shrdq $10, %rsi, %rax
+; X64-NEXT: retq
+;
+; X64-SLOW-LABEL: insert_10_i64_commute:
+; X64-SLOW: # %bb.0:
+; X64-SLOW-NEXT: movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT: andq %rdi, %rax
+; X64-SLOW-NEXT: shlq $54, %rsi
+; X64-SLOW-NEXT: orq %rsi, %rax
+; X64-SLOW-NEXT: retq
+;
+; X64-AVX2-LABEL: insert_10_i64_commute:
+; X64-AVX2: # %bb.0:
+; X64-AVX2-NEXT: movq %rdi, %rax
+; X64-AVX2-NEXT: shlq $10, %rax
+; X64-AVX2-NEXT: shrdq $10, %rsi, %rax
+; X64-AVX2-NEXT: retq
+;
+; X64-AVX512-LABEL: insert_10_i64_commute:
+; X64-AVX512: # %bb.0:
+; X64-AVX512-NEXT: movq %rdi, %rax
+; X64-AVX512-NEXT: shlq $10, %rax
+; X64-AVX512-NEXT: shrdq $10, %rsi, %rax
+; X64-AVX512-NEXT: retq
+;
+; X64-AVX512-BMI2-LABEL: insert_10_i64_commute:
+; X64-AVX512-BMI2: # %bb.0:
+; X64-AVX512-BMI2-NEXT: movq %rdi, %rax
+; X64-AVX512-BMI2-NEXT: shlq $10, %rax
+; X64-AVX512-BMI2-NEXT: shrdq $10, %rsi, %rax
+; X64-AVX512-BMI2-NEXT: retq
+ %and = and i64 %a, 18014398509481983
+ %shl = shl i64 %b, 54
+ %or = or i64 %and, %shl
+ ret i64 %or
+}
+
+define i64 @insert_33_i64(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_33_i64:
+; X64: # %bb.0:
+; X64-NEXT: movq %rdi, %rax
+; X64-NEXT: shlq $33, %rax
+; X64-NEXT: shrdq $33, %rsi, %rax
+; X64-NEXT: retq
+;
+; X64-SLOW-LABEL: insert_33_i64:
+; X64-SLOW: # %bb.0:
+; X64-SLOW-NEXT: shlq $31, %rsi
+; X64-SLOW-NEXT: andl $2147483647, %edi # imm = 0x7FFFFFFF
+; X64-SLOW-NEXT: leaq (%rdi,%rsi), %rax
+; X64-SLOW-NEXT: retq
+;
+; X64-AVX2-LABEL: insert_33_i64:
+; X64-AVX2: # %bb.0:
+; X64-AVX2-NEXT: movq %rdi, %rax
+; X64-AVX2-NEXT: shlq $33, %rax
+; X64-AVX2-NEXT: shrdq $33, %rsi, %rax
+; X64-AVX2-NEXT: retq
+;
+; X64-AVX512-LABEL: insert_33_i64:
+; X64-AVX512: # %bb.0:
+; X64-AVX512-NEXT: movq %rdi, %rax
+; X64-AVX512-NEXT: shlq $33, %rax
+; X64-AVX512-NEXT: shrdq $33, %rsi, %rax
+; X64-AVX512-NEXT: retq
+;
+; X64-AVX512-BMI2-LABEL: insert_33_i64:
+; X64-AVX512-BMI2: # %bb.0:
+; X64-AVX512-BMI2-NEXT: movq %rdi, %rax
+; X64-AVX512-BMI2-NEXT: shlq $33, %rax
+; X64-AVX512-BMI2-NEXT: shrdq $33, %rsi, %rax
+; X64-AVX512-BMI2-NEXT: retq
+ %and = and i64 %a, 2147483647
+ %shl = shl i64 %b, 31
+ %or = or i64 %shl, %and
+ ret i64 %or
+}
+
+define i64 @insert_1_i64(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_1_i64:
+; X64: # %bb.0:
+; X64-NEXT: leaq (%rdi,%rdi), %rax
+; X64-NEXT: shrdq $1, %rsi, %rax
+; X64-NEXT: retq
+;
+; X64-SLOW-LABEL: insert_1_i64:
+; X64-SLOW: # %bb.0:
+; X64-SLOW-NEXT: movabsq $9223372036854775807, %rax # imm = 0x7FFFFFFFFFFFFFFF
+; X64-SLOW-NEXT: andq %rdi, %rax
+; X64-SLOW-NEXT: shlq $63, %rsi
+; X64-SLOW-NEXT: orq %rsi, %rax
+; X64-SLOW-NEXT: retq
+;
+; X64-AVX2-LABEL: insert_1_i64:
+; X64-AVX2: # %bb.0:
+; X64-AVX2-NEXT: leaq (%rdi,%rdi), %rax
+; X64-AVX2-NEXT: shrdq $1, %rsi, %rax
+; X64-AVX2-NEXT: retq
+;
+; X64-AVX512-LABEL: insert_1_i64:
+; X64-AVX512: # %bb.0:
+; X64-AVX512-NEXT: leaq (%rdi,%rdi), %rax
+; X64-AVX512-NEXT: shrdq $1, %rsi, %rax
+; X64-AVX512-NEXT: retq
+;
+; X64-AVX512-BMI2-LABEL: insert_1_i64:
+; X64-AVX512-BMI2: # %bb.0:
+; X64-AVX512-BMI2-NEXT: leaq (%rdi,%rdi), %rax
+; X64-AVX512-BMI2-NEXT: shrdq $1, %rsi, %rax
+; X64-AVX512-BMI2-NEXT: retq
+ %and = and i64 %a, 9223372036854775807
+ %shl = shl i64 %b, 63
+ %or = or i64 %shl, %and
+ ret i64 %or
+}
+
+; i32 inputs and mask
+define i32 @insert_10_i32(i32 %a, i32 %b) nounwind {
+; X64-LABEL: insert_10_i32:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: shll $10, %eax
+; X64-NEXT: shrdl $10, %esi, %eax
+; X64-NEXT: retq
+;
+; X64-SLOW-LABEL: insert_10_i32:
+; X64-SLOW: # %bb.0:
+; X64-SLOW-NEXT: # kill: def $esi killed $esi def $rsi
+; X64-SLOW-NEXT: # kill: def $edi killed $edi def $rdi
+; X64-SLOW-NEXT: shll $22, %esi
+; X64-SLOW-NEXT: andl $4194303, %edi # imm = 0x3FFFFF
+; X64-SLOW-NEXT: leal (%rdi,%rsi), %eax
+; X64-SLOW-NEXT: retq
+;
+; X64-AVX2-LABEL: insert_10_i32:
+; X64-AVX2: # %bb.0:
+; X64-AVX2-NEXT: movl %edi, %eax
+; X64-AVX2-NEXT: shll $10, %eax
+; X64-AVX2-NEXT: shrdl $10, %esi, %eax
+; X64-AVX2-NEXT: retq
+;
+; X64-AVX512-LABEL: insert_10_i32:
+; X64-AVX512: # %bb.0:
+; X64-AVX512-NEXT: movl %edi, %eax
+; X64-AVX512-NEXT: shll $10, %eax
+; X64-AVX512-NEXT: shrdl $10, %esi, %eax
+; X64-AVX512-NEXT: retq
+;
+; X64-AVX512-BMI2-LABEL: insert_10_i32:
+; X64-AVX512-BMI2: # %bb.0:
+; X64-AVX512-BMI2-NEXT: movl %edi, %eax
+; X64-AVX512-BMI2-NEXT: shll $10, %eax
+; X64-AVX512-BMI2-NEXT: shrdl $10, %esi, %eax
+; X64-AVX512-BMI2-NEXT: retq
+ %and = and i32 %a, 4194303
+ %shl = shl i32 %b, 22
+ %or = or i32 %shl, %and
+ ret i32 %or
+}
+
+; i16 inputs and mask
+define i16 @insert_6_i16(i16 %a, i16 %b) nounwind {
+; X64-LABEL: insert_6_i16:
+; X64: # %bb.0:
+; X64-NEXT: movl %edi, %eax
+; X64-NEXT: shll $6, %eax
+; X64-NEXT: shrdw $6, %si, %ax
+; X64-NEXT: # kill: def $ax killed $ax killed $eax
+; X64-NEXT: retq
+;
+; X64-SLOW-LABEL: insert_6_i16:
+; X64-SLOW: # %bb.0:
+; X64-SLOW-NEXT: # kill: def $esi killed $esi def $rsi
+; X64-SLOW-NEXT: # kill: def $edi killed $edi def $rdi
+; X64-SLOW-NEXT: shll $10, %esi
+; X64-SLOW-NEXT: andl $1023, %edi # imm = 0x3FF
+; X64-SLOW-NEXT: leal (%rdi,%rsi), %eax
+; X64-SLOW-NEXT: # kill: def $ax killed $ax killed $eax
+; X64-SLOW-NEXT: retq
+;
+; X64-AVX2-LABEL: insert_6_i16:
+; X64-AVX2: # %bb.0:
+; X64-AVX2-NEXT: movl %edi, %eax
+; X64-AVX2-NEXT: shll $6, %eax
+; X64-AVX2-NEXT: shrdw $6, %si, %ax
+; X64-AVX2-NEXT: # kill: def $ax killed $ax killed $eax
+; X64-AVX2-NEXT: retq
+;
+; X64-AVX512-LABEL: insert_6_i16:
+; X64-AVX512: # %bb.0:
+; X64-AVX512-NEXT: movl %edi, %eax
+; X64-AVX512-NEXT: shll $6, %eax
+; X64-AVX512-NEXT: shrdw $6, %si, %ax
+; X64-AVX512-NEXT: # kill: def $ax killed $ax killed $eax
+; X64-AVX512-NEXT: retq
+;
+; X64-AVX512-BMI2-LABEL: insert_6_i16:
+; X64-AVX512-BMI2: # %bb.0:
+; X64-AVX512-BMI2-NEXT: movl %edi, %eax
+; X64-AVX512-BMI2-NEXT: shll $6, %eax
+; X64-AVX512-BMI2-NEXT: shrdw $6, %si, %ax
+; X64-AVX512-BMI2-NEXT: # kill: def $ax killed $ax killed $eax
+; X64-AVX512-BMI2-NEXT: retq
+ %and = and i16 %a, 1023
+ %shl = shl i16 %b, 10
+ %or = or i16 %shl, %and
+ ret i16 %or
+}
+
+; Negative test for i8: SHRD does not accept an 8-bit operand
+define i8 @insert_3_i8(i8 %a, i8 %b) nounwind {
+; X64-LABEL: insert_3_i8:
+; X64: # %bb.0:
+; X64-NEXT: # kill: def $esi killed $esi def $rsi
+; X64-NEXT: # kill: def $edi killed $edi def $rdi
+; X64-NEXT: shlb $5, %sil
+; X64-NEXT: andb $31, %dil
+; X64-NEXT: leal (%rdi,%rsi), %eax
+; X64-NEXT: # kill: def $al killed $al killed $eax
+; X64-NEXT: retq
+;
+; X64-SLOW-LABEL: insert_3_i8:
+; X64-SLOW: # %bb.0:
+; X64-SLOW-NEXT: # kill: def $esi killed $esi def $rsi
+; X64-SLOW-NEXT: # kill: def $edi killed $edi def $rdi
+; X64-SLOW-NEXT: shlb $5, %sil
+; X64-SLOW-NEXT: andb $31, %dil
+; X64-SLOW-NEXT: leal (%rdi,%rsi), %eax
+; X64-SLOW-NEXT: # kill: def $al killed $al killed $eax
+; X64-SLOW-NEXT: retq
+;
+; X64-AVX2-LABEL: insert_3_i8:
+; X64-AVX2: # %bb.0:
+; X64-AVX2-NEXT: # kill: def $esi killed $esi def $rsi
+; X64-AVX2-NEXT: # kill: def $edi killed $edi def $rdi
+; X64-AVX2-NEXT: shlb $5, %sil
+; X64-AVX2-NEXT: andb $31, %dil
+; X64-AVX2-NEXT: leal (%rdi,%rsi), %eax
+; X64-AVX2-NEXT: # kill: def $al killed $al killed $eax
+; X64-AVX2-NEXT: retq
+;
+; X64-AVX512-LABEL: insert_3_i8:
+; X64-AVX512: # %bb.0:
+; X64-AVX512-NEXT: # kill: def $esi killed $esi def $rsi
+; X64-AVX512-NEXT: # kill: def $edi killed $edi def $rdi
+; X64-AVX512-NEXT: shlb $5, %sil
+; X64-AVX512-NEXT: andb $31, %dil
+; X64-AVX512-NEXT: leal (%rdi,%rsi), %eax
+; X64-AVX512-NEXT: # kill: def $al killed $al killed $eax
+; X64-AVX512-NEXT: retq
+;
+; X64-AVX512-BMI2-LABEL: insert_3_i8:
+; X64-AVX512-BMI2: # %bb.0:
+; X64-AVX512-BMI2-NEXT: # kill: def $esi killed $esi def $rsi
+; X64-AVX512-BMI2-NEXT: # kill: def $edi killed $edi def $rdi
+; X64-AVX512-BMI2-NEXT: shlb $5, %sil
+; X64-AVX512-BMI2-NEXT: andb $31, %dil
+; X64-AVX512-BMI2-NEXT: leal (%rdi,%rsi), %eax
+; X64-AVX512-BMI2-NEXT: # kill: def $al killed $al killed $eax
+; X64-AVX512-BMI2-NEXT: retq
+ %and = and i8 %a, 31
+ %shl = shl i8 %b, 5
+ %or = or i8 %shl, %and
+ ret i8 %or
+}
+
+; Negative test: the mask and the shift amount describe different split points,
+; so the result is not a plain concatenation of A and B.
+define i64 @mask_shift_mismatch(i64 %a, i64 %b) nounwind {
+; X64-LABEL: mask_shift_mismatch:
+; X64: # %bb.0:
+; X64-NEXT: movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-NEXT: andq %rdi, %rax
+; X64-NEXT: shlq $55, %rsi
+; X64-NEXT: orq %rsi, %rax
+; X64-NEXT: retq
+;
+; X64-SLOW-LABEL: mask_shift_mismatch:
+; X64-SLOW: # %bb.0:
+; X64-SLOW-NEXT: movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT: andq %rdi, %rax
+; X64-SLOW-NEXT: shlq $55, %rsi
+; X64-SLOW-NEXT: orq %rsi, %rax
+; X64-SLOW-NEXT: retq
+;
+; X64-AVX2-LABEL: mask_shift_mismatch:
+; X64-AVX2: # %bb.0:
+; X64-AVX2-NEXT: movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-AVX2-NEXT: andq %rdi, %rax
+; X64-AVX2-NEXT: shlq $55, %rsi
+; X64-AVX2-NEXT: orq %rsi, %rax
+; X64-AVX2-NEXT: retq
+;
+; X64-AVX512-LABEL: mask_shift_mismatch:
+; X64-AVX512: # %bb.0:
+; X64-AVX512-NEXT: movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-AVX512-NEXT: andq %rdi, %rax
+; X64-AVX512-NEXT: shlq $55, %rsi
+; X64-AVX512-NEXT: orq %rsi, %rax
+; X64-AVX512-NEXT: retq
+;
+; X64-AVX512-BMI2-LABEL: mask_shift_mismatch:
+; X64-AVX512-BMI2: # %bb.0:
+; X64-AVX512-BMI2-NEXT: movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-AVX512-BMI2-NEXT: andq %rdi, %rax
+; X64-AVX512-BMI2-NEXT: shlq $55, %rsi
+; X64-AVX512-BMI2-NEXT: orq %rsi, %rax
+; X64-AVX512-BMI2-NEXT: retq
+ %and = and i64 %a, 18014398509481983
+ %shl = shl i64 %b, 55
+ %or = or i64 %shl, %and
+ ret i64 %or
+}
+
+; Negative test: the mask keeps too many bits, so the shifted-in value would be
+; OR'd on top of bits of A instead of replacing them.
+define i64 @mask_too_wide(i64 %a, i64 %b) nounwind {
+; X64-LABEL: mask_too_wide:
+; X64: # %bb.0:
+; X64-NEXT: movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-NEXT: andq %rdi, %rax
+; X64-NEXT: shlq $54, %rsi
+; X64-NEXT: orq %rsi, %rax
+; X64-NEXT: retq
+;
+; X64-SLOW-LABEL: mask_too_wide:
+; X64-SLOW: # %bb.0:
+; X64-SLOW-NEXT: movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-SLOW-NEXT: andq %rdi, %rax
+; X64-SLOW-NEXT: shlq $54, %rsi
+; X64-SLOW-NEXT: orq %rsi, %rax
+; X64-SLOW-NEXT: retq
+;
+; X64-AVX2-LABEL: mask_too_wide:
+; X64-AVX2: # %bb.0:
+; X64-AVX2-NEXT: movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-AVX2-NEXT: andq %rdi, %rax
+; X64-AVX2-NEXT: shlq $54, %rsi
+; X64-AVX2-NEXT: orq %rsi, %rax
+; X64-AVX2-NEXT: retq
+;
+; X64-AVX512-LABEL: mask_too_wide:
+; X64-AVX512: # %bb.0:
+; X64-AVX512-NEXT: movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-AVX512-NEXT: andq %rdi, %rax
+; X64-AVX512-NEXT: shlq $54, %rsi
+; X64-AVX512-NEXT: orq %rsi, %rax
+; X64-AVX512-NEXT: retq
+;
+; X64-AVX512-BMI2-LABEL: mask_too_wide:
+; X64-AVX512-BMI2: # %bb.0:
+; X64-AVX512-BMI2-NEXT: movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-AVX512-BMI2-NEXT: andq %rdi, %rax
+; X64-AVX512-BMI2-NEXT: shlq $54, %rsi
+; X64-AVX512-BMI2-NEXT: orq %rsi, %rax
+; X64-AVX512-BMI2-NEXT: retq
+ %and = and i64 %a, 36028797018963967
+ %shl = shl i64 %b, 54
+ %or = or i64 %shl, %and
+ ret i64 %or
+}
+
+; Negative test: e...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/220182
More information about the llvm-commits
mailing list