[llvm] [DAG][X86] Bitfield insertion can combine into SHL + SHRD (PR #220182)

via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 15 01:01:38 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-x86

Author: Pratyay Pande (pratyaypande)

<details>
<summary>Changes</summary>

This change combines the following pattern:

```llvm
define dso_local noundef i64 @<!-- -->updateTop10Bits(unsigned long, unsigned long)(i64 noundef %A, i64 noundef %B) local_unnamed_addr {
entry:
  %and = and i64 %A, 18014398509481983
  %shl = shl i64 %B, 54
  %or = or disjoint i64 %shl, %and
  ret i64 %or
}
```

Into an `shrd` usage instead of `bhziq`.

This change is intended to fix issue #<!-- -->112488 

---

Patch is 62.50 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/220182.diff


3 Files Affected:

- (modified) llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp (+42) 
- (modified) llvm/lib/Target/X86/X86ISelLowering.cpp (+36-1) 
- (added) llvm/test/CodeGen/X86/insert-bitfield.ll (+1614) 


``````````diff
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index a829d7a34d5a6..6f62442690542 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -8844,6 +8844,44 @@ static SDValue visitORCommutative(SelectionDAG &DAG, SDValue N0, SDValue N1,
   return SDValue();
 }
 
+// Fold an OR with a masked destination and a left-shifted
+// source into a shift + double-precision shift (SHRD):
+static SDValue combineOrOnSHLToFSHR(SDNode *N, SDLoc &DL, SelectionDAG &DAG) {
+  EVT VT = N->getValueType(0);
+
+  APInt Mask, ShiftAmount;
+  SDValue X, Y;
+
+  // Check for the following pattern:
+  //   (or (and X, HighBitsMask(C)), (srl Y, C))
+  // Do not combine if there are multi-use AND and OR.
+  // It does not result in more performant code.
+  if (!sd_match(N, m_Or(m_OneUse(m_And(m_Value(X), m_ConstInt(Mask))),
+                        m_OneUse(m_Shl(m_Value(Y), m_ConstInt(ShiftAmount))))))
+    return SDValue();
+
+  // Max bit-width of operands
+  uint64_t MaxMaskBitWidth = VT.getScalarSizeInBits();
+
+  // Check for Mask and ShiftAmount
+  //
+  // (shl Y, ShiftAmount) fills the top (MaxMaskBitWidth - ShiftAmount) bits,
+  // so X must keep exactly the low ShiftAmount.
+  APInt ExpectedMask =
+      APInt::getLowBitsSet(MaxMaskBitWidth, ShiftAmount.getZExtValue());
+
+  if (!((ShiftAmount.getZExtValue() > 0) &&
+        (ShiftAmount.getZExtValue() < MaxMaskBitWidth) &&
+        (Mask == ExpectedMask)))
+    return SDValue();
+
+  uint64_t InvShAmt = MaxMaskBitWidth - ShiftAmount.getZExtValue();
+  SDValue ShAConst = DAG.getShiftAmountConstant(InvShAmt, VT, DL);
+  SDValue SHLVal = DAG.getNode(ISD::SHL, DL, VT, X, ShAConst);
+
+  return DAG.getNode(ISD::FSHR, DL, VT, Y, SHLVal, ShAConst);
+}
+
 SDValue DAGCombiner::visitOR(SDNode *N) {
   SDValue N0 = N->getOperand(0);
   SDValue N1 = N->getOperand(1);
@@ -9043,6 +9081,10 @@ SDValue DAGCombiner::visitOR(SDNode *N) {
   if (SDValue Load = MatchLoadCombine(N))
     return Load;
 
+  if (Level == CombineLevel::BeforeLegalizeTypes)
+    if (SDValue Res = combineOrOnSHLToFSHR(N, DL, DAG))
+      return Res;
+
   // Simplify the operands using demanded-bits information.
   if (SimplifyDemandedBits(SDValue(N, 0)))
     return SDValue(N, 0);
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 775fd7642f040..4d2b4ca836e4e 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -32343,8 +32343,43 @@ static SDValue LowerFunnelShift(SDValue Op, const X86Subtarget &Subtarget,
     return DAG.getZExtOrTrunc(Res, DL, VT);
   }
 
-  if (VT == MVT::i8 || ExpandFunnel)
+  // If expanding the funnel shift is required OR the value type is unsupported
+  if (VT == MVT::i8 || ExpandFunnel) {
+    // DAG Combiner would combine an OR on a masked value and a shifted value
+    // into a shift followed by a funnel shift. The following fold reverses that
+    // operation in case `ExpandFunnel` is set, i.e., if we are not optimizing
+    // for size and if funnel-shift is slow. The fold applies if one operand is
+    // the result of an ISD::SHL or the node is MVT::i8, for which X86 does not
+    // have a funnel-shift instruction.
+    if (Op1.getOpcode() == ISD::SHL && isa<ConstantSDNode>(Amt.getNode())) {
+
+      auto *C = dyn_cast<ConstantSDNode>(Amt.getNode());
+      SDValue SHLOperandShiftAmount = Op1->getOperand(1);
+      uint64_t InvMaskWidth = C->getAPIntValue().urem(EltSizeInBits);
+
+      if (ConstantSDNode *EC =
+              dyn_cast<ConstantSDNode>(SHLOperandShiftAmount.getNode())) {
+        const APInt &ExpectedShiftAmount = EC->getAPIntValue();
+
+        // Check if the shift amounts match.
+        if (ExpectedShiftAmount != InvMaskWidth)
+          return SDValue();
+      }
+
+      uint64_t ShiftAmount = EltSizeInBits - InvMaskWidth;
+      SDValue SHLOperand = Op1.getOperand(0);
+
+      APInt Mask = APInt::getLowBitsSet(EltSizeInBits, ShiftAmount);
+
+      SDValue MaskBitNum = DAG.getShiftAmountConstant(
+          ShiftAmount, SHLOperand.getValueType(), DL);
+      SDValue MaskNode = DAG.getConstant(Mask, DL, VT);
+      SDValue AndMask = DAG.getNode(ISD::AND, DL, VT, SHLOperand, MaskNode);
+      SDValue SHL = DAG.getNode(ISD::SHL, DL, VT, Op0, MaskBitNum);
+      return DAG.getNode(ISD::OR, DL, VT, AndMask, SHL);
+    }
     return SDValue();
+  }
 
   // i16 needs to modulo the shift amount, but i32/i64 have implicit modulo.
   if (VT == MVT::i16) {
diff --git a/llvm/test/CodeGen/X86/insert-bitfield.ll b/llvm/test/CodeGen/X86/insert-bitfield.ll
new file mode 100644
index 0000000000000..ddb0dc85d89a6
--- /dev/null
+++ b/llvm/test/CodeGen/X86/insert-bitfield.ll
@@ -0,0 +1,1614 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-- | FileCheck %s --check-prefixes=X64
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+slow-shld | FileCheck %s --check-prefixes=X64-SLOW
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s --check-prefixes=X64-AVX2
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+avx512f,+avx512bw,+avx512vl | FileCheck %s --check-prefixes=X64-AVX512
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+avx512vbmi2,+avx512vl | FileCheck %s --check-prefixes=X64-AVX512-BMI2
+
+define i64 @insert_10_i64(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_10_i64:
+; X64:       # %bb.0:
+; X64-NEXT:    movq %rdi, %rax
+; X64-NEXT:    shlq $10, %rax
+; X64-NEXT:    shrdq $10, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_10_i64:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+;
+; X64-AVX2-LABEL: insert_10_i64:
+; X64-AVX2:       # %bb.0:
+; X64-AVX2-NEXT:    movq %rdi, %rax
+; X64-AVX2-NEXT:    shlq $10, %rax
+; X64-AVX2-NEXT:    shrdq $10, %rsi, %rax
+; X64-AVX2-NEXT:    retq
+;
+; X64-AVX512-LABEL: insert_10_i64:
+; X64-AVX512:       # %bb.0:
+; X64-AVX512-NEXT:    movq %rdi, %rax
+; X64-AVX512-NEXT:    shlq $10, %rax
+; X64-AVX512-NEXT:    shrdq $10, %rsi, %rax
+; X64-AVX512-NEXT:    retq
+;
+; X64-AVX512-BMI2-LABEL: insert_10_i64:
+; X64-AVX512-BMI2:       # %bb.0:
+; X64-AVX512-BMI2-NEXT:    movq %rdi, %rax
+; X64-AVX512-BMI2-NEXT:    shlq $10, %rax
+; X64-AVX512-BMI2-NEXT:    shrdq $10, %rsi, %rax
+; X64-AVX512-BMI2-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 54
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+define i64 @insert_10_i64_disjoint(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_10_i64_disjoint:
+; X64:       # %bb.0:
+; X64-NEXT:    movq %rdi, %rax
+; X64-NEXT:    shlq $10, %rax
+; X64-NEXT:    shrdq $10, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_10_i64_disjoint:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+;
+; X64-AVX2-LABEL: insert_10_i64_disjoint:
+; X64-AVX2:       # %bb.0:
+; X64-AVX2-NEXT:    movq %rdi, %rax
+; X64-AVX2-NEXT:    shlq $10, %rax
+; X64-AVX2-NEXT:    shrdq $10, %rsi, %rax
+; X64-AVX2-NEXT:    retq
+;
+; X64-AVX512-LABEL: insert_10_i64_disjoint:
+; X64-AVX512:       # %bb.0:
+; X64-AVX512-NEXT:    movq %rdi, %rax
+; X64-AVX512-NEXT:    shlq $10, %rax
+; X64-AVX512-NEXT:    shrdq $10, %rsi, %rax
+; X64-AVX512-NEXT:    retq
+;
+; X64-AVX512-BMI2-LABEL: insert_10_i64_disjoint:
+; X64-AVX512-BMI2:       # %bb.0:
+; X64-AVX512-BMI2-NEXT:    movq %rdi, %rax
+; X64-AVX512-BMI2-NEXT:    shlq $10, %rax
+; X64-AVX512-BMI2-NEXT:    shrdq $10, %rsi, %rax
+; X64-AVX512-BMI2-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 54
+  %or = or disjoint i64 %shl, %and
+  ret i64 %or
+}
+
+; Commuted operands.
+define i64 @insert_10_i64_commute(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_10_i64_commute:
+; X64:       # %bb.0:
+; X64-NEXT:    movq %rdi, %rax
+; X64-NEXT:    shlq $10, %rax
+; X64-NEXT:    shrdq $10, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_10_i64_commute:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+;
+; X64-AVX2-LABEL: insert_10_i64_commute:
+; X64-AVX2:       # %bb.0:
+; X64-AVX2-NEXT:    movq %rdi, %rax
+; X64-AVX2-NEXT:    shlq $10, %rax
+; X64-AVX2-NEXT:    shrdq $10, %rsi, %rax
+; X64-AVX2-NEXT:    retq
+;
+; X64-AVX512-LABEL: insert_10_i64_commute:
+; X64-AVX512:       # %bb.0:
+; X64-AVX512-NEXT:    movq %rdi, %rax
+; X64-AVX512-NEXT:    shlq $10, %rax
+; X64-AVX512-NEXT:    shrdq $10, %rsi, %rax
+; X64-AVX512-NEXT:    retq
+;
+; X64-AVX512-BMI2-LABEL: insert_10_i64_commute:
+; X64-AVX512-BMI2:       # %bb.0:
+; X64-AVX512-BMI2-NEXT:    movq %rdi, %rax
+; X64-AVX512-BMI2-NEXT:    shlq $10, %rax
+; X64-AVX512-BMI2-NEXT:    shrdq $10, %rsi, %rax
+; X64-AVX512-BMI2-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 54
+  %or = or i64 %and, %shl
+  ret i64 %or
+}
+
+define i64 @insert_33_i64(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_33_i64:
+; X64:       # %bb.0:
+; X64-NEXT:    movq %rdi, %rax
+; X64-NEXT:    shlq $33, %rax
+; X64-NEXT:    shrdq $33, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_33_i64:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    shlq $31, %rsi
+; X64-SLOW-NEXT:    andl $2147483647, %edi # imm = 0x7FFFFFFF
+; X64-SLOW-NEXT:    leaq (%rdi,%rsi), %rax
+; X64-SLOW-NEXT:    retq
+;
+; X64-AVX2-LABEL: insert_33_i64:
+; X64-AVX2:       # %bb.0:
+; X64-AVX2-NEXT:    movq %rdi, %rax
+; X64-AVX2-NEXT:    shlq $33, %rax
+; X64-AVX2-NEXT:    shrdq $33, %rsi, %rax
+; X64-AVX2-NEXT:    retq
+;
+; X64-AVX512-LABEL: insert_33_i64:
+; X64-AVX512:       # %bb.0:
+; X64-AVX512-NEXT:    movq %rdi, %rax
+; X64-AVX512-NEXT:    shlq $33, %rax
+; X64-AVX512-NEXT:    shrdq $33, %rsi, %rax
+; X64-AVX512-NEXT:    retq
+;
+; X64-AVX512-BMI2-LABEL: insert_33_i64:
+; X64-AVX512-BMI2:       # %bb.0:
+; X64-AVX512-BMI2-NEXT:    movq %rdi, %rax
+; X64-AVX512-BMI2-NEXT:    shlq $33, %rax
+; X64-AVX512-BMI2-NEXT:    shrdq $33, %rsi, %rax
+; X64-AVX512-BMI2-NEXT:    retq
+  %and = and i64 %a, 2147483647
+  %shl = shl i64 %b, 31
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+define i64 @insert_1_i64(i64 %a, i64 %b) nounwind {
+; X64-LABEL: insert_1_i64:
+; X64:       # %bb.0:
+; X64-NEXT:    leaq (%rdi,%rdi), %rax
+; X64-NEXT:    shrdq $1, %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_1_i64:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $9223372036854775807, %rax # imm = 0x7FFFFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $63, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+;
+; X64-AVX2-LABEL: insert_1_i64:
+; X64-AVX2:       # %bb.0:
+; X64-AVX2-NEXT:    leaq (%rdi,%rdi), %rax
+; X64-AVX2-NEXT:    shrdq $1, %rsi, %rax
+; X64-AVX2-NEXT:    retq
+;
+; X64-AVX512-LABEL: insert_1_i64:
+; X64-AVX512:       # %bb.0:
+; X64-AVX512-NEXT:    leaq (%rdi,%rdi), %rax
+; X64-AVX512-NEXT:    shrdq $1, %rsi, %rax
+; X64-AVX512-NEXT:    retq
+;
+; X64-AVX512-BMI2-LABEL: insert_1_i64:
+; X64-AVX512-BMI2:       # %bb.0:
+; X64-AVX512-BMI2-NEXT:    leaq (%rdi,%rdi), %rax
+; X64-AVX512-BMI2-NEXT:    shrdq $1, %rsi, %rax
+; X64-AVX512-BMI2-NEXT:    retq
+  %and = and i64 %a, 9223372036854775807
+  %shl = shl i64 %b, 63
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+; i32 inputs and mask
+define i32 @insert_10_i32(i32 %a, i32 %b) nounwind {
+; X64-LABEL: insert_10_i32:
+; X64:       # %bb.0:
+; X64-NEXT:    movl %edi, %eax
+; X64-NEXT:    shll $10, %eax
+; X64-NEXT:    shrdl $10, %esi, %eax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_10_i32:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    # kill: def $esi killed $esi def $rsi
+; X64-SLOW-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-SLOW-NEXT:    shll $22, %esi
+; X64-SLOW-NEXT:    andl $4194303, %edi # imm = 0x3FFFFF
+; X64-SLOW-NEXT:    leal (%rdi,%rsi), %eax
+; X64-SLOW-NEXT:    retq
+;
+; X64-AVX2-LABEL: insert_10_i32:
+; X64-AVX2:       # %bb.0:
+; X64-AVX2-NEXT:    movl %edi, %eax
+; X64-AVX2-NEXT:    shll $10, %eax
+; X64-AVX2-NEXT:    shrdl $10, %esi, %eax
+; X64-AVX2-NEXT:    retq
+;
+; X64-AVX512-LABEL: insert_10_i32:
+; X64-AVX512:       # %bb.0:
+; X64-AVX512-NEXT:    movl %edi, %eax
+; X64-AVX512-NEXT:    shll $10, %eax
+; X64-AVX512-NEXT:    shrdl $10, %esi, %eax
+; X64-AVX512-NEXT:    retq
+;
+; X64-AVX512-BMI2-LABEL: insert_10_i32:
+; X64-AVX512-BMI2:       # %bb.0:
+; X64-AVX512-BMI2-NEXT:    movl %edi, %eax
+; X64-AVX512-BMI2-NEXT:    shll $10, %eax
+; X64-AVX512-BMI2-NEXT:    shrdl $10, %esi, %eax
+; X64-AVX512-BMI2-NEXT:    retq
+  %and = and i32 %a, 4194303
+  %shl = shl i32 %b, 22
+  %or = or i32 %shl, %and
+  ret i32 %or
+}
+
+; i16 inputs and mask
+define i16 @insert_6_i16(i16 %a, i16 %b) nounwind {
+; X64-LABEL: insert_6_i16:
+; X64:       # %bb.0:
+; X64-NEXT:    movl %edi, %eax
+; X64-NEXT:    shll $6, %eax
+; X64-NEXT:    shrdw $6, %si, %ax
+; X64-NEXT:    # kill: def $ax killed $ax killed $eax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_6_i16:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    # kill: def $esi killed $esi def $rsi
+; X64-SLOW-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-SLOW-NEXT:    shll $10, %esi
+; X64-SLOW-NEXT:    andl $1023, %edi # imm = 0x3FF
+; X64-SLOW-NEXT:    leal (%rdi,%rsi), %eax
+; X64-SLOW-NEXT:    # kill: def $ax killed $ax killed $eax
+; X64-SLOW-NEXT:    retq
+;
+; X64-AVX2-LABEL: insert_6_i16:
+; X64-AVX2:       # %bb.0:
+; X64-AVX2-NEXT:    movl %edi, %eax
+; X64-AVX2-NEXT:    shll $6, %eax
+; X64-AVX2-NEXT:    shrdw $6, %si, %ax
+; X64-AVX2-NEXT:    # kill: def $ax killed $ax killed $eax
+; X64-AVX2-NEXT:    retq
+;
+; X64-AVX512-LABEL: insert_6_i16:
+; X64-AVX512:       # %bb.0:
+; X64-AVX512-NEXT:    movl %edi, %eax
+; X64-AVX512-NEXT:    shll $6, %eax
+; X64-AVX512-NEXT:    shrdw $6, %si, %ax
+; X64-AVX512-NEXT:    # kill: def $ax killed $ax killed $eax
+; X64-AVX512-NEXT:    retq
+;
+; X64-AVX512-BMI2-LABEL: insert_6_i16:
+; X64-AVX512-BMI2:       # %bb.0:
+; X64-AVX512-BMI2-NEXT:    movl %edi, %eax
+; X64-AVX512-BMI2-NEXT:    shll $6, %eax
+; X64-AVX512-BMI2-NEXT:    shrdw $6, %si, %ax
+; X64-AVX512-BMI2-NEXT:    # kill: def $ax killed $ax killed $eax
+; X64-AVX512-BMI2-NEXT:    retq
+  %and = and i16 %a, 1023
+  %shl = shl i16 %b, 10
+  %or = or i16 %shl, %and
+  ret i16 %or
+}
+
+; Negative test for i8: SHRD does not accept an 8-bit operand
+define i8 @insert_3_i8(i8 %a, i8 %b) nounwind {
+; X64-LABEL: insert_3_i8:
+; X64:       # %bb.0:
+; X64-NEXT:    # kill: def $esi killed $esi def $rsi
+; X64-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-NEXT:    shlb $5, %sil
+; X64-NEXT:    andb $31, %dil
+; X64-NEXT:    leal (%rdi,%rsi), %eax
+; X64-NEXT:    # kill: def $al killed $al killed $eax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: insert_3_i8:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    # kill: def $esi killed $esi def $rsi
+; X64-SLOW-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-SLOW-NEXT:    shlb $5, %sil
+; X64-SLOW-NEXT:    andb $31, %dil
+; X64-SLOW-NEXT:    leal (%rdi,%rsi), %eax
+; X64-SLOW-NEXT:    # kill: def $al killed $al killed $eax
+; X64-SLOW-NEXT:    retq
+;
+; X64-AVX2-LABEL: insert_3_i8:
+; X64-AVX2:       # %bb.0:
+; X64-AVX2-NEXT:    # kill: def $esi killed $esi def $rsi
+; X64-AVX2-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-AVX2-NEXT:    shlb $5, %sil
+; X64-AVX2-NEXT:    andb $31, %dil
+; X64-AVX2-NEXT:    leal (%rdi,%rsi), %eax
+; X64-AVX2-NEXT:    # kill: def $al killed $al killed $eax
+; X64-AVX2-NEXT:    retq
+;
+; X64-AVX512-LABEL: insert_3_i8:
+; X64-AVX512:       # %bb.0:
+; X64-AVX512-NEXT:    # kill: def $esi killed $esi def $rsi
+; X64-AVX512-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-AVX512-NEXT:    shlb $5, %sil
+; X64-AVX512-NEXT:    andb $31, %dil
+; X64-AVX512-NEXT:    leal (%rdi,%rsi), %eax
+; X64-AVX512-NEXT:    # kill: def $al killed $al killed $eax
+; X64-AVX512-NEXT:    retq
+;
+; X64-AVX512-BMI2-LABEL: insert_3_i8:
+; X64-AVX512-BMI2:       # %bb.0:
+; X64-AVX512-BMI2-NEXT:    # kill: def $esi killed $esi def $rsi
+; X64-AVX512-BMI2-NEXT:    # kill: def $edi killed $edi def $rdi
+; X64-AVX512-BMI2-NEXT:    shlb $5, %sil
+; X64-AVX512-BMI2-NEXT:    andb $31, %dil
+; X64-AVX512-BMI2-NEXT:    leal (%rdi,%rsi), %eax
+; X64-AVX512-BMI2-NEXT:    # kill: def $al killed $al killed $eax
+; X64-AVX512-BMI2-NEXT:    retq
+  %and = and i8 %a, 31
+  %shl = shl i8 %b, 5
+  %or = or i8 %shl, %and
+  ret i8 %or
+}
+
+; Negative test: the mask and the shift amount describe different split points,
+; so the result is not a plain concatenation of A and B.
+define i64 @mask_shift_mismatch(i64 %a, i64 %b) nounwind {
+; X64-LABEL: mask_shift_mismatch:
+; X64:       # %bb.0:
+; X64-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-NEXT:    andq %rdi, %rax
+; X64-NEXT:    shlq $55, %rsi
+; X64-NEXT:    orq %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: mask_shift_mismatch:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $55, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+;
+; X64-AVX2-LABEL: mask_shift_mismatch:
+; X64-AVX2:       # %bb.0:
+; X64-AVX2-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-AVX2-NEXT:    andq %rdi, %rax
+; X64-AVX2-NEXT:    shlq $55, %rsi
+; X64-AVX2-NEXT:    orq %rsi, %rax
+; X64-AVX2-NEXT:    retq
+;
+; X64-AVX512-LABEL: mask_shift_mismatch:
+; X64-AVX512:       # %bb.0:
+; X64-AVX512-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-AVX512-NEXT:    andq %rdi, %rax
+; X64-AVX512-NEXT:    shlq $55, %rsi
+; X64-AVX512-NEXT:    orq %rsi, %rax
+; X64-AVX512-NEXT:    retq
+;
+; X64-AVX512-BMI2-LABEL: mask_shift_mismatch:
+; X64-AVX512-BMI2:       # %bb.0:
+; X64-AVX512-BMI2-NEXT:    movabsq $18014398509481983, %rax # imm = 0x3FFFFFFFFFFFFF
+; X64-AVX512-BMI2-NEXT:    andq %rdi, %rax
+; X64-AVX512-BMI2-NEXT:    shlq $55, %rsi
+; X64-AVX512-BMI2-NEXT:    orq %rsi, %rax
+; X64-AVX512-BMI2-NEXT:    retq
+  %and = and i64 %a, 18014398509481983
+  %shl = shl i64 %b, 55
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+; Negative test: the mask keeps too many bits, so the shifted-in value would be
+; OR'd on top of bits of A instead of replacing them.
+define i64 @mask_too_wide(i64 %a, i64 %b) nounwind {
+; X64-LABEL: mask_too_wide:
+; X64:       # %bb.0:
+; X64-NEXT:    movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-NEXT:    andq %rdi, %rax
+; X64-NEXT:    shlq $54, %rsi
+; X64-NEXT:    orq %rsi, %rax
+; X64-NEXT:    retq
+;
+; X64-SLOW-LABEL: mask_too_wide:
+; X64-SLOW:       # %bb.0:
+; X64-SLOW-NEXT:    movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-SLOW-NEXT:    andq %rdi, %rax
+; X64-SLOW-NEXT:    shlq $54, %rsi
+; X64-SLOW-NEXT:    orq %rsi, %rax
+; X64-SLOW-NEXT:    retq
+;
+; X64-AVX2-LABEL: mask_too_wide:
+; X64-AVX2:       # %bb.0:
+; X64-AVX2-NEXT:    movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-AVX2-NEXT:    andq %rdi, %rax
+; X64-AVX2-NEXT:    shlq $54, %rsi
+; X64-AVX2-NEXT:    orq %rsi, %rax
+; X64-AVX2-NEXT:    retq
+;
+; X64-AVX512-LABEL: mask_too_wide:
+; X64-AVX512:       # %bb.0:
+; X64-AVX512-NEXT:    movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-AVX512-NEXT:    andq %rdi, %rax
+; X64-AVX512-NEXT:    shlq $54, %rsi
+; X64-AVX512-NEXT:    orq %rsi, %rax
+; X64-AVX512-NEXT:    retq
+;
+; X64-AVX512-BMI2-LABEL: mask_too_wide:
+; X64-AVX512-BMI2:       # %bb.0:
+; X64-AVX512-BMI2-NEXT:    movabsq $36028797018963967, %rax # imm = 0x7FFFFFFFFFFFFF
+; X64-AVX512-BMI2-NEXT:    andq %rdi, %rax
+; X64-AVX512-BMI2-NEXT:    shlq $54, %rsi
+; X64-AVX512-BMI2-NEXT:    orq %rsi, %rax
+; X64-AVX512-BMI2-NEXT:    retq
+  %and = and i64 %a, 36028797018963967
+  %shl = shl i64 %b, 54
+  %or = or i64 %shl, %and
+  ret i64 %or
+}
+
+; Negative test: e...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/220182


More information about the llvm-commits mailing list