[llvm] [LegalizeIntegerTypes] Add `PromoteIntOp_ANY_EXTEND_VECTOR_INREG` (PR #178144)

via llvm-commits llvm-commits at lists.llvm.org
Wed Jan 28 07:04:34 PST 2026


llvmbot wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-x86

Author: Abhishek Kaushik (abhishek-kaushik22)

<details>
<summary>Changes</summary>

Fixes #<!-- -->161013

---

Patch is 49.65 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/178144.diff


4 Files Affected:

- (modified) llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp (+13) 
- (modified) llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h (+1) 
- (added) llvm/test/CodeGen/AArch64/pr161013.ll (+49) 
- (added) llvm/test/CodeGen/X86/pr161013.ll (+1124) 


``````````diff
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
index 8ce41df6be69b..5b32c5f945a75 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
@@ -2044,6 +2044,9 @@ bool DAGTypeLegalizer::PromoteIntegerOperand(SDNode *N, unsigned OpNo) {
     report_fatal_error("Do not know how to promote this operator's operand!");
 
   case ISD::ANY_EXTEND:   Res = PromoteIntOp_ANY_EXTEND(N); break;
+  case ISD::ANY_EXTEND_VECTOR_INREG:
+    Res = PromoteIntOp_ANY_EXTEND_VECTOR_INREG(N);
+    break;
   case ISD::ATOMIC_STORE:
     Res = PromoteIntOp_ATOMIC_STORE(cast<AtomicSDNode>(N));
     break;
@@ -2284,6 +2287,16 @@ SDValue DAGTypeLegalizer::PromoteIntOp_ANY_EXTEND(SDNode *N) {
   return DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), N->getValueType(0), Op);
 }
 
+SDValue DAGTypeLegalizer::PromoteIntOp_ANY_EXTEND_VECTOR_INREG(SDNode *N) {
+  SDValue Op = GetPromotedInteger(N->getOperand(0));
+  EVT ResVT = N->getValueType(0);
+  EVT OpVT = Op.getValueType();
+  EVT NewVT = EVT::getVectorVT(*DAG.getContext(), OpVT.getScalarType(),
+                               ResVT.getVectorNumElements());
+  Op = DAG.getExtractSubvector(SDLoc(Op), NewVT, Op, 0);
+  return DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), ResVT, Op);
+}
+
 SDValue DAGTypeLegalizer::PromoteIntOp_ATOMIC_STORE(AtomicSDNode *N) {
   SDValue Op1 = GetPromotedInteger(N->getOperand(1));
   return DAG.getAtomic(N->getOpcode(), SDLoc(N), N->getMemoryVT(),
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index a39e419e5ad1c..681ceb22c0ad3 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -389,6 +389,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
   // Integer Operand Promotion.
   bool PromoteIntegerOperand(SDNode *N, unsigned OpNo);
   SDValue PromoteIntOp_ANY_EXTEND(SDNode *N);
+  SDValue PromoteIntOp_ANY_EXTEND_VECTOR_INREG(SDNode *N);
   SDValue PromoteIntOp_ATOMIC_STORE(AtomicSDNode *N);
   SDValue PromoteIntOp_BITCAST(SDNode *N);
   SDValue PromoteIntOp_BUILD_PAIR(SDNode *N);
diff --git a/llvm/test/CodeGen/AArch64/pr161013.ll b/llvm/test/CodeGen/AArch64/pr161013.ll
new file mode 100644
index 0000000000000..d163914f1ac0e
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/pr161013.ll
@@ -0,0 +1,49 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=aarch64-- < %s | FileCheck %s
+
+define <16 x i4> @avir_v2i4_v16i4(<2 x i4> %arg) nounwind {
+; CHECK-LABEL: avir_v2i4_v16i4:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    sub sp, sp, #16
+; CHECK-NEXT:    uzp1 v0.4h, v0.4h, v0.4h
+; CHECK-NEXT:    str d0, [sp, #8]
+; CHECK-NEXT:    ldr x8, [sp, #8]
+; CHECK-NEXT:    and w10, w8, #0xf
+; CHECK-NEXT:    ubfx w9, w8, #4, #4
+; CHECK-NEXT:    fmov s0, w10
+; CHECK-NEXT:    mov v0.b[1], w9
+; CHECK-NEXT:    ubfx w9, w8, #8, #4
+; CHECK-NEXT:    mov v0.b[2], w9
+; CHECK-NEXT:    ubfx w9, w8, #12, #4
+; CHECK-NEXT:    mov v0.b[3], w9
+; CHECK-NEXT:    ubfx w9, w8, #16, #4
+; CHECK-NEXT:    mov v0.b[4], w9
+; CHECK-NEXT:    ubfx w9, w8, #20, #4
+; CHECK-NEXT:    mov v0.b[5], w9
+; CHECK-NEXT:    ubfx w9, w8, #24, #4
+; CHECK-NEXT:    mov v0.b[6], w9
+; CHECK-NEXT:    lsr w9, w8, #28
+; CHECK-NEXT:    mov v0.b[7], w9
+; CHECK-NEXT:    ubfx x9, x8, #32, #4
+; CHECK-NEXT:    mov v0.b[8], w9
+; CHECK-NEXT:    ubfx x9, x8, #36, #4
+; CHECK-NEXT:    mov v0.b[9], w9
+; CHECK-NEXT:    ubfx x9, x8, #40, #4
+; CHECK-NEXT:    mov v0.b[10], w9
+; CHECK-NEXT:    ubfx x9, x8, #44, #4
+; CHECK-NEXT:    mov v0.b[11], w9
+; CHECK-NEXT:    ubfx x9, x8, #48, #4
+; CHECK-NEXT:    mov v0.b[12], w9
+; CHECK-NEXT:    ubfx x9, x8, #52, #4
+; CHECK-NEXT:    mov v0.b[13], w9
+; CHECK-NEXT:    ubfx x9, x8, #56, #4
+; CHECK-NEXT:    lsr x8, x8, #60
+; CHECK-NEXT:    mov v0.b[14], w9
+; CHECK-NEXT:    mov v0.b[15], w8
+; CHECK-NEXT:    add sp, sp, #16
+; CHECK-NEXT:    ret
+  %res = shufflevector <2 x i4> %arg, <2 x i4> poison,
+  <16 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 1     , i32 poison, i32 poison, i32 poison,
+              i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+  ret <16 x i4> %res
+}
diff --git a/llvm/test/CodeGen/X86/pr161013.ll b/llvm/test/CodeGen/X86/pr161013.ll
new file mode 100644
index 0000000000000..2e805047f5842
--- /dev/null
+++ b/llvm/test/CodeGen/X86/pr161013.ll
@@ -0,0 +1,1124 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+avx              | FileCheck %s --check-prefixes=AVX,AVX1
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+avx2             | FileCheck %s --check-prefixes=AVX,AVX2
+; RUN: llc < %s -mtriple=x86_64-- -mattr=+avx512f          | FileCheck %s --check-prefixes=AVX,AVX512
+
+
+define <32 x i4> @avir_v4i4_to_v32i4(<4 x i4> %arg) {
+; AVX1-LABEL: avir_v4i4_to_v32i4:
+; AVX1:       # %bb.0:
+; AVX1-NEXT:    vpshufb {{.*#+}} xmm0 = xmm0[0,1,4,5,8,9,12,13,8,9,12,13,12,13,14,15]
+; AVX1-NEXT:    vmovdqa %xmm0, -{{[0-9]+}}(%rsp)
+; AVX1-NEXT:    movq -{{[0-9]+}}(%rsp), %rax
+; AVX1-NEXT:    movq -{{[0-9]+}}(%rsp), %rcx
+; AVX1-NEXT:    movl %ecx, %edx
+; AVX1-NEXT:    shrl $4, %edx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    movl %ecx, %esi
+; AVX1-NEXT:    andl $15, %esi
+; AVX1-NEXT:    vmovd %esi, %xmm0
+; AVX1-NEXT:    vpinsrb $1, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movl %ecx, %edx
+; AVX1-NEXT:    shrl $8, %edx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $2, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movl %ecx, %edx
+; AVX1-NEXT:    shrl $12, %edx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $3, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movl %ecx, %edx
+; AVX1-NEXT:    shrl $16, %edx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $4, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movl %ecx, %edx
+; AVX1-NEXT:    shrl $20, %edx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $5, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movl %ecx, %edx
+; AVX1-NEXT:    shrl $24, %edx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $6, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movl %ecx, %edx
+; AVX1-NEXT:    shrl $28, %edx
+; AVX1-NEXT:    vpinsrb $7, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movq %rcx, %rdx
+; AVX1-NEXT:    shrq $32, %rdx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $8, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movq %rcx, %rdx
+; AVX1-NEXT:    shrq $36, %rdx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $9, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movq %rcx, %rdx
+; AVX1-NEXT:    shrq $40, %rdx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $10, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movq %rcx, %rdx
+; AVX1-NEXT:    shrq $44, %rdx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $11, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movq %rcx, %rdx
+; AVX1-NEXT:    shrq $48, %rdx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $12, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movq %rcx, %rdx
+; AVX1-NEXT:    shrq $52, %rdx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $13, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    movq %rcx, %rdx
+; AVX1-NEXT:    shrq $56, %rdx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vpinsrb $14, %edx, %xmm0, %xmm0
+; AVX1-NEXT:    shrq $60, %rcx
+; AVX1-NEXT:    vpinsrb $15, %ecx, %xmm0, %xmm0
+; AVX1-NEXT:    movl %eax, %ecx
+; AVX1-NEXT:    shrl $4, %ecx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    movl %eax, %edx
+; AVX1-NEXT:    andl $15, %edx
+; AVX1-NEXT:    vmovd %edx, %xmm1
+; AVX1-NEXT:    vpinsrb $1, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movl %eax, %ecx
+; AVX1-NEXT:    shrl $8, %ecx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $2, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movl %eax, %ecx
+; AVX1-NEXT:    shrl $12, %ecx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $3, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movl %eax, %ecx
+; AVX1-NEXT:    shrl $16, %ecx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $4, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movl %eax, %ecx
+; AVX1-NEXT:    shrl $20, %ecx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $5, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movl %eax, %ecx
+; AVX1-NEXT:    shrl $24, %ecx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $6, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movl %eax, %ecx
+; AVX1-NEXT:    shrl $28, %ecx
+; AVX1-NEXT:    vpinsrb $7, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movq %rax, %rcx
+; AVX1-NEXT:    shrq $32, %rcx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $8, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movq %rax, %rcx
+; AVX1-NEXT:    shrq $36, %rcx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $9, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movq %rax, %rcx
+; AVX1-NEXT:    shrq $40, %rcx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $10, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movq %rax, %rcx
+; AVX1-NEXT:    shrq $44, %rcx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $11, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movq %rax, %rcx
+; AVX1-NEXT:    shrq $48, %rcx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $12, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movq %rax, %rcx
+; AVX1-NEXT:    shrq $52, %rcx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $13, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    movq %rax, %rcx
+; AVX1-NEXT:    shrq $56, %rcx
+; AVX1-NEXT:    andl $15, %ecx
+; AVX1-NEXT:    vpinsrb $14, %ecx, %xmm1, %xmm1
+; AVX1-NEXT:    shrq $60, %rax
+; AVX1-NEXT:    vpinsrb $15, %eax, %xmm1, %xmm1
+; AVX1-NEXT:    vinsertf128 $1, %xmm0, %ymm1, %ymm0
+; AVX1-NEXT:    retq
+;
+; AVX2-LABEL: avir_v4i4_to_v32i4:
+; AVX2:       # %bb.0:
+; AVX2-NEXT:    vpshufb {{.*#+}} xmm0 = xmm0[0,1,4,5,8,9,12,13,8,9,12,13,12,13,14,15]
+; AVX2-NEXT:    vmovdqa %xmm0, -{{[0-9]+}}(%rsp)
+; AVX2-NEXT:    movq -{{[0-9]+}}(%rsp), %rax
+; AVX2-NEXT:    movq -{{[0-9]+}}(%rsp), %rcx
+; AVX2-NEXT:    movl %ecx, %edx
+; AVX2-NEXT:    shrl $4, %edx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    movl %ecx, %esi
+; AVX2-NEXT:    andl $15, %esi
+; AVX2-NEXT:    vmovd %esi, %xmm0
+; AVX2-NEXT:    vpinsrb $1, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movl %ecx, %edx
+; AVX2-NEXT:    shrl $8, %edx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $2, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movl %ecx, %edx
+; AVX2-NEXT:    shrl $12, %edx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $3, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movl %ecx, %edx
+; AVX2-NEXT:    shrl $16, %edx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $4, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movl %ecx, %edx
+; AVX2-NEXT:    shrl $20, %edx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $5, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movl %ecx, %edx
+; AVX2-NEXT:    shrl $24, %edx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $6, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movl %ecx, %edx
+; AVX2-NEXT:    shrl $28, %edx
+; AVX2-NEXT:    vpinsrb $7, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movq %rcx, %rdx
+; AVX2-NEXT:    shrq $32, %rdx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $8, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movq %rcx, %rdx
+; AVX2-NEXT:    shrq $36, %rdx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $9, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movq %rcx, %rdx
+; AVX2-NEXT:    shrq $40, %rdx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $10, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movq %rcx, %rdx
+; AVX2-NEXT:    shrq $44, %rdx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $11, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movq %rcx, %rdx
+; AVX2-NEXT:    shrq $48, %rdx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $12, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movq %rcx, %rdx
+; AVX2-NEXT:    shrq $52, %rdx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $13, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    movq %rcx, %rdx
+; AVX2-NEXT:    shrq $56, %rdx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vpinsrb $14, %edx, %xmm0, %xmm0
+; AVX2-NEXT:    shrq $60, %rcx
+; AVX2-NEXT:    vpinsrb $15, %ecx, %xmm0, %xmm0
+; AVX2-NEXT:    movl %eax, %ecx
+; AVX2-NEXT:    shrl $4, %ecx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    movl %eax, %edx
+; AVX2-NEXT:    andl $15, %edx
+; AVX2-NEXT:    vmovd %edx, %xmm1
+; AVX2-NEXT:    vpinsrb $1, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movl %eax, %ecx
+; AVX2-NEXT:    shrl $8, %ecx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $2, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movl %eax, %ecx
+; AVX2-NEXT:    shrl $12, %ecx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $3, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movl %eax, %ecx
+; AVX2-NEXT:    shrl $16, %ecx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $4, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movl %eax, %ecx
+; AVX2-NEXT:    shrl $20, %ecx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $5, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movl %eax, %ecx
+; AVX2-NEXT:    shrl $24, %ecx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $6, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movl %eax, %ecx
+; AVX2-NEXT:    shrl $28, %ecx
+; AVX2-NEXT:    vpinsrb $7, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movq %rax, %rcx
+; AVX2-NEXT:    shrq $32, %rcx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $8, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movq %rax, %rcx
+; AVX2-NEXT:    shrq $36, %rcx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $9, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movq %rax, %rcx
+; AVX2-NEXT:    shrq $40, %rcx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $10, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movq %rax, %rcx
+; AVX2-NEXT:    shrq $44, %rcx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $11, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movq %rax, %rcx
+; AVX2-NEXT:    shrq $48, %rcx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $12, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movq %rax, %rcx
+; AVX2-NEXT:    shrq $52, %rcx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $13, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    movq %rax, %rcx
+; AVX2-NEXT:    shrq $56, %rcx
+; AVX2-NEXT:    andl $15, %ecx
+; AVX2-NEXT:    vpinsrb $14, %ecx, %xmm1, %xmm1
+; AVX2-NEXT:    shrq $60, %rax
+; AVX2-NEXT:    vpinsrb $15, %eax, %xmm1, %xmm1
+; AVX2-NEXT:    vinserti128 $1, %xmm0, %ymm1, %ymm0
+; AVX2-NEXT:    retq
+;
+; AVX512-LABEL: avir_v4i4_to_v32i4:
+; AVX512:       # %bb.0:
+; AVX512-NEXT:    vpshufb {{.*#+}} xmm0 = xmm0[0,1,4,5,8,9,12,13,8,9,12,13,12,13,14,15]
+; AVX512-NEXT:    vmovdqa %xmm0, -{{[0-9]+}}(%rsp)
+; AVX512-NEXT:    movq -{{[0-9]+}}(%rsp), %rax
+; AVX512-NEXT:    movq -{{[0-9]+}}(%rsp), %rcx
+; AVX512-NEXT:    movl %ecx, %edx
+; AVX512-NEXT:    shrl $4, %edx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    movl %ecx, %esi
+; AVX512-NEXT:    andl $15, %esi
+; AVX512-NEXT:    vmovd %esi, %xmm0
+; AVX512-NEXT:    vpinsrb $1, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movl %ecx, %edx
+; AVX512-NEXT:    shrl $8, %edx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $2, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movl %ecx, %edx
+; AVX512-NEXT:    shrl $12, %edx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $3, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movl %ecx, %edx
+; AVX512-NEXT:    shrl $16, %edx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $4, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movl %ecx, %edx
+; AVX512-NEXT:    shrl $20, %edx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $5, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movl %ecx, %edx
+; AVX512-NEXT:    shrl $24, %edx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $6, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movl %ecx, %edx
+; AVX512-NEXT:    shrl $28, %edx
+; AVX512-NEXT:    vpinsrb $7, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movq %rcx, %rdx
+; AVX512-NEXT:    shrq $32, %rdx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $8, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movq %rcx, %rdx
+; AVX512-NEXT:    shrq $36, %rdx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $9, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movq %rcx, %rdx
+; AVX512-NEXT:    shrq $40, %rdx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $10, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movq %rcx, %rdx
+; AVX512-NEXT:    shrq $44, %rdx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $11, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movq %rcx, %rdx
+; AVX512-NEXT:    shrq $48, %rdx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $12, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movq %rcx, %rdx
+; AVX512-NEXT:    shrq $52, %rdx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $13, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    movq %rcx, %rdx
+; AVX512-NEXT:    shrq $56, %rdx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vpinsrb $14, %edx, %xmm0, %xmm0
+; AVX512-NEXT:    shrq $60, %rcx
+; AVX512-NEXT:    vpinsrb $15, %ecx, %xmm0, %xmm0
+; AVX512-NEXT:    movl %eax, %ecx
+; AVX512-NEXT:    shrl $4, %ecx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    movl %eax, %edx
+; AVX512-NEXT:    andl $15, %edx
+; AVX512-NEXT:    vmovd %edx, %xmm1
+; AVX512-NEXT:    vpinsrb $1, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movl %eax, %ecx
+; AVX512-NEXT:    shrl $8, %ecx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $2, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movl %eax, %ecx
+; AVX512-NEXT:    shrl $12, %ecx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $3, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movl %eax, %ecx
+; AVX512-NEXT:    shrl $16, %ecx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $4, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movl %eax, %ecx
+; AVX512-NEXT:    shrl $20, %ecx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $5, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movl %eax, %ecx
+; AVX512-NEXT:    shrl $24, %ecx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $6, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movl %eax, %ecx
+; AVX512-NEXT:    shrl $28, %ecx
+; AVX512-NEXT:    vpinsrb $7, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movq %rax, %rcx
+; AVX512-NEXT:    shrq $32, %rcx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $8, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movq %rax, %rcx
+; AVX512-NEXT:    shrq $36, %rcx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $9, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movq %rax, %rcx
+; AVX512-NEXT:    shrq $40, %rcx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $10, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movq %rax, %rcx
+; AVX512-NEXT:    shrq $44, %rcx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $11, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movq %rax, %rcx
+; AVX512-NEXT:    shrq $48, %rcx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $12, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movq %rax, %rcx
+; AVX512-NEXT:    shrq $52, %rcx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $13, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    movq %rax, %rcx
+; AVX512-NEXT:    shrq $56, %rcx
+; AVX512-NEXT:    andl $15, %ecx
+; AVX512-NEXT:    vpinsrb $14, %ecx, %xmm1, %xmm1
+; AVX512-NEXT:    shrq $60, %rax
+; AVX512-NEXT:    vpinsrb $15, %eax, %xmm1, %xmm1
+; AVX512-NEXT:    vinserti128 $1, %xmm0, %ymm1, %ymm0
+; AVX512-NEXT:    retq
+  %res = shufflevector <4 x i4> %arg, <4 x i4> poison,
+  <32 x i32> <i32 poison, i32 poison, i32 poison, i32 poison, i32 1     , i32 poison, i32 poison, i32 poison,
+              i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison,
+              i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison,
+              i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+  ret <32 x i4> %res
+}
+
+define <64 x i4> @avir_v4i4_to_v64i4(<4 x i4> %arg) {
+; AVX-LABEL: avir_v4i4_to_v64i4:
+; AVX:       # %bb.0:
+; AVX-NEXT:    movq %rdi, %rax
+; AVX-NEXT:    vpshufb {{.*#+}} xmm0 = xmm0[0,4,8,12,u,u,u,u,u,u,u,u,u,u,u,u]
+; AVX-NEXT:    vmovdqa %xmm0, (%rdi)
+; AVX-NEXT:    retq
+  %res = shufflevector <4 x i4> %arg, <4 x i4> poison,
+  <64 x i32> <i32 0     , i32 poison, i32 1     , i32 poison, i32 2     , i32 poison, i32 3     , i32 poison,
+     ...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/178144


More information about the llvm-commits mailing list