[llvm] [RISCV] Model Zvzip costs for fixed-length shuffles (PR #227677)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 30 05:02:02 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-risc-v

Author: renndong

<details>
<summary>Changes</summary>

Teach the RISC-V TTI to recognize interleave and deinterleave shuffles
that can be implemented with vzip.vv or vunzipe.v/vunzipo.v. Compute
their costs using the operand containing the interleaved data: the
destination of vzip.vv and the source of vunzipe.v/vunzipo.v.

Assisted-By: Trae CLI (GPT-5)

---

Patch is 42.41 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/227677.diff


3 Files Affected:

- (modified) llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp (+140-9) 
- (modified) llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h (+6) 
- (modified) llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll (+165-72) 


``````````diff
diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index 69554ca2b8155..87dab23c37b27 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -744,29 +744,138 @@ InstructionCost RISCVTTIImpl::getSlideCost(FixedVectorType *Tp,
 }
 
 std::optional<MVT> RISCVTTIImpl::getZvzipVZIPCostVT(MVT InterleavedVT) const {
-  assert(InterleavedVT.isScalableVector() && "Expected a scalable vector type");
   if (!InterleavedVT.getVectorElementCount().isKnownEven())
     return std::nullopt;
 
-  unsigned EltBits = InterleavedVT.getScalarSizeInBits();
-  unsigned MinSize = InterleavedVT.getSizeInBits().getKnownMinValue();
+  MVT CostVT = InterleavedVT;
+  if (InterleavedVT.isFixedLengthVector()) {
+    MVT SourceVT = InterleavedVT.getHalfNumVectorElementsVT();
+    CostVT = TLI->getContainerForFixedLengthVector(SourceVT)
+                 .getDoubleNumVectorElementsVT();
+  }
+
+  unsigned EltBits = CostVT.getScalarSizeInBits();
+  unsigned MinSize = CostVT.getSizeInBits().getKnownMinValue();
   unsigned LMULOctuple = MinSize / (RISCV::RVVBitsPerBlock / 8);
   // Perform the 2 * SEW <= LMUL * min(ELEN, VLEN) check.
-  if (EltBits * 16 >
-      LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
+  if (EltBits * 16 > LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
     return std::nullopt;
-  return InterleavedVT;
+  return CostVT;
 }
 
 std::optional<MVT> RISCVTTIImpl::getZvzipVUNZIPCostVT(MVT InterleavedVT) const {
-  assert(InterleavedVT.isScalableVector() && "Expected a scalable vector type");
   if (!InterleavedVT.getVectorElementCount().isKnownEven())
     return std::nullopt;
 
-  MVT DeinterleavedVT = InterleavedVT.getHalfNumVectorElementsVT();
+  MVT CostVT = InterleavedVT;
+  // lowerZvzipVUNZIP widens the source container if halving it would produce
+  // an illegal result type. Apply the same rule here so the cost uses the
+  // source LMUL selected by ISel.
+  if (InterleavedVT.isFixedLengthVector()) {
+    CostVT = TLI->getContainerForFixedLengthVector(InterleavedVT);
+    if (CostVT.getVectorMinNumElements() == 1 ||
+        !TLI->isTypeLegal(CostVT.getHalfNumVectorElementsVT()))
+      CostVT = CostVT.getDoubleNumVectorElementsVT();
+  }
+
+  MVT DeinterleavedVT = CostVT.getHalfNumVectorElementsVT();
   if (RISCVTargetLowering::getLMUL(DeinterleavedVT) == RISCVVType::LMUL_8)
     return std::nullopt;
-  return InterleavedVT;
+  return CostVT;
+}
+
+InstructionCost RISCVTTIImpl::getVZIPCost(TTI::ShuffleKind Kind,
+                                          VectorType *DstTy, VectorType *SrcTy,
+                                          ArrayRef<int> Mask,
+                                          TTI::TargetCostKind CostKind) const {
+  assert(isa<FixedVectorType>(SrcTy) && "Expected fixed-length vector types");
+
+  if (!ST->hasStdExtZvzip() || Mask.size() < 2 ||
+      DstTy->getScalarSizeInBits() == 1)
+    return InstructionCost::getInvalid();
+
+  unsigned NumSrcElts = SrcTy->getElementCount().getKnownMinValue();
+  unsigned NumDstElts = DstTy->getElementCount().getKnownMinValue();
+  unsigned NumInputElts;
+
+  if (Kind == TTI::SK_PermuteSingleSrc) {
+    // A single-source vzip shuffle interleaves two halves of the same vector,
+    // so its source and destination must have the same number of elements.
+    if (NumSrcElts != NumDstElts)
+      return InstructionCost::getInvalid();
+    NumInputElts = NumSrcElts;
+  } else if (Kind == TTI::SK_PermuteTwoSrc) {
+    // Accept both forms that can be lowered with vzip.vv:
+    //
+    // 1. The destination has twice as many elements as each source, e.g.
+    //      %r = shufflevector <4 x i32> %a, <4 x i32> %b,
+    //                         <8 x i32> <0, 4, 1, 5, 2, 6, 3, 7>
+    //    The two operands are already the half-sized inputs of vzip.vv.
+    //
+    // 2. Each source has the same number of elements as the destination, e.g.
+    //      %r = shufflevector <4 x i32> %a, <4 x i32> %b,
+    //                         <4 x i32> <0, 4, 1, 5>
+    //    ISel first extracts a <2 x i32> half from each source and then uses
+    //    those halves as the inputs of vzip.vv.
+    if (NumSrcElts != NumDstElts && NumSrcElts * 2 != NumDstElts)
+      return InstructionCost::getInvalid();
+    NumInputElts = NumSrcElts * 2;
+  } else {
+    return InstructionCost::getInvalid();
+  }
+
+  // ISel deliberately excludes the two-element single-source identity case.
+  if (Kind == TTI::SK_PermuteSingleSrc && Mask.size() == 2 &&
+      ShuffleVectorInst::isSingleSourceMask(Mask, Mask.size()))
+    return InstructionCost::getInvalid();
+
+  // A deinterleave shuffle for an interleaved load may have a mask such as
+  //   %even = shufflevector <4 x i32> %wide, <4 x i32> poison,
+  //                         <4 x i32> <i32 0, i32 2, i32 poison, i32 poison>
+  // Because isInterleaveMask ignores poison lanes, this mask may also satisfy
+  // the interleave matcher. Do not cost it as vzip.vv.
+  unsigned DeinterleaveIndex;
+  if (ShuffleVectorInst::isDeInterleaveMaskOfFactor(Mask, 2,
+                                                    DeinterleaveIndex) &&
+      count_if(Mask, [](int Idx) { return Idx >= 0; }) > 1)
+    return InstructionCost::getInvalid();
+
+  // Require the two start indices found in the mask to be aligned to
+  // half-vector boundaries, with at least one equal to zero.
+  SmallVector<unsigned, 2> StartIndexes;
+  if (!ShuffleVectorInst::isInterleaveMask(Mask, 2, NumInputElts, StartIndexes))
+    return InstructionCost::getInvalid();
+
+  unsigned HalfNumElts = Mask.size() / 2;
+  if ((StartIndexes[0] != 0 && StartIndexes[1] != 0) ||
+      StartIndexes[0] % HalfNumElts != 0 || StartIndexes[1] % HalfNumElts != 0)
+    return InstructionCost::getInvalid();
+
+  std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(DstTy);
+  std::optional<MVT> CostVT = getZvzipVZIPCostVT(DstLT.second);
+  if (!CostVT)
+    return InstructionCost::getInvalid();
+
+  InstructionCost Cost =
+      DstLT.first * getRISCVInstructionCost(RISCV::VZIP_VV, *CostVT, CostKind);
+
+  // For a shuffle such as
+  //   %r = shufflevector <4 x i32> %v, <4 x i32> poison,
+  //                      <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+  // vzip.vv needs two <2 x i32> inputs. Include the cost of extracting each
+  // upper half; extracting a low half is free.
+  if (NumSrcElts == NumDstElts) {
+    auto *HalfSrcTy = FixedVectorType::getHalfElementsVectorType(
+        cast<FixedVectorType>(SrcTy));
+    for (unsigned Start : StartIndexes) {
+      unsigned ExtractIndex = Start % NumSrcElts;
+      if (ExtractIndex != 0)
+        Cost += getShuffleCost(TTI::SK_ExtractSubvector, HalfSrcTy, SrcTy,
+                               CostKind, {}, ExtractIndex, HalfSrcTy);
+    }
+  }
+
+  return Cost;
 }
 
 InstructionCost RISCVTTIImpl::getShuffleCost(
@@ -809,6 +918,14 @@ InstructionCost RISCVTTIImpl::getShuffleCost(
     case TTI::SK_PermuteSingleSrc: {
       if (Mask.size() >= 2) {
         MVT EltTp = LT.second.getVectorElementType();
+
+        // Try the vzip.vv cost first, otherwise use the widening-interleave
+        // cost below.
+        if (InstructionCost VZipCost =
+                getVZIPCost(Kind, DstTy, SrcTy, Mask, CostKind);
+            VZipCost.isValid())
+          return VZipCost;
+
         // If the size of the element is < ELEN then shuffles of interleaves and
         // deinterleaves of 2 vectors can be lowered into the following
         // sequences
@@ -830,6 +947,16 @@ InstructionCost RISCVTTIImpl::getShuffleCost(
                                                         LT.second, CostKind);
           }
         }
+
+        unsigned Index;
+        if (ST->hasStdExtZvzip() && EltTp.getScalarSizeInBits() != 1 &&
+            ShuffleVectorInst::isDeInterleaveMaskOfFactor(Mask, 2, Index)) {
+          unsigned Opcode = Index == 0 ? RISCV::VUNZIPE_V : RISCV::VUNZIPO_V;
+          if (auto CostVT = getZvzipVUNZIPCostVT(LT.second))
+            return LT.first *
+                   getRISCVInstructionCost(Opcode, *CostVT, CostKind);
+        }
+
         int SubVectorSize;
         if (LT.second.getScalarSizeInBits() != 1 &&
             isRepeatedConcatMask(Mask, SubVectorSize)) {
@@ -878,6 +1005,10 @@ InstructionCost RISCVTTIImpl::getShuffleCost(
           SlideCost.isValid())
         return SlideCost;
 
+      if (InstructionCost VZipCost =
+              getVZIPCost(Kind, DstTy, SrcTy, Mask, CostKind);
+          VZipCost.isValid())
+        return VZipCost;
       // 2 x (vrgather + cost of generating the mask constant) + cost of mask
       // register for the second vrgather. We model this for an unknown
       // (shuffle) mask.
diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
index 2d179479565b5..2a7387e208219 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
@@ -80,6 +80,12 @@ class RISCVTTIImpl final : public BasicTTIImplBase<RISCVTTIImpl> {
   /// the interleaved source EMUL. Return std::nullopt if illegal.
   std::optional<MVT> getZvzipVUNZIPCostVT(MVT InterleavedVT) const;
 
+  /// If a fixed-length shuffle can be lowered to a vzip.vv instruction,
+  /// return its cost.
+  InstructionCost getVZIPCost(TTI::ShuffleKind Kind, VectorType *DstTy,
+                              VectorType *SrcTy, ArrayRef<int> Mask,
+                              TTI::TargetCostKind CostKind) const;
+
 public:
   explicit RISCVTTIImpl(const RISCVTargetMachine *TM, const Function &F)
       : BaseT(TM, F.getDataLayout()), ST(TM->getSubtargetImpl(F)),
diff --git a/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll b/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
index af9b7ee4f6e3f..bfb07dfdc8a04 100644
--- a/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
+++ b/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
@@ -28,20 +28,10 @@ define <8 x i8> @interleave2_v8i8(<4 x i8> %v0, <4 x i8> %v1) {
 }
 
 define <8 x i32> @interleave2_v8i32(<4 x i32> %v0, <4 x i32> %v1) {
-; NOZVZIP-LABEL: 'interleave2_v8i32'
-; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
-;
-; V-LABEL: 'interleave2_v8i32'
-; V-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; V-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
-;
-; ZVE32-LABEL: 'interleave2_v8i32'
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+; CHECK-LABEL: 'interleave2_v8i32'
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
 ;
   %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
   %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
@@ -62,7 +52,7 @@ define <8 x i64> @interleave2_v8i64(<4 x i64> %v0, <4 x i64> %v1) {
 ;
 ; V-LABEL: 'interleave2_v8i64'
 ; V-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %concat = shufflevector <4 x i64> %v0, <4 x i64> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; V-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %res = shufflevector <8 x i64> %concat, <8 x i64> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; V-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %res = shufflevector <8 x i64> %concat, <8 x i64> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
 ; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i64> %res
 ;
 ; ZVE32-LABEL: 'interleave2_v8i64'
@@ -75,14 +65,73 @@ define <8 x i64> @interleave2_v8i64(<4 x i64> %v0, <4 x i64> %v1) {
   ret <8 x i64> %res
 }
 
+define <8 x i32> @interleave2_2src_v8i32(<4 x i32> %a, <4 x i32> %b) {
+; NOZVZIP-LABEL: 'interleave2_2src_v8i32'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_v8i32'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+  %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+  ret <8 x i32> %res
+}
+
+define <8 x i32> @interleave2_2src_swapped_v8i32(<4 x i32> %a, <4 x i32> %b) {
+; NOZVZIP-LABEL: 'interleave2_2src_swapped_v8i32'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_swapped_v8i32'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+  %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
+  ret <8 x i32> %res
+}
+
+define <8 x i32> @interleave2_2src_low_halves_v8i32(<8 x i32> %a, <8 x i32> %b) {
+; NOZVZIP-LABEL: 'interleave2_2src_low_halves_v8i32'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_low_halves_v8i32'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+  %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+  ret <8 x i32> %res
+}
+
+define <8 x i32> @interleave2_2src_upper_half_v8i32(<8 x i32> %a, <8 x i32> %b) {
+; NOZVZIP-LABEL: 'interleave2_2src_upper_half_v8i32'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_upper_half_v8i32'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+  %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
+  ret <8 x i32> %res
+}
+
 ; TODO: getInstructionCost doesn't call getShuffleCost here because the shuffle changes length
 define {<4 x i8>, <4 x i8>} @deinterleave_2(<8 x i8> %v) {
-; CHECK-LABEL: 'deinterleave_2'
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i8>, <4 x i8> } poison, <4 x i8> %v0, 0
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i8>, <4 x i8> } %res0, <4 x i8> %v1, 1
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i8>, <4 x i8> } %res1
+; NOZVZIP-LABEL: 'deinterleave_2'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i8>, <4 x i8> } poison, <4 x i8> %v0, 0
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i8>, <4 x i8> } %res0, <4 x i8> %v1, 1
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i8>, <4 x i8> } %res1
+;
+; ZVZIP-LABEL: 'deinterleave_2'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i8>, <4 x i8> } poison, <4 x i8> %v0, 0
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i8>, <4 x i8> } %res0, <4 x i8> %v1, 1
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i8>, <4 x i8> } %res1
 ;
   %v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
   %v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
@@ -92,12 +141,19 @@ define {<4 x i8>, <4 x i8>} @deinterleave_2(<8 x i8> %v) {
 }
 
 define {<4 x i32>, <4 x i32>} @deinterleave_2_m1_dest(<8 x i32> %v) {
-; CHECK-LABEL: 'deinterleave_2_m1_dest'
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %v0 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %v1 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i32>, <4 x i32> } poison, <4 x i32> %v0, 0
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instructio...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/227677


More information about the llvm-commits mailing list