[llvm] [RISCV] Model Zvzip costs for fixed-length shuffles (PR #227677)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 30 05:09:52 PDT 2026


https://github.com/renndong updated https://github.com/llvm/llvm-project/pull/227677

>From 1ff1c574c126769db454ab90dd2388c71a32a6de Mon Sep 17 00:00:00 2001
From: Mingliang Liu <liumingliang.dev at bytedance.com>
Date: Wed, 30 Sep 2026 17:19:03 +0800
Subject: [PATCH 1/3] [RISCV] Precommit the test

---
 .../CostModel/RISCV/shuffle-interleave.ll     | 74 ++++++++++++++++++-
 1 file changed, 72 insertions(+), 2 deletions(-)

diff --git a/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll b/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
index af9b7ee4f6e3f..dbf93a78196e7 100644
--- a/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
+++ b/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
@@ -75,6 +75,42 @@ define <8 x i64> @interleave2_v8i64(<4 x i64> %v0, <4 x i64> %v1) {
   ret <8 x i64> %res
 }
 
+define <8 x i32> @interleave2_2src_v8i32(<4 x i32> %a, <4 x i32> %b) {
+; CHECK-LABEL: 'interleave2_2src_v8i32'
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+  %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+  ret <8 x i32> %res
+}
+
+define <8 x i32> @interleave2_2src_swapped_v8i32(<4 x i32> %a, <4 x i32> %b) {
+; CHECK-LABEL: 'interleave2_2src_swapped_v8i32'
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+  %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
+  ret <8 x i32> %res
+}
+
+define <8 x i32> @interleave2_2src_low_halves_v8i32(<8 x i32> %a, <8 x i32> %b) {
+; CHECK-LABEL: 'interleave2_2src_low_halves_v8i32'
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+  %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+  ret <8 x i32> %res
+}
+
+define <8 x i32> @interleave2_2src_upper_half_v8i32(<8 x i32> %a, <8 x i32> %b) {
+; CHECK-LABEL: 'interleave2_2src_upper_half_v8i32'
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+  %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
+  ret <8 x i32> %res
+}
+
 ; TODO: getInstructionCost doesn't call getShuffleCost here because the shuffle changes length
 define {<4 x i8>, <4 x i8>} @deinterleave_2(<8 x i8> %v) {
 ; CHECK-LABEL: 'deinterleave_2'
@@ -121,9 +157,43 @@ define {<8 x i32>, <8 x i32>} @deinterleave_2_m2_dest(<16 x i32> %v) {
   ret {<8 x i32>, <8 x i32>} %res1
 }
 
+define {<4 x i64>, <4 x i64>} @deinterleave_v8i64(<8 x i64> %v) {
+; RV32-LABEL: 'deinterleave_v8i64'
+; RV32-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %even = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; RV32-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %odd = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; RV32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i64>, <4 x i64> } poison, <4 x i64> %even, 0
+; RV32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i64>, <4 x i64> } %res0, <4 x i64> %odd, 1
+; RV32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i64>, <4 x i64> } %res1
+;
+; RV64-LABEL: 'deinterleave_v8i64'
+; RV64-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %even = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; RV64-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %odd = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; RV64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i64>, <4 x i64> } poison, <4 x i64> %even, 0
+; RV64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i64>, <4 x i64> } %res0, <4 x i64> %odd, 1
+; RV64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i64>, <4 x i64> } %res1
+;
+; V-LABEL: 'deinterleave_v8i64'
+; V-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %even = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; V-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %odd = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i64>, <4 x i64> } poison, <4 x i64> %even, 0
+; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i64>, <4 x i64> } %res0, <4 x i64> %odd, 1
+; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i64>, <4 x i64> } %res1
+;
+; ZVE32-LABEL: 'deinterleave_v8i64'
+; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %even = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %odd = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i64>, <4 x i64> } poison, <4 x i64> %even, 0
+; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i64>, <4 x i64> } %res0, <4 x i64> %odd, 1
+; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i64>, <4 x i64> } %res1
+;
+  %even = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+  %odd = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+  %res0 = insertvalue {<4 x i64>, <4 x i64>} poison, <4 x i64> %even, 0
+  %res1 = insertvalue {<4 x i64>, <4 x i64>} %res0, <4 x i64> %odd, 1
+  ret {<4 x i64>, <4 x i64>} %res1
+}
+
 ; Fixed-length interleave2 intrinsics.
-; TODO: we haven't supported the cost calculation of Zvzip in getShuffleCost
-; yet, so the cost of fixed vector may seem weird.
 
 define <2 x i32> @interleave2_intrinsic_v2i32(<1 x i32> %a, <1 x i32> %b) {
 ; NOZVZIP-LABEL: 'interleave2_intrinsic_v2i32'

>From 004c09aeabf3e6dd7557dec60eef4141f7eb66cc Mon Sep 17 00:00:00 2001
From: Mingliang Liu <liumingliang.dev at bytedance.com>
Date: Wed, 30 Sep 2026 19:14:35 +0800
Subject: [PATCH 2/3] [RISCV] Model Zvzip costs for fixed-length shuffles

Teach the RISC-V TTI to recognize interleave and deinterleave shuffles
that can be implemented with vzip.vv or vunzipe.v/vunzipo.v. Compute
their costs using the operand containing the interleaved data: the
destination of vzip.vv and the source of vunzipe.v/vunzipo.v.

Assisted-By: Trae CLI (GPT-5)
---
 .../Target/RISCV/RISCVTargetTransformInfo.cpp | 149 +++++++++++++-
 .../Target/RISCV/RISCVTargetTransformInfo.h   |   6 +
 .../CostModel/RISCV/shuffle-interleave.ll     | 191 ++++++++++--------
 3 files changed, 253 insertions(+), 93 deletions(-)

diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index 69554ca2b8155..87dab23c37b27 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -744,29 +744,138 @@ InstructionCost RISCVTTIImpl::getSlideCost(FixedVectorType *Tp,
 }
 
 std::optional<MVT> RISCVTTIImpl::getZvzipVZIPCostVT(MVT InterleavedVT) const {
-  assert(InterleavedVT.isScalableVector() && "Expected a scalable vector type");
   if (!InterleavedVT.getVectorElementCount().isKnownEven())
     return std::nullopt;
 
-  unsigned EltBits = InterleavedVT.getScalarSizeInBits();
-  unsigned MinSize = InterleavedVT.getSizeInBits().getKnownMinValue();
+  MVT CostVT = InterleavedVT;
+  if (InterleavedVT.isFixedLengthVector()) {
+    MVT SourceVT = InterleavedVT.getHalfNumVectorElementsVT();
+    CostVT = TLI->getContainerForFixedLengthVector(SourceVT)
+                 .getDoubleNumVectorElementsVT();
+  }
+
+  unsigned EltBits = CostVT.getScalarSizeInBits();
+  unsigned MinSize = CostVT.getSizeInBits().getKnownMinValue();
   unsigned LMULOctuple = MinSize / (RISCV::RVVBitsPerBlock / 8);
   // Perform the 2 * SEW <= LMUL * min(ELEN, VLEN) check.
-  if (EltBits * 16 >
-      LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
+  if (EltBits * 16 > LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
     return std::nullopt;
-  return InterleavedVT;
+  return CostVT;
 }
 
 std::optional<MVT> RISCVTTIImpl::getZvzipVUNZIPCostVT(MVT InterleavedVT) const {
-  assert(InterleavedVT.isScalableVector() && "Expected a scalable vector type");
   if (!InterleavedVT.getVectorElementCount().isKnownEven())
     return std::nullopt;
 
-  MVT DeinterleavedVT = InterleavedVT.getHalfNumVectorElementsVT();
+  MVT CostVT = InterleavedVT;
+  // lowerZvzipVUNZIP widens the source container if halving it would produce
+  // an illegal result type. Apply the same rule here so the cost uses the
+  // source LMUL selected by ISel.
+  if (InterleavedVT.isFixedLengthVector()) {
+    CostVT = TLI->getContainerForFixedLengthVector(InterleavedVT);
+    if (CostVT.getVectorMinNumElements() == 1 ||
+        !TLI->isTypeLegal(CostVT.getHalfNumVectorElementsVT()))
+      CostVT = CostVT.getDoubleNumVectorElementsVT();
+  }
+
+  MVT DeinterleavedVT = CostVT.getHalfNumVectorElementsVT();
   if (RISCVTargetLowering::getLMUL(DeinterleavedVT) == RISCVVType::LMUL_8)
     return std::nullopt;
-  return InterleavedVT;
+  return CostVT;
+}
+
+InstructionCost RISCVTTIImpl::getVZIPCost(TTI::ShuffleKind Kind,
+                                          VectorType *DstTy, VectorType *SrcTy,
+                                          ArrayRef<int> Mask,
+                                          TTI::TargetCostKind CostKind) const {
+  assert(isa<FixedVectorType>(SrcTy) && "Expected fixed-length vector types");
+
+  if (!ST->hasStdExtZvzip() || Mask.size() < 2 ||
+      DstTy->getScalarSizeInBits() == 1)
+    return InstructionCost::getInvalid();
+
+  unsigned NumSrcElts = SrcTy->getElementCount().getKnownMinValue();
+  unsigned NumDstElts = DstTy->getElementCount().getKnownMinValue();
+  unsigned NumInputElts;
+
+  if (Kind == TTI::SK_PermuteSingleSrc) {
+    // A single-source vzip shuffle interleaves two halves of the same vector,
+    // so its source and destination must have the same number of elements.
+    if (NumSrcElts != NumDstElts)
+      return InstructionCost::getInvalid();
+    NumInputElts = NumSrcElts;
+  } else if (Kind == TTI::SK_PermuteTwoSrc) {
+    // Accept both forms that can be lowered with vzip.vv:
+    //
+    // 1. The destination has twice as many elements as each source, e.g.
+    //      %r = shufflevector <4 x i32> %a, <4 x i32> %b,
+    //                         <8 x i32> <0, 4, 1, 5, 2, 6, 3, 7>
+    //    The two operands are already the half-sized inputs of vzip.vv.
+    //
+    // 2. Each source has the same number of elements as the destination, e.g.
+    //      %r = shufflevector <4 x i32> %a, <4 x i32> %b,
+    //                         <4 x i32> <0, 4, 1, 5>
+    //    ISel first extracts a <2 x i32> half from each source and then uses
+    //    those halves as the inputs of vzip.vv.
+    if (NumSrcElts != NumDstElts && NumSrcElts * 2 != NumDstElts)
+      return InstructionCost::getInvalid();
+    NumInputElts = NumSrcElts * 2;
+  } else {
+    return InstructionCost::getInvalid();
+  }
+
+  // ISel deliberately excludes the two-element single-source identity case.
+  if (Kind == TTI::SK_PermuteSingleSrc && Mask.size() == 2 &&
+      ShuffleVectorInst::isSingleSourceMask(Mask, Mask.size()))
+    return InstructionCost::getInvalid();
+
+  // A deinterleave shuffle for an interleaved load may have a mask such as
+  //   %even = shufflevector <4 x i32> %wide, <4 x i32> poison,
+  //                         <4 x i32> <i32 0, i32 2, i32 poison, i32 poison>
+  // Because isInterleaveMask ignores poison lanes, this mask may also satisfy
+  // the interleave matcher. Do not cost it as vzip.vv.
+  unsigned DeinterleaveIndex;
+  if (ShuffleVectorInst::isDeInterleaveMaskOfFactor(Mask, 2,
+                                                    DeinterleaveIndex) &&
+      count_if(Mask, [](int Idx) { return Idx >= 0; }) > 1)
+    return InstructionCost::getInvalid();
+
+  // Require the two start indices found in the mask to be aligned to
+  // half-vector boundaries, with at least one equal to zero.
+  SmallVector<unsigned, 2> StartIndexes;
+  if (!ShuffleVectorInst::isInterleaveMask(Mask, 2, NumInputElts, StartIndexes))
+    return InstructionCost::getInvalid();
+
+  unsigned HalfNumElts = Mask.size() / 2;
+  if ((StartIndexes[0] != 0 && StartIndexes[1] != 0) ||
+      StartIndexes[0] % HalfNumElts != 0 || StartIndexes[1] % HalfNumElts != 0)
+    return InstructionCost::getInvalid();
+
+  std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(DstTy);
+  std::optional<MVT> CostVT = getZvzipVZIPCostVT(DstLT.second);
+  if (!CostVT)
+    return InstructionCost::getInvalid();
+
+  InstructionCost Cost =
+      DstLT.first * getRISCVInstructionCost(RISCV::VZIP_VV, *CostVT, CostKind);
+
+  // For a shuffle such as
+  //   %r = shufflevector <4 x i32> %v, <4 x i32> poison,
+  //                      <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+  // vzip.vv needs two <2 x i32> inputs. Include the cost of extracting each
+  // upper half; extracting a low half is free.
+  if (NumSrcElts == NumDstElts) {
+    auto *HalfSrcTy = FixedVectorType::getHalfElementsVectorType(
+        cast<FixedVectorType>(SrcTy));
+    for (unsigned Start : StartIndexes) {
+      unsigned ExtractIndex = Start % NumSrcElts;
+      if (ExtractIndex != 0)
+        Cost += getShuffleCost(TTI::SK_ExtractSubvector, HalfSrcTy, SrcTy,
+                               CostKind, {}, ExtractIndex, HalfSrcTy);
+    }
+  }
+
+  return Cost;
 }
 
 InstructionCost RISCVTTIImpl::getShuffleCost(
@@ -809,6 +918,14 @@ InstructionCost RISCVTTIImpl::getShuffleCost(
     case TTI::SK_PermuteSingleSrc: {
       if (Mask.size() >= 2) {
         MVT EltTp = LT.second.getVectorElementType();
+
+        // Try the vzip.vv cost first, otherwise use the widening-interleave
+        // cost below.
+        if (InstructionCost VZipCost =
+                getVZIPCost(Kind, DstTy, SrcTy, Mask, CostKind);
+            VZipCost.isValid())
+          return VZipCost;
+
         // If the size of the element is < ELEN then shuffles of interleaves and
         // deinterleaves of 2 vectors can be lowered into the following
         // sequences
@@ -830,6 +947,16 @@ InstructionCost RISCVTTIImpl::getShuffleCost(
                                                         LT.second, CostKind);
           }
         }
+
+        unsigned Index;
+        if (ST->hasStdExtZvzip() && EltTp.getScalarSizeInBits() != 1 &&
+            ShuffleVectorInst::isDeInterleaveMaskOfFactor(Mask, 2, Index)) {
+          unsigned Opcode = Index == 0 ? RISCV::VUNZIPE_V : RISCV::VUNZIPO_V;
+          if (auto CostVT = getZvzipVUNZIPCostVT(LT.second))
+            return LT.first *
+                   getRISCVInstructionCost(Opcode, *CostVT, CostKind);
+        }
+
         int SubVectorSize;
         if (LT.second.getScalarSizeInBits() != 1 &&
             isRepeatedConcatMask(Mask, SubVectorSize)) {
@@ -878,6 +1005,10 @@ InstructionCost RISCVTTIImpl::getShuffleCost(
           SlideCost.isValid())
         return SlideCost;
 
+      if (InstructionCost VZipCost =
+              getVZIPCost(Kind, DstTy, SrcTy, Mask, CostKind);
+          VZipCost.isValid())
+        return VZipCost;
       // 2 x (vrgather + cost of generating the mask constant) + cost of mask
       // register for the second vrgather. We model this for an unknown
       // (shuffle) mask.
diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
index 2d179479565b5..2a7387e208219 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
@@ -80,6 +80,12 @@ class RISCVTTIImpl final : public BasicTTIImplBase<RISCVTTIImpl> {
   /// the interleaved source EMUL. Return std::nullopt if illegal.
   std::optional<MVT> getZvzipVUNZIPCostVT(MVT InterleavedVT) const;
 
+  /// If a fixed-length shuffle can be lowered to a vzip.vv instruction,
+  /// return its cost.
+  InstructionCost getVZIPCost(TTI::ShuffleKind Kind, VectorType *DstTy,
+                              VectorType *SrcTy, ArrayRef<int> Mask,
+                              TTI::TargetCostKind CostKind) const;
+
 public:
   explicit RISCVTTIImpl(const RISCVTargetMachine *TM, const Function &F)
       : BaseT(TM, F.getDataLayout()), ST(TM->getSubtargetImpl(F)),
diff --git a/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll b/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
index dbf93a78196e7..bfb07dfdc8a04 100644
--- a/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
+++ b/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
@@ -28,20 +28,10 @@ define <8 x i8> @interleave2_v8i8(<4 x i8> %v0, <4 x i8> %v1) {
 }
 
 define <8 x i32> @interleave2_v8i32(<4 x i32> %v0, <4 x i32> %v1) {
-; NOZVZIP-LABEL: 'interleave2_v8i32'
-; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
-;
-; V-LABEL: 'interleave2_v8i32'
-; V-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; V-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
-;
-; ZVE32-LABEL: 'interleave2_v8i32'
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+; CHECK-LABEL: 'interleave2_v8i32'
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
 ;
   %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
   %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
@@ -62,7 +52,7 @@ define <8 x i64> @interleave2_v8i64(<4 x i64> %v0, <4 x i64> %v1) {
 ;
 ; V-LABEL: 'interleave2_v8i64'
 ; V-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %concat = shufflevector <4 x i64> %v0, <4 x i64> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; V-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %res = shufflevector <8 x i64> %concat, <8 x i64> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; V-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %res = shufflevector <8 x i64> %concat, <8 x i64> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
 ; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i64> %res
 ;
 ; ZVE32-LABEL: 'interleave2_v8i64'
@@ -76,36 +66,52 @@ define <8 x i64> @interleave2_v8i64(<4 x i64> %v0, <4 x i64> %v1) {
 }
 
 define <8 x i32> @interleave2_2src_v8i32(<4 x i32> %a, <4 x i32> %b) {
-; CHECK-LABEL: 'interleave2_2src_v8i32'
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+; NOZVZIP-LABEL: 'interleave2_2src_v8i32'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_v8i32'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
 ;
   %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
   ret <8 x i32> %res
 }
 
 define <8 x i32> @interleave2_2src_swapped_v8i32(<4 x i32> %a, <4 x i32> %b) {
-; CHECK-LABEL: 'interleave2_2src_swapped_v8i32'
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+; NOZVZIP-LABEL: 'interleave2_2src_swapped_v8i32'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_swapped_v8i32'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
 ;
   %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
   ret <8 x i32> %res
 }
 
 define <8 x i32> @interleave2_2src_low_halves_v8i32(<8 x i32> %a, <8 x i32> %b) {
-; CHECK-LABEL: 'interleave2_2src_low_halves_v8i32'
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+; NOZVZIP-LABEL: 'interleave2_2src_low_halves_v8i32'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_low_halves_v8i32'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
 ;
   %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
   ret <8 x i32> %res
 }
 
 define <8 x i32> @interleave2_2src_upper_half_v8i32(<8 x i32> %a, <8 x i32> %b) {
-; CHECK-LABEL: 'interleave2_2src_upper_half_v8i32'
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+; NOZVZIP-LABEL: 'interleave2_2src_upper_half_v8i32'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_upper_half_v8i32'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
 ;
   %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
   ret <8 x i32> %res
@@ -113,12 +119,19 @@ define <8 x i32> @interleave2_2src_upper_half_v8i32(<8 x i32> %a, <8 x i32> %b)
 
 ; TODO: getInstructionCost doesn't call getShuffleCost here because the shuffle changes length
 define {<4 x i8>, <4 x i8>} @deinterleave_2(<8 x i8> %v) {
-; CHECK-LABEL: 'deinterleave_2'
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i8>, <4 x i8> } poison, <4 x i8> %v0, 0
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i8>, <4 x i8> } %res0, <4 x i8> %v1, 1
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i8>, <4 x i8> } %res1
+; NOZVZIP-LABEL: 'deinterleave_2'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i8>, <4 x i8> } poison, <4 x i8> %v0, 0
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i8>, <4 x i8> } %res0, <4 x i8> %v1, 1
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i8>, <4 x i8> } %res1
+;
+; ZVZIP-LABEL: 'deinterleave_2'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i8>, <4 x i8> } poison, <4 x i8> %v0, 0
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i8>, <4 x i8> } %res0, <4 x i8> %v1, 1
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i8>, <4 x i8> } %res1
 ;
   %v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
   %v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
@@ -128,12 +141,19 @@ define {<4 x i8>, <4 x i8>} @deinterleave_2(<8 x i8> %v) {
 }
 
 define {<4 x i32>, <4 x i32>} @deinterleave_2_m1_dest(<8 x i32> %v) {
-; CHECK-LABEL: 'deinterleave_2_m1_dest'
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %v0 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %v1 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i32>, <4 x i32> } poison, <4 x i32> %v0, 0
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i32>, <4 x i32> } %res0, <4 x i32> %v1, 1
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i32>, <4 x i32> } %res1
+; NOZVZIP-LABEL: 'deinterleave_2_m1_dest'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %v0 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %v1 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i32>, <4 x i32> } poison, <4 x i32> %v0, 0
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i32>, <4 x i32> } %res0, <4 x i32> %v1, 1
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i32>, <4 x i32> } %res1
+;
+; ZVZIP-LABEL: 'deinterleave_2_m1_dest'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %v0 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %v1 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i32>, <4 x i32> } poison, <4 x i32> %v0, 0
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i32>, <4 x i32> } %res0, <4 x i32> %v1, 1
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i32>, <4 x i32> } %res1
 ;
   %v0 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
   %v1 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
@@ -143,12 +163,19 @@ define {<4 x i32>, <4 x i32>} @deinterleave_2_m1_dest(<8 x i32> %v) {
 }
 
 define {<8 x i32>, <8 x i32>} @deinterleave_2_m2_dest(<16 x i32> %v) {
-; CHECK-LABEL: 'deinterleave_2_m2_dest'
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %v0 = shufflevector <16 x i32> %v, <16 x i32> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %v1 = shufflevector <16 x i32> %v, <16 x i32> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <8 x i32>, <8 x i32> } poison, <8 x i32> %v0, 0
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <8 x i32>, <8 x i32> } %res0, <8 x i32> %v1, 1
-; CHECK-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <8 x i32>, <8 x i32> } %res1
+; NOZVZIP-LABEL: 'deinterleave_2_m2_dest'
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %v0 = shufflevector <16 x i32> %v, <16 x i32> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %v1 = shufflevector <16 x i32> %v, <16 x i32> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <8 x i32>, <8 x i32> } poison, <8 x i32> %v0, 0
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <8 x i32>, <8 x i32> } %res0, <8 x i32> %v1, 1
+; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <8 x i32>, <8 x i32> } %res1
+;
+; ZVZIP-LABEL: 'deinterleave_2_m2_dest'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %v0 = shufflevector <16 x i32> %v, <16 x i32> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %v1 = shufflevector <16 x i32> %v, <16 x i32> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <8 x i32>, <8 x i32> } poison, <8 x i32> %v0, 0
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <8 x i32>, <8 x i32> } %res0, <8 x i32> %v1, 1
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <8 x i32>, <8 x i32> } %res1
 ;
   %v0 = shufflevector <16 x i32> %v, <16 x i32> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
   %v1 = shufflevector <16 x i32> %v, <16 x i32> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
@@ -173,8 +200,8 @@ define {<4 x i64>, <4 x i64>} @deinterleave_v8i64(<8 x i64> %v) {
 ; RV64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i64>, <4 x i64> } %res1
 ;
 ; V-LABEL: 'deinterleave_v8i64'
-; V-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %even = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
-; V-NEXT:  Cost Model: Found an estimated cost of 22 for instruction: %odd = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; V-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %even = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; V-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %odd = shufflevector <8 x i64> %v, <8 x i64> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
 ; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i64>, <4 x i64> } poison, <4 x i64> %even, 0
 ; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i64>, <4 x i64> } %res0, <4 x i64> %odd, 1
 ; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i64>, <4 x i64> } %res1
@@ -200,9 +227,13 @@ define <2 x i32> @interleave2_intrinsic_v2i32(<1 x i32> %a, <1 x i32> %b) {
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %res = call <2 x i32> @llvm.vector.interleave2.v2i32(<1 x i32> %a, <1 x i32> %b)
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <2 x i32> %res
 ;
-; ZVZIP-LABEL: 'interleave2_intrinsic_v2i32'
-; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 11 for instruction: %res = call <2 x i32> @llvm.vector.interleave2.v2i32(<1 x i32> %a, <1 x i32> %b)
-; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <2 x i32> %res
+; V-LABEL: 'interleave2_intrinsic_v2i32'
+; V-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %res = call <2 x i32> @llvm.vector.interleave2.v2i32(<1 x i32> %a, <1 x i32> %b)
+; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <2 x i32> %res
+;
+; ZVE32-LABEL: 'interleave2_intrinsic_v2i32'
+; ZVE32-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %res = call <2 x i32> @llvm.vector.interleave2.v2i32(<1 x i32> %a, <1 x i32> %b)
+; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <2 x i32> %res
 ;
   %res = call <2 x i32> @llvm.vector.interleave2.v2i32(<1 x i32> %a, <1 x i32> %b)
   ret <2 x i32> %res
@@ -218,7 +249,7 @@ define <2 x i64> @interleave2_intrinsic_v2i64(<1 x i64> %a, <1 x i64> %b) {
 ; RV64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <2 x i64> %res
 ;
 ; V-LABEL: 'interleave2_intrinsic_v2i64'
-; V-NEXT:  Cost Model: Found an estimated cost of 11 for instruction: %res = call <2 x i64> @llvm.vector.interleave2.v2i64(<1 x i64> %a, <1 x i64> %b)
+; V-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %res = call <2 x i64> @llvm.vector.interleave2.v2i64(<1 x i64> %a, <1 x i64> %b)
 ; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <2 x i64> %res
 ;
 ; ZVE32-LABEL: 'interleave2_intrinsic_v2i64'
@@ -244,7 +275,7 @@ define <4 x i8> @interleave2_intrinsic_v4i8(<2 x i8> %a, <2 x i8> %b) {
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <4 x i8> %res
 ;
 ; ZVZIP-LABEL: 'interleave2_intrinsic_v4i8'
-; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 11 for instruction: %res = call <4 x i8> @llvm.vector.interleave2.v4i8(<2 x i8> %a, <2 x i8> %b)
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %res = call <4 x i8> @llvm.vector.interleave2.v4i8(<2 x i8> %a, <2 x i8> %b)
 ; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <4 x i8> %res
 ;
   %res = call <4 x i8> @llvm.vector.interleave2.v4i8(<2 x i8> %a, <2 x i8> %b)
@@ -256,9 +287,13 @@ define <4 x i32> @interleave2_intrinsic_v4i32(<2 x i32> %a, <2 x i32> %b) {
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 18 for instruction: %res = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %a, <2 x i32> %b)
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <4 x i32> %res
 ;
-; ZVZIP-LABEL: 'interleave2_intrinsic_v4i32'
-; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 11 for instruction: %res = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %a, <2 x i32> %b)
-; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <4 x i32> %res
+; V-LABEL: 'interleave2_intrinsic_v4i32'
+; V-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %res = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %a, <2 x i32> %b)
+; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <4 x i32> %res
+;
+; ZVE32-LABEL: 'interleave2_intrinsic_v4i32'
+; ZVE32-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %res = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %a, <2 x i32> %b)
+; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <4 x i32> %res
 ;
   %res = call <4 x i32> @llvm.vector.interleave2.v4i32(<2 x i32> %a, <2 x i32> %b)
   ret <4 x i32> %res
@@ -274,7 +309,7 @@ define <4 x i64> @interleave2_intrinsic_v4i64(<2 x i64> %a, <2 x i64> %b) {
 ; RV64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <4 x i64> %res
 ;
 ; V-LABEL: 'interleave2_intrinsic_v4i64'
-; V-NEXT:  Cost Model: Found an estimated cost of 11 for instruction: %res = call <4 x i64> @llvm.vector.interleave2.v4i64(<2 x i64> %a, <2 x i64> %b)
+; V-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %res = call <4 x i64> @llvm.vector.interleave2.v4i64(<2 x i64> %a, <2 x i64> %b)
 ; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <4 x i64> %res
 ;
 ; ZVE32-LABEL: 'interleave2_intrinsic_v4i64'
@@ -291,7 +326,7 @@ define <8 x i16> @interleave2_intrinsic_v8i16(<4 x i16> %a, <4 x i16> %b) {
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i16> %res
 ;
 ; ZVZIP-LABEL: 'interleave2_intrinsic_v8i16'
-; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 11 for instruction: %res = call <8 x i16> @llvm.vector.interleave2.v8i16(<4 x i16> %a, <4 x i16> %b)
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 1 for instruction: %res = call <8 x i16> @llvm.vector.interleave2.v8i16(<4 x i16> %a, <4 x i16> %b)
 ; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i16> %res
 ;
   %res = call <8 x i16> @llvm.vector.interleave2.v8i16(<4 x i16> %a, <4 x i16> %b)
@@ -304,7 +339,7 @@ define <8 x i32> @interleave2_intrinsic_v8i32(<4 x i32> %a, <4 x i32> %b) {
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
 ;
 ; ZVZIP-LABEL: 'interleave2_intrinsic_v8i32'
-; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 11 for instruction: %res = call <8 x i32> @llvm.vector.interleave2.v8i32(<4 x i32> %a, <4 x i32> %b)
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 2 for instruction: %res = call <8 x i32> @llvm.vector.interleave2.v8i32(<4 x i32> %a, <4 x i32> %b)
 ; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
 ;
   %res = call <8 x i32> @llvm.vector.interleave2.v8i32(<4 x i32> %a, <4 x i32> %b)
@@ -326,7 +361,7 @@ define <16 x i32> @interleave2_intrinsic_v16i32(<8 x i32> %a, <8 x i32> %b) {
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i32> %res
 ;
 ; ZVZIP-LABEL: 'interleave2_intrinsic_v16i32'
-; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 19 for instruction: %res = call <16 x i32> @llvm.vector.interleave2.v16i32(<8 x i32> %a, <8 x i32> %b)
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = call <16 x i32> @llvm.vector.interleave2.v16i32(<8 x i32> %a, <8 x i32> %b)
 ; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i32> %res
 ;
   %res = call <16 x i32> @llvm.vector.interleave2.v16i32(<8 x i32> %a, <8 x i32> %b)
@@ -339,7 +374,7 @@ define <32 x i32> @interleave2_intrinsic_v32i32(<16 x i32> %a, <16 x i32> %b) {
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <32 x i32> %res
 ;
 ; ZVZIP-LABEL: 'interleave2_intrinsic_v32i32'
-; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 47 for instruction: %res = call <32 x i32> @llvm.vector.interleave2.v32i32(<16 x i32> %a, <16 x i32> %b)
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %res = call <32 x i32> @llvm.vector.interleave2.v32i32(<16 x i32> %a, <16 x i32> %b)
 ; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret <32 x i32> %res
 ;
   %res = call <32 x i32> @llvm.vector.interleave2.v32i32(<16 x i32> %a, <16 x i32> %b)
@@ -584,7 +619,7 @@ define { <2 x i32>, <2 x i32> } @deinterleave2_intrinsic_v4i32(<4 x i32> %v) {
 ; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <2 x i32>, <2 x i32> } %res
 ;
 ; ZVE32-LABEL: 'deinterleave2_intrinsic_v4i32'
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %res = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %v)
+; ZVE32-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %v)
 ; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <2 x i32>, <2 x i32> } %res
 ;
   %res = call { <2 x i32>, <2 x i32> } @llvm.vector.deinterleave2.v4i32(<4 x i32> %v)
@@ -601,7 +636,7 @@ define { <2 x i64>, <2 x i64> } @deinterleave2_intrinsic_v4i64(<4 x i64> %v) {
 ; RV64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <2 x i64>, <2 x i64> } %res
 ;
 ; V-LABEL: 'deinterleave2_intrinsic_v4i64'
-; V-NEXT:  Cost Model: Found an estimated cost of 16 for instruction: %res = call { <2 x i64>, <2 x i64> } @llvm.vector.deinterleave2.v4i64(<4 x i64> %v)
+; V-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = call { <2 x i64>, <2 x i64> } @llvm.vector.deinterleave2.v4i64(<4 x i64> %v)
 ; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <2 x i64>, <2 x i64> } %res
 ;
 ; ZVE32-LABEL: 'deinterleave2_intrinsic_v4i64'
@@ -630,13 +665,9 @@ define { <4 x i32>, <4 x i32> } @deinterleave2_intrinsic_v8i32(<8 x i32> %v) {
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 39 for instruction: %res = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %v)
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i32>, <4 x i32> } %res
 ;
-; V-LABEL: 'deinterleave2_intrinsic_v8i32'
-; V-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %v)
-; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i32>, <4 x i32> } %res
-;
-; ZVE32-LABEL: 'deinterleave2_intrinsic_v8i32'
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 16 for instruction: %res = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %v)
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i32>, <4 x i32> } %res
+; ZVZIP-LABEL: 'deinterleave2_intrinsic_v8i32'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 4 for instruction: %res = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %v)
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i32>, <4 x i32> } %res
 ;
   %res = call { <4 x i32>, <4 x i32> } @llvm.vector.deinterleave2.v8i32(<8 x i32> %v)
   ret { <4 x i32>, <4 x i32> } %res
@@ -652,7 +683,7 @@ define { <4 x i64>, <4 x i64> } @deinterleave2_intrinsic_v8i64(<8 x i64> %v) {
 ; RV64-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i64>, <4 x i64> } %res
 ;
 ; V-LABEL: 'deinterleave2_intrinsic_v8i64'
-; V-NEXT:  Cost Model: Found an estimated cost of 44 for instruction: %res = call { <4 x i64>, <4 x i64> } @llvm.vector.deinterleave2.v8i64(<8 x i64> %v)
+; V-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %res = call { <4 x i64>, <4 x i64> } @llvm.vector.deinterleave2.v8i64(<8 x i64> %v)
 ; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i64>, <4 x i64> } %res
 ;
 ; ZVE32-LABEL: 'deinterleave2_intrinsic_v8i64'
@@ -668,13 +699,9 @@ define { <8 x i32>, <8 x i32> } @deinterleave2_intrinsic_v16i32(<16 x i32> %v) {
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 93 for instruction: %res = call { <8 x i32>, <8 x i32> } @llvm.vector.deinterleave2.v16i32(<16 x i32> %v)
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <8 x i32>, <8 x i32> } %res
 ;
-; V-LABEL: 'deinterleave2_intrinsic_v16i32'
-; V-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %res = call { <8 x i32>, <8 x i32> } @llvm.vector.deinterleave2.v16i32(<16 x i32> %v)
-; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <8 x i32>, <8 x i32> } %res
-;
-; ZVE32-LABEL: 'deinterleave2_intrinsic_v16i32'
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 44 for instruction: %res = call { <8 x i32>, <8 x i32> } @llvm.vector.deinterleave2.v16i32(<16 x i32> %v)
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <8 x i32>, <8 x i32> } %res
+; ZVZIP-LABEL: 'deinterleave2_intrinsic_v16i32'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 8 for instruction: %res = call { <8 x i32>, <8 x i32> } @llvm.vector.deinterleave2.v16i32(<16 x i32> %v)
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <8 x i32>, <8 x i32> } %res
 ;
   %res = call { <8 x i32>, <8 x i32> } @llvm.vector.deinterleave2.v16i32(<16 x i32> %v)
   ret { <8 x i32>, <8 x i32> } %res
@@ -685,13 +712,9 @@ define { <16 x i32>, <16 x i32> } @deinterleave2_intrinsic_v32i32(<32 x i32> %v)
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 189 for instruction: %res = call { <16 x i32>, <16 x i32> } @llvm.vector.deinterleave2.v32i32(<32 x i32> %v)
 ; NOZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <16 x i32>, <16 x i32> } %res
 ;
-; V-LABEL: 'deinterleave2_intrinsic_v32i32'
-; V-NEXT:  Cost Model: Found an estimated cost of 16 for instruction: %res = call { <16 x i32>, <16 x i32> } @llvm.vector.deinterleave2.v32i32(<32 x i32> %v)
-; V-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <16 x i32>, <16 x i32> } %res
-;
-; ZVE32-LABEL: 'deinterleave2_intrinsic_v32i32'
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 148 for instruction: %res = call { <16 x i32>, <16 x i32> } @llvm.vector.deinterleave2.v32i32(<32 x i32> %v)
-; ZVE32-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <16 x i32>, <16 x i32> } %res
+; ZVZIP-LABEL: 'deinterleave2_intrinsic_v32i32'
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 16 for instruction: %res = call { <16 x i32>, <16 x i32> } @llvm.vector.deinterleave2.v32i32(<32 x i32> %v)
+; ZVZIP-NEXT:  Cost Model: Found an estimated cost of 0 for instruction: ret { <16 x i32>, <16 x i32> } %res
 ;
   %res = call { <16 x i32>, <16 x i32> } @llvm.vector.deinterleave2.v32i32(<32 x i32> %v)
   ret { <16 x i32>, <16 x i32> } %res

>From d788e131f500dc0249314c72e934c872118af7e1 Mon Sep 17 00:00:00 2001
From: Mingliang Liu <liumingliang.dev at bytedance.com>
Date: Wed, 30 Sep 2026 20:09:20 +0800
Subject: [PATCH 3/3] [RISCV] Fix code format

---
 llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp | 3 ++-
 1 file changed, 2 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index 87dab23c37b27..a3acf56583f93 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -758,7 +758,8 @@ std::optional<MVT> RISCVTTIImpl::getZvzipVZIPCostVT(MVT InterleavedVT) const {
   unsigned MinSize = CostVT.getSizeInBits().getKnownMinValue();
   unsigned LMULOctuple = MinSize / (RISCV::RVVBitsPerBlock / 8);
   // Perform the 2 * SEW <= LMUL * min(ELEN, VLEN) check.
-  if (EltBits * 16 > LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
+  if (EltBits * 16 >
+      LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
     return std::nullopt;
   return CostVT;
 }



More information about the llvm-commits mailing list