[llvm] [RISCV] Model Zvzip costs for fixed-length shuffles (PR #227677)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 05:02:02 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-risc-v
Author: renndong
<details>
<summary>Changes</summary>
Teach the RISC-V TTI to recognize interleave and deinterleave shuffles
that can be implemented with vzip.vv or vunzipe.v/vunzipo.v. Compute
their costs using the operand containing the interleaved data: the
destination of vzip.vv and the source of vunzipe.v/vunzipo.v.
Assisted-By: Trae CLI (GPT-5)
---
Patch is 42.41 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/227677.diff
3 Files Affected:
- (modified) llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp (+140-9)
- (modified) llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h (+6)
- (modified) llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll (+165-72)
``````````diff
diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index 69554ca2b8155..87dab23c37b27 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -744,29 +744,138 @@ InstructionCost RISCVTTIImpl::getSlideCost(FixedVectorType *Tp,
}
std::optional<MVT> RISCVTTIImpl::getZvzipVZIPCostVT(MVT InterleavedVT) const {
- assert(InterleavedVT.isScalableVector() && "Expected a scalable vector type");
if (!InterleavedVT.getVectorElementCount().isKnownEven())
return std::nullopt;
- unsigned EltBits = InterleavedVT.getScalarSizeInBits();
- unsigned MinSize = InterleavedVT.getSizeInBits().getKnownMinValue();
+ MVT CostVT = InterleavedVT;
+ if (InterleavedVT.isFixedLengthVector()) {
+ MVT SourceVT = InterleavedVT.getHalfNumVectorElementsVT();
+ CostVT = TLI->getContainerForFixedLengthVector(SourceVT)
+ .getDoubleNumVectorElementsVT();
+ }
+
+ unsigned EltBits = CostVT.getScalarSizeInBits();
+ unsigned MinSize = CostVT.getSizeInBits().getKnownMinValue();
unsigned LMULOctuple = MinSize / (RISCV::RVVBitsPerBlock / 8);
// Perform the 2 * SEW <= LMUL * min(ELEN, VLEN) check.
- if (EltBits * 16 >
- LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
+ if (EltBits * 16 > LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
return std::nullopt;
- return InterleavedVT;
+ return CostVT;
}
std::optional<MVT> RISCVTTIImpl::getZvzipVUNZIPCostVT(MVT InterleavedVT) const {
- assert(InterleavedVT.isScalableVector() && "Expected a scalable vector type");
if (!InterleavedVT.getVectorElementCount().isKnownEven())
return std::nullopt;
- MVT DeinterleavedVT = InterleavedVT.getHalfNumVectorElementsVT();
+ MVT CostVT = InterleavedVT;
+ // lowerZvzipVUNZIP widens the source container if halving it would produce
+ // an illegal result type. Apply the same rule here so the cost uses the
+ // source LMUL selected by ISel.
+ if (InterleavedVT.isFixedLengthVector()) {
+ CostVT = TLI->getContainerForFixedLengthVector(InterleavedVT);
+ if (CostVT.getVectorMinNumElements() == 1 ||
+ !TLI->isTypeLegal(CostVT.getHalfNumVectorElementsVT()))
+ CostVT = CostVT.getDoubleNumVectorElementsVT();
+ }
+
+ MVT DeinterleavedVT = CostVT.getHalfNumVectorElementsVT();
if (RISCVTargetLowering::getLMUL(DeinterleavedVT) == RISCVVType::LMUL_8)
return std::nullopt;
- return InterleavedVT;
+ return CostVT;
+}
+
+InstructionCost RISCVTTIImpl::getVZIPCost(TTI::ShuffleKind Kind,
+ VectorType *DstTy, VectorType *SrcTy,
+ ArrayRef<int> Mask,
+ TTI::TargetCostKind CostKind) const {
+ assert(isa<FixedVectorType>(SrcTy) && "Expected fixed-length vector types");
+
+ if (!ST->hasStdExtZvzip() || Mask.size() < 2 ||
+ DstTy->getScalarSizeInBits() == 1)
+ return InstructionCost::getInvalid();
+
+ unsigned NumSrcElts = SrcTy->getElementCount().getKnownMinValue();
+ unsigned NumDstElts = DstTy->getElementCount().getKnownMinValue();
+ unsigned NumInputElts;
+
+ if (Kind == TTI::SK_PermuteSingleSrc) {
+ // A single-source vzip shuffle interleaves two halves of the same vector,
+ // so its source and destination must have the same number of elements.
+ if (NumSrcElts != NumDstElts)
+ return InstructionCost::getInvalid();
+ NumInputElts = NumSrcElts;
+ } else if (Kind == TTI::SK_PermuteTwoSrc) {
+ // Accept both forms that can be lowered with vzip.vv:
+ //
+ // 1. The destination has twice as many elements as each source, e.g.
+ // %r = shufflevector <4 x i32> %a, <4 x i32> %b,
+ // <8 x i32> <0, 4, 1, 5, 2, 6, 3, 7>
+ // The two operands are already the half-sized inputs of vzip.vv.
+ //
+ // 2. Each source has the same number of elements as the destination, e.g.
+ // %r = shufflevector <4 x i32> %a, <4 x i32> %b,
+ // <4 x i32> <0, 4, 1, 5>
+ // ISel first extracts a <2 x i32> half from each source and then uses
+ // those halves as the inputs of vzip.vv.
+ if (NumSrcElts != NumDstElts && NumSrcElts * 2 != NumDstElts)
+ return InstructionCost::getInvalid();
+ NumInputElts = NumSrcElts * 2;
+ } else {
+ return InstructionCost::getInvalid();
+ }
+
+ // ISel deliberately excludes the two-element single-source identity case.
+ if (Kind == TTI::SK_PermuteSingleSrc && Mask.size() == 2 &&
+ ShuffleVectorInst::isSingleSourceMask(Mask, Mask.size()))
+ return InstructionCost::getInvalid();
+
+ // A deinterleave shuffle for an interleaved load may have a mask such as
+ // %even = shufflevector <4 x i32> %wide, <4 x i32> poison,
+ // <4 x i32> <i32 0, i32 2, i32 poison, i32 poison>
+ // Because isInterleaveMask ignores poison lanes, this mask may also satisfy
+ // the interleave matcher. Do not cost it as vzip.vv.
+ unsigned DeinterleaveIndex;
+ if (ShuffleVectorInst::isDeInterleaveMaskOfFactor(Mask, 2,
+ DeinterleaveIndex) &&
+ count_if(Mask, [](int Idx) { return Idx >= 0; }) > 1)
+ return InstructionCost::getInvalid();
+
+ // Require the two start indices found in the mask to be aligned to
+ // half-vector boundaries, with at least one equal to zero.
+ SmallVector<unsigned, 2> StartIndexes;
+ if (!ShuffleVectorInst::isInterleaveMask(Mask, 2, NumInputElts, StartIndexes))
+ return InstructionCost::getInvalid();
+
+ unsigned HalfNumElts = Mask.size() / 2;
+ if ((StartIndexes[0] != 0 && StartIndexes[1] != 0) ||
+ StartIndexes[0] % HalfNumElts != 0 || StartIndexes[1] % HalfNumElts != 0)
+ return InstructionCost::getInvalid();
+
+ std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(DstTy);
+ std::optional<MVT> CostVT = getZvzipVZIPCostVT(DstLT.second);
+ if (!CostVT)
+ return InstructionCost::getInvalid();
+
+ InstructionCost Cost =
+ DstLT.first * getRISCVInstructionCost(RISCV::VZIP_VV, *CostVT, CostKind);
+
+ // For a shuffle such as
+ // %r = shufflevector <4 x i32> %v, <4 x i32> poison,
+ // <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+ // vzip.vv needs two <2 x i32> inputs. Include the cost of extracting each
+ // upper half; extracting a low half is free.
+ if (NumSrcElts == NumDstElts) {
+ auto *HalfSrcTy = FixedVectorType::getHalfElementsVectorType(
+ cast<FixedVectorType>(SrcTy));
+ for (unsigned Start : StartIndexes) {
+ unsigned ExtractIndex = Start % NumSrcElts;
+ if (ExtractIndex != 0)
+ Cost += getShuffleCost(TTI::SK_ExtractSubvector, HalfSrcTy, SrcTy,
+ CostKind, {}, ExtractIndex, HalfSrcTy);
+ }
+ }
+
+ return Cost;
}
InstructionCost RISCVTTIImpl::getShuffleCost(
@@ -809,6 +918,14 @@ InstructionCost RISCVTTIImpl::getShuffleCost(
case TTI::SK_PermuteSingleSrc: {
if (Mask.size() >= 2) {
MVT EltTp = LT.second.getVectorElementType();
+
+ // Try the vzip.vv cost first, otherwise use the widening-interleave
+ // cost below.
+ if (InstructionCost VZipCost =
+ getVZIPCost(Kind, DstTy, SrcTy, Mask, CostKind);
+ VZipCost.isValid())
+ return VZipCost;
+
// If the size of the element is < ELEN then shuffles of interleaves and
// deinterleaves of 2 vectors can be lowered into the following
// sequences
@@ -830,6 +947,16 @@ InstructionCost RISCVTTIImpl::getShuffleCost(
LT.second, CostKind);
}
}
+
+ unsigned Index;
+ if (ST->hasStdExtZvzip() && EltTp.getScalarSizeInBits() != 1 &&
+ ShuffleVectorInst::isDeInterleaveMaskOfFactor(Mask, 2, Index)) {
+ unsigned Opcode = Index == 0 ? RISCV::VUNZIPE_V : RISCV::VUNZIPO_V;
+ if (auto CostVT = getZvzipVUNZIPCostVT(LT.second))
+ return LT.first *
+ getRISCVInstructionCost(Opcode, *CostVT, CostKind);
+ }
+
int SubVectorSize;
if (LT.second.getScalarSizeInBits() != 1 &&
isRepeatedConcatMask(Mask, SubVectorSize)) {
@@ -878,6 +1005,10 @@ InstructionCost RISCVTTIImpl::getShuffleCost(
SlideCost.isValid())
return SlideCost;
+ if (InstructionCost VZipCost =
+ getVZIPCost(Kind, DstTy, SrcTy, Mask, CostKind);
+ VZipCost.isValid())
+ return VZipCost;
// 2 x (vrgather + cost of generating the mask constant) + cost of mask
// register for the second vrgather. We model this for an unknown
// (shuffle) mask.
diff --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
index 2d179479565b5..2a7387e208219 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.h
@@ -80,6 +80,12 @@ class RISCVTTIImpl final : public BasicTTIImplBase<RISCVTTIImpl> {
/// the interleaved source EMUL. Return std::nullopt if illegal.
std::optional<MVT> getZvzipVUNZIPCostVT(MVT InterleavedVT) const;
+ /// If a fixed-length shuffle can be lowered to a vzip.vv instruction,
+ /// return its cost.
+ InstructionCost getVZIPCost(TTI::ShuffleKind Kind, VectorType *DstTy,
+ VectorType *SrcTy, ArrayRef<int> Mask,
+ TTI::TargetCostKind CostKind) const;
+
public:
explicit RISCVTTIImpl(const RISCVTargetMachine *TM, const Function &F)
: BaseT(TM, F.getDataLayout()), ST(TM->getSubtargetImpl(F)),
diff --git a/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll b/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
index af9b7ee4f6e3f..bfb07dfdc8a04 100644
--- a/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
+++ b/llvm/test/Analysis/CostModel/RISCV/shuffle-interleave.ll
@@ -28,20 +28,10 @@ define <8 x i8> @interleave2_v8i8(<4 x i8> %v0, <4 x i8> %v1) {
}
define <8 x i32> @interleave2_v8i32(<4 x i32> %v0, <4 x i32> %v1) {
-; NOZVZIP-LABEL: 'interleave2_v8i32'
-; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
-;
-; V-LABEL: 'interleave2_v8i32'
-; V-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; V-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; V-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
-;
-; ZVE32-LABEL: 'interleave2_v8i32'
-; ZVE32-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; ZVE32-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; ZVE32-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+; CHECK-LABEL: 'interleave2_v8i32'
+; CHECK-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
;
%concat = shufflevector <4 x i32> %v0, <4 x i32> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
%res = shufflevector <8 x i32> %concat, <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
@@ -62,7 +52,7 @@ define <8 x i64> @interleave2_v8i64(<4 x i64> %v0, <4 x i64> %v1) {
;
; V-LABEL: 'interleave2_v8i64'
; V-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %concat = shufflevector <4 x i64> %v0, <4 x i64> %v1, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; V-NEXT: Cost Model: Found an estimated cost of 22 for instruction: %res = shufflevector <8 x i64> %concat, <8 x i64> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; V-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %res = shufflevector <8 x i64> %concat, <8 x i64> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
; V-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i64> %res
;
; ZVE32-LABEL: 'interleave2_v8i64'
@@ -75,14 +65,73 @@ define <8 x i64> @interleave2_v8i64(<4 x i64> %v0, <4 x i64> %v1) {
ret <8 x i64> %res
}
+define <8 x i32> @interleave2_2src_v8i32(<4 x i32> %a, <4 x i32> %b) {
+; NOZVZIP-LABEL: 'interleave2_2src_v8i32'
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_v8i32'
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+ %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+ ret <8 x i32> %res
+}
+
+define <8 x i32> @interleave2_2src_swapped_v8i32(<4 x i32> %a, <4 x i32> %b) {
+; NOZVZIP-LABEL: 'interleave2_2src_swapped_v8i32'
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_swapped_v8i32'
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+ %res = shufflevector <4 x i32> %a, <4 x i32> %b, <8 x i32> <i32 4, i32 0, i32 5, i32 1, i32 6, i32 2, i32 7, i32 3>
+ ret <8 x i32> %res
+}
+
+define <8 x i32> @interleave2_2src_low_halves_v8i32(<8 x i32> %a, <8 x i32> %b) {
+; NOZVZIP-LABEL: 'interleave2_2src_low_halves_v8i32'
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_low_halves_v8i32'
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+ %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+ ret <8 x i32> %res
+}
+
+define <8 x i32> @interleave2_2src_upper_half_v8i32(<8 x i32> %a, <8 x i32> %b) {
+; NOZVZIP-LABEL: 'interleave2_2src_upper_half_v8i32'
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 19 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+; ZVZIP-LABEL: 'interleave2_2src_upper_half_v8i32'
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i32> %res
+;
+ %res = shufflevector <8 x i32> %a, <8 x i32> %b, <8 x i32> <i32 0, i32 12, i32 1, i32 13, i32 2, i32 14, i32 3, i32 15>
+ ret <8 x i32> %res
+}
+
; TODO: getInstructionCost doesn't call getShuffleCost here because the shuffle changes length
define {<4 x i8>, <4 x i8>} @deinterleave_2(<8 x i8> %v) {
-; CHECK-LABEL: 'deinterleave_2'
-; CHECK-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
-; CHECK-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
-; CHECK-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i8>, <4 x i8> } poison, <4 x i8> %v0, 0
-; CHECK-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i8>, <4 x i8> } %res0, <4 x i8> %v1, 1
-; CHECK-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i8>, <4 x i8> } %res1
+; NOZVZIP-LABEL: 'deinterleave_2'
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i8>, <4 x i8> } poison, <4 x i8> %v0, 0
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i8>, <4 x i8> } %res0, <4 x i8> %v1, 1
+; NOZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i8>, <4 x i8> } %res1
+;
+; ZVZIP-LABEL: 'deinterleave_2'
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i8>, <4 x i8> } poison, <4 x i8> %v0, 0
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %res1 = insertvalue { <4 x i8>, <4 x i8> } %res0, <4 x i8> %v1, 1
+; ZVZIP-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret { <4 x i8>, <4 x i8> } %res1
;
%v0 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
%v1 = shufflevector <8 x i8> %v, <8 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
@@ -92,12 +141,19 @@ define {<4 x i8>, <4 x i8>} @deinterleave_2(<8 x i8> %v) {
}
define {<4 x i32>, <4 x i32>} @deinterleave_2_m1_dest(<8 x i32> %v) {
-; CHECK-LABEL: 'deinterleave_2_m1_dest'
-; CHECK-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v0 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
-; CHECK-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %v1 = shufflevector <8 x i32> %v, <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
-; CHECK-NEXT: Cost Model: Found an estimated cost of 0 for instruction: %res0 = insertvalue { <4 x i32>, <4 x i32> } poison, <4 x i32> %v0, 0
-; CHECK-NEXT: Cost Model: Found an estimated cost of 0 for instructio...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/227677
More information about the llvm-commits
mailing list