[llvm] [ARM] Add vtrn/vzip/vuzp shuffle costs. (PR #223611)
David Green via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 27 23:52:23 PDT 2026
https://github.com/davemgreen updated https://github.com/llvm/llvm-project/pull/223611
>From 3b7079d88e81fdaeb24f17ac2cfc134d0f448fa0 Mon Sep 17 00:00:00 2001
From: David Green <david.green at arm.com>
Date: Mon, 28 Sep 2026 07:52:06 +0100
Subject: [PATCH] [ARM] Add vtrn/vzip/vuzp shuffle costs.
---
llvm/lib/Target/ARM/ARMISelLowering.cpp | 226 -----------------
.../lib/Target/ARM/ARMTargetTransformInfo.cpp | 9 +-
llvm/lib/Target/ARM/ARMTargetTransformInfo.h | 230 ++++++++++++++++++
llvm/test/Analysis/CostModel/ARM/shuffle.ll | 104 ++++----
4 files changed, 290 insertions(+), 279 deletions(-)
diff --git a/llvm/lib/Target/ARM/ARMISelLowering.cpp b/llvm/lib/Target/ARM/ARMISelLowering.cpp
index 7e5f6134fc88be..5375f170a3c055 100644
--- a/llvm/lib/Target/ARM/ARMISelLowering.cpp
+++ b/llvm/lib/Target/ARM/ARMISelLowering.cpp
@@ -7155,232 +7155,6 @@ static bool isVTBLMask(ArrayRef<int> M, EVT VT) {
return VT == MVT::v8i8 && M.size() == 8;
}
-static unsigned SelectPairHalf(unsigned Elements, ArrayRef<int> Mask,
- unsigned Index) {
- if (Mask.size() == Elements * 2)
- return Index / Elements;
- return Mask[Index] == 0 ? 0 : 1;
-}
-
-// Checks whether the shuffle mask represents a vector transpose (VTRN) by
-// checking that pairs of elements in the shuffle mask represent the same index
-// in each vector, incrementing the expected index by 2 at each step.
-// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 2, 6]
-// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,c,g}
-// v2={e,f,g,h}
-// WhichResult gives the offset for each element in the mask based on which
-// of the two results it belongs to.
-//
-// The transpose can be represented either as:
-// result1 = shufflevector v1, v2, result1_shuffle_mask
-// result2 = shufflevector v1, v2, result2_shuffle_mask
-// where v1/v2 and the shuffle masks have the same number of elements
-// (here WhichResult (see below) indicates which result is being checked)
-//
-// or as:
-// results = shufflevector v1, v2, shuffle_mask
-// where both results are returned in one vector and the shuffle mask has twice
-// as many elements as v1/v2 (here WhichResult will always be 0 if true) here we
-// want to check the low half and high half of the shuffle mask as if it were
-// the other case
-static bool isVTRNMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
- return false;
-
- // If the mask is twice as long as the input vector then we need to check the
- // upper and lower parts of the mask with a matching value for WhichResult
- // FIXME: A mask with only even values will be rejected in case the first
- // element is undefined, e.g. [-1, 4, 2, 6] will be rejected, because only
- // M[0] is used to determine WhichResult
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- for (unsigned j = 0; j < NumElts; j += 2) {
- if ((M[i+j] >= 0 && (unsigned) M[i+j] != j + WhichResult) ||
- (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != j + NumElts + WhichResult))
- return false;
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- return true;
-}
-
-/// isVTRN_v_undef_Mask - Special case of isVTRNMask for canonical form of
-/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
-/// Mask is e.g., <0, 0, 2, 2> instead of <0, 4, 2, 6>.
-static bool isVTRN_v_undef_Mask(ArrayRef<int> M, EVT VT, unsigned &WhichResult){
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
- return false;
-
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- for (unsigned j = 0; j < NumElts; j += 2) {
- if ((M[i+j] >= 0 && (unsigned) M[i+j] != j + WhichResult) ||
- (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != j + WhichResult))
- return false;
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- return true;
-}
-
-// Checks whether the shuffle mask represents a vector unzip (VUZP) by checking
-// that the mask elements are either all even and in steps of size 2 or all odd
-// and in steps of size 2.
-// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 2, 4, 6]
-// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,c,e,g}
-// v2={e,f,g,h}
-// Requires similar checks to that of isVTRNMask with
-// respect the how results are returned.
-static bool isVUZPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if (M.size() != NumElts && M.size() != NumElts*2)
- return false;
-
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- for (unsigned j = 0; j < NumElts; ++j) {
- if (M[i+j] >= 0 && (unsigned) M[i+j] != 2 * j + WhichResult)
- return false;
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
- if (VT.is64BitVector() && EltSz == 32)
- return false;
-
- return true;
-}
-
-/// isVUZP_v_undef_Mask - Special case of isVUZPMask for canonical form of
-/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
-/// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>,
-static bool isVUZP_v_undef_Mask(ArrayRef<int> M, EVT VT, unsigned &WhichResult){
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if (M.size() != NumElts && M.size() != NumElts*2)
- return false;
-
- unsigned Half = NumElts / 2;
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- for (unsigned j = 0; j < NumElts; j += Half) {
- unsigned Idx = WhichResult;
- for (unsigned k = 0; k < Half; ++k) {
- int MIdx = M[i + j + k];
- if (MIdx >= 0 && (unsigned) MIdx != Idx)
- return false;
- Idx += 2;
- }
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
- if (VT.is64BitVector() && EltSz == 32)
- return false;
-
- return true;
-}
-
-// Checks whether the shuffle mask represents a vector zip (VZIP) by checking
-// that pairs of elements of the shufflemask represent the same index in each
-// vector incrementing sequentially through the vectors.
-// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 1, 5]
-// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,b,f}
-// v2={e,f,g,h}
-// Requires similar checks to that of isVTRNMask with respect the how results
-// are returned.
-static bool isVZIPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
- return false;
-
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- unsigned Idx = WhichResult * NumElts / 2;
- for (unsigned j = 0; j < NumElts; j += 2) {
- if ((M[i+j] >= 0 && (unsigned) M[i+j] != Idx) ||
- (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != Idx + NumElts))
- return false;
- Idx += 1;
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
- if (VT.is64BitVector() && EltSz == 32)
- return false;
-
- return true;
-}
-
-/// isVZIP_v_undef_Mask - Special case of isVZIPMask for canonical form of
-/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
-/// Mask is e.g., <0, 0, 1, 1> instead of <0, 4, 1, 5>.
-static bool isVZIP_v_undef_Mask(ArrayRef<int> M, EVT VT, unsigned &WhichResult){
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
- return false;
-
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- unsigned Idx = WhichResult * NumElts / 2;
- for (unsigned j = 0; j < NumElts; j += 2) {
- if ((M[i+j] >= 0 && (unsigned) M[i+j] != Idx) ||
- (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != Idx))
- return false;
- Idx += 1;
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
- if (VT.is64BitVector() && EltSz == 32)
- return false;
-
- return true;
-}
-
/// Check if \p ShuffleMask is a NEON two-result shuffle (VZIP, VUZP, VTRN),
/// and return the corresponding ARMISD opcode if it is, or 0 if it isn't.
static unsigned isNEONTwoResultShuffleMask(ArrayRef<int> ShuffleMask, EVT VT,
diff --git a/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp b/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
index 3ea4ddd10a00a8..1599d9d751b221 100644
--- a/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
+++ b/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
@@ -1311,10 +1311,17 @@ InstructionCost ARMTTIImpl::getShuffleCost(
// instructions for, for example REV.
if (!Mask.empty()) {
std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(SrcTy);
+ unsigned Unused;
if (LT.second.isVector() &&
Mask.size() <= LT.second.getVectorNumElements() &&
(isVREVMask(Mask, LT.second, 16) || isVREVMask(Mask, LT.second, 32) ||
- isVREVMask(Mask, LT.second, 64)))
+ isVREVMask(Mask, LT.second, 64) ||
+ isVTRNMask(Mask, LT.second, Unused) ||
+ isVTRN_v_undef_Mask(Mask, LT.second, Unused) ||
+ isVZIPMask(Mask, LT.second, Unused) ||
+ isVZIP_v_undef_Mask(Mask, LT.second, Unused) ||
+ isVUZPMask(Mask, LT.second, Unused) ||
+ isVUZP_v_undef_Mask(Mask, LT.second, Unused)))
return LT.first;
}
}
diff --git a/llvm/lib/Target/ARM/ARMTargetTransformInfo.h b/llvm/lib/Target/ARM/ARMTargetTransformInfo.h
index 7dacff87fae36d..9c74d1b91a1fe9 100644
--- a/llvm/lib/Target/ARM/ARMTargetTransformInfo.h
+++ b/llvm/lib/Target/ARM/ARMTargetTransformInfo.h
@@ -366,6 +366,236 @@ inline bool isVREVMask(ArrayRef<int> M, EVT VT, unsigned BlockSize) {
return true;
}
+inline unsigned SelectPairHalf(unsigned Elements, ArrayRef<int> Mask,
+ unsigned Index) {
+ if (Mask.size() == Elements * 2)
+ return Index / Elements;
+ return Mask[Index] == 0 ? 0 : 1;
+}
+
+// Checks whether the shuffle mask represents a vector transpose (VTRN) by
+// checking that pairs of elements in the shuffle mask represent the same index
+// in each vector, incrementing the expected index by 2 at each step.
+// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 2, 6]
+// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,c,g}
+// v2={e,f,g,h}
+// WhichResult gives the offset for each element in the mask based on which
+// of the two results it belongs to.
+//
+// The transpose can be represented either as:
+// result1 = shufflevector v1, v2, result1_shuffle_mask
+// result2 = shufflevector v1, v2, result2_shuffle_mask
+// where v1/v2 and the shuffle masks have the same number of elements
+// (here WhichResult (see below) indicates which result is being checked)
+//
+// or as:
+// results = shufflevector v1, v2, shuffle_mask
+// where both results are returned in one vector and the shuffle mask has twice
+// as many elements as v1/v2 (here WhichResult will always be 0 if true) here we
+// want to check the low half and high half of the shuffle mask as if it were
+// the other case
+inline bool isVTRNMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
+ return false;
+
+ // If the mask is twice as long as the input vector then we need to check the
+ // upper and lower parts of the mask with a matching value for WhichResult
+ // FIXME: A mask with only even values will be rejected in case the first
+ // element is undefined, e.g. [-1, 4, 2, 6] will be rejected, because only
+ // M[0] is used to determine WhichResult
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ for (unsigned j = 0; j < NumElts; j += 2) {
+ if ((M[i + j] >= 0 && (unsigned)M[i + j] != j + WhichResult) ||
+ (M[i + j + 1] >= 0 &&
+ (unsigned)M[i + j + 1] != j + NumElts + WhichResult))
+ return false;
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ return true;
+}
+
+/// isVTRN_v_undef_Mask - Special case of isVTRNMask for canonical form of
+/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
+/// Mask is e.g., <0, 0, 2, 2> instead of <0, 4, 2, 6>.
+inline bool isVTRN_v_undef_Mask(ArrayRef<int> M, EVT VT,
+ unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
+ return false;
+
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ for (unsigned j = 0; j < NumElts; j += 2) {
+ if ((M[i + j] >= 0 && (unsigned)M[i + j] != j + WhichResult) ||
+ (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != j + WhichResult))
+ return false;
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ return true;
+}
+
+// Checks whether the shuffle mask represents a vector unzip (VUZP) by checking
+// that the mask elements are either all even and in steps of size 2 or all odd
+// and in steps of size 2.
+// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 2, 4, 6]
+// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,c,e,g}
+// v2={e,f,g,h}
+// Requires similar checks to that of isVTRNMask with
+// respect the how results are returned.
+inline bool isVUZPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if (M.size() != NumElts && M.size() != NumElts * 2)
+ return false;
+
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ for (unsigned j = 0; j < NumElts; ++j) {
+ if (M[i + j] >= 0 && (unsigned)M[i + j] != 2 * j + WhichResult)
+ return false;
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
+ if (VT.is64BitVector() && EltSz == 32)
+ return false;
+
+ return true;
+}
+
+/// isVUZP_v_undef_Mask - Special case of isVUZPMask for canonical form of
+/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
+/// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>,
+inline bool isVUZP_v_undef_Mask(ArrayRef<int> M, EVT VT,
+ unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if (M.size() != NumElts && M.size() != NumElts * 2)
+ return false;
+
+ unsigned Half = NumElts / 2;
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ for (unsigned j = 0; j < NumElts; j += Half) {
+ unsigned Idx = WhichResult;
+ for (unsigned k = 0; k < Half; ++k) {
+ int MIdx = M[i + j + k];
+ if (MIdx >= 0 && (unsigned)MIdx != Idx)
+ return false;
+ Idx += 2;
+ }
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
+ if (VT.is64BitVector() && EltSz == 32)
+ return false;
+
+ return true;
+}
+
+// Checks whether the shuffle mask represents a vector zip (VZIP) by checking
+// that pairs of elements of the shufflemask represent the same index in each
+// vector incrementing sequentially through the vectors.
+// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 1, 5]
+// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,b,f}
+// v2={e,f,g,h}
+// Requires similar checks to that of isVTRNMask with respect the how results
+// are returned.
+inline bool isVZIPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
+ return false;
+
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ unsigned Idx = WhichResult * NumElts / 2;
+ for (unsigned j = 0; j < NumElts; j += 2) {
+ if ((M[i + j] >= 0 && (unsigned)M[i + j] != Idx) ||
+ (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != Idx + NumElts))
+ return false;
+ Idx += 1;
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
+ if (VT.is64BitVector() && EltSz == 32)
+ return false;
+
+ return true;
+}
+
+/// isVZIP_v_undef_Mask - Special case of isVZIPMask for canonical form of
+/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
+/// Mask is e.g., <0, 0, 1, 1> instead of <0, 4, 1, 5>.
+inline bool isVZIP_v_undef_Mask(ArrayRef<int> M, EVT VT,
+ unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
+ return false;
+
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ unsigned Idx = WhichResult * NumElts / 2;
+ for (unsigned j = 0; j < NumElts; j += 2) {
+ if ((M[i + j] >= 0 && (unsigned)M[i + j] != Idx) ||
+ (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != Idx))
+ return false;
+ Idx += 1;
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
+ if (VT.is64BitVector() && EltSz == 32)
+ return false;
+
+ return true;
+}
+
} // end namespace llvm
#endif // LLVM_LIB_TARGET_ARM_ARMTARGETTRANSFORMINFO_H
diff --git a/llvm/test/Analysis/CostModel/ARM/shuffle.ll b/llvm/test/Analysis/CostModel/ARM/shuffle.ll
index c172a0edd0e913..b051def3e0c47b 100644
--- a/llvm/test/Analysis/CostModel/ARM/shuffle.ll
+++ b/llvm/test/Analysis/CostModel/ARM/shuffle.ll
@@ -476,35 +476,35 @@ define void @zip() {
; CHECK-MVE-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; CHECK-NEON-LABEL: 'zip'
-; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %zip1v2i8 = shufflevector <2 x i8> poison, <2 x i8> poison, <2 x i32> <i32 0, i32 2>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %zip2v2i8 = shufflevector <2 x i8> poison, <2 x i8> poison, <2 x i32> <i32 1, i32 3>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %zipv2i8 = shufflevector <2 x i8> poison, <2 x i8> poison, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %zip1v4i8 = shufflevector <4 x i8> poison, <4 x i8> poison, <4 x i32> <i32 0, i32 4, i32 1, i32 5>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %zip2v4i8 = shufflevector <4 x i8> poison, <4 x i8> poison, <4 x i32> <i32 2, i32 6, i32 3, i32 7>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %zipv4i8 = shufflevector <4 x i8> poison, <4 x i8> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %zip1v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %zip2v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %zipv8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <16 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11, i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %zip1v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 0, i32 16, i32 1, i32 17, i32 2, i32 18, i32 3, i32 19, i32 4, i32 20, i32 5, i32 21, i32 6, i32 22, i32 7, i32 23>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %zip2v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 8, i32 24, i32 9, i32 25, i32 10, i32 26, i32 11, i32 27, i32 12, i32 28, i32 13, i32 29, i32 14, i32 30, i32 15, i32 31>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip1v2i8 = shufflevector <2 x i8> poison, <2 x i8> poison, <2 x i32> <i32 0, i32 2>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip2v2i8 = shufflevector <2 x i8> poison, <2 x i8> poison, <2 x i32> <i32 1, i32 3>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zipv2i8 = shufflevector <2 x i8> poison, <2 x i8> poison, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip1v4i8 = shufflevector <4 x i8> poison, <4 x i8> poison, <4 x i32> <i32 0, i32 4, i32 1, i32 5>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip2v4i8 = shufflevector <4 x i8> poison, <4 x i8> poison, <4 x i32> <i32 2, i32 6, i32 3, i32 7>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zipv4i8 = shufflevector <4 x i8> poison, <4 x i8> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip1v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip2v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zipv8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <16 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11, i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip1v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 0, i32 16, i32 1, i32 17, i32 2, i32 18, i32 3, i32 19, i32 4, i32 20, i32 5, i32 21, i32 6, i32 22, i32 7, i32 23>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip2v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 8, i32 24, i32 9, i32 25, i32 10, i32 26, i32 11, i32 27, i32 12, i32 28, i32 13, i32 29, i32 14, i32 30, i32 15, i32 31>
; CHECK-NEON-NEXT: Cost Model: Found costs of 192 for: %zipv16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <32 x i32> <i32 0, i32 16, i32 1, i32 17, i32 2, i32 18, i32 3, i32 19, i32 4, i32 20, i32 5, i32 21, i32 6, i32 22, i32 7, i32 23, i32 8, i32 24, i32 9, i32 25, i32 10, i32 26, i32 11, i32 27, i32 12, i32 28, i32 13, i32 29, i32 14, i32 30, i32 15, i32 31>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %zip1v2i16 = shufflevector <2 x i16> poison, <2 x i16> poison, <2 x i32> <i32 0, i32 2>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %zip2v2i16 = shufflevector <2 x i16> poison, <2 x i16> poison, <2 x i32> <i32 1, i32 3>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %zipv2i16 = shufflevector <2 x i16> poison, <2 x i16> poison, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %zip1v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 0, i32 4, i32 1, i32 5>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %zip2v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 2, i32 6, i32 3, i32 7>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %zipv4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %zip1v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %zip2v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip1v2i16 = shufflevector <2 x i16> poison, <2 x i16> poison, <2 x i32> <i32 0, i32 2>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip2v2i16 = shufflevector <2 x i16> poison, <2 x i16> poison, <2 x i32> <i32 1, i32 3>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zipv2i16 = shufflevector <2 x i16> poison, <2 x i16> poison, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip1v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 0, i32 4, i32 1, i32 5>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip2v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 2, i32 6, i32 3, i32 7>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zipv4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip1v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip2v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %zipv8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <16 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11, i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %zip1v16i16 = shufflevector <16 x i16> poison, <16 x i16> poison, <16 x i32> <i32 0, i32 16, i32 1, i32 17, i32 2, i32 18, i32 3, i32 19, i32 4, i32 20, i32 5, i32 21, i32 6, i32 22, i32 7, i32 23>
; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %zip2v16i16 = shufflevector <16 x i16> poison, <16 x i16> poison, <16 x i32> <i32 8, i32 24, i32 9, i32 25, i32 10, i32 26, i32 11, i32 27, i32 12, i32 28, i32 13, i32 29, i32 14, i32 30, i32 15, i32 31>
; CHECK-NEON-NEXT: Cost Model: Found costs of 192 for: %zipv16i16 = shufflevector <16 x i16> poison, <16 x i16> poison, <32 x i32> <i32 0, i32 16, i32 1, i32 17, i32 2, i32 18, i32 3, i32 19, i32 4, i32 20, i32 5, i32 21, i32 6, i32 22, i32 7, i32 23, i32 8, i32 24, i32 9, i32 25, i32 10, i32 26, i32 11, i32 27, i32 12, i32 28, i32 13, i32 29, i32 14, i32 30, i32 15, i32 31>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %zip1v2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <2 x i32> <i32 0, i32 2>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %zip2v2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <2 x i32> <i32 1, i32 3>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %zipv2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %zip1v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 0, i32 4, i32 1, i32 5>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %zip2v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 2, i32 6, i32 3, i32 7>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip1v2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <2 x i32> <i32 0, i32 2>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip2v2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <2 x i32> <i32 1, i32 3>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zipv2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip1v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 0, i32 4, i32 1, i32 5>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %zip2v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 2, i32 6, i32 3, i32 7>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %zipv4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %zip1v8i32 = shufflevector <8 x i32> poison, <8 x i32> poison, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %zip2v8i32 = shufflevector <8 x i32> poison, <8 x i32> poison, <8 x i32> <i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
@@ -622,26 +622,26 @@ define void @uzp() {
; CHECK-MVE-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; CHECK-NEON-LABEL: 'uzp'
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %uzp1v4i8 = shufflevector <4 x i8> poison, <4 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %uzp2v4i8 = shufflevector <4 x i8> poison, <4 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp1v4i8 = shufflevector <4 x i8> poison, <4 x i8> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp2v4i8 = shufflevector <4 x i8> poison, <4 x i8> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %uzpv4i8 = shufflevector <4 x i8> poison, <4 x i8> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 1, i32 3, i32 5, i32 7>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %uzp1v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %uzp2v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp1v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp2v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %uzpv8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %uzp1v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %uzp2v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp1v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp2v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
; CHECK-NEON-NEXT: Cost Model: Found costs of 192 for: %uzpv16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <32 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30, i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %uzp1v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %uzp2v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp1v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp2v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %uzpv4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 1, i32 3, i32 5, i32 7>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %uzp1v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %uzp2v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp1v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp2v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %uzpv8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %uzp1v16i16 = shufflevector <16 x i16> poison, <16 x i16> poison, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30>
; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %uzp2v16i16 = shufflevector <16 x i16> poison, <16 x i16> poison, <16 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
; CHECK-NEON-NEXT: Cost Model: Found costs of 192 for: %uzpv16i16 = shufflevector <16 x i16> poison, <16 x i16> poison, <32 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30, i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %uzp1v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %uzp2v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp1v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %uzp2v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %uzpv4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 1, i32 3, i32 5, i32 7>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %uzp1v8i32 = shufflevector <8 x i32> poison, <8 x i32> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %uzp2v8i32 = shufflevector <8 x i32> poison, <8 x i32> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
@@ -756,41 +756,41 @@ define void @trn() {
; CHECK-MVE-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; CHECK-NEON-LABEL: 'trn'
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %trn1v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 0, i32 8, i32 2, i32 10, i32 4, i32 12, i32 6, i32 14>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn1v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 0, i32 8, i32 2, i32 10, i32 4, i32 12, i32 6, i32 14>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %trn2v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 8, i32 0, i32 10, i32 2, i32 12, i32 4, i32 14, i32 6>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %trn3v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 1, i32 9, i32 3, i32 11, i32 5, i32 13, i32 7, i32 15>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn3v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 1, i32 9, i32 3, i32 11, i32 5, i32 13, i32 7, i32 15>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %trn4v8i8 = shufflevector <8 x i8> poison, <8 x i8> poison, <8 x i32> <i32 9, i32 1, i32 11, i32 3, i32 13, i32 5, i32 15, i32 7>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %trn1v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 0, i32 16, i32 2, i32 18, i32 4, i32 20, i32 6, i32 22, i32 8, i32 24, i32 10, i32 26, i32 12, i32 28, i32 14, i32 30>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn1v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 0, i32 16, i32 2, i32 18, i32 4, i32 20, i32 6, i32 22, i32 8, i32 24, i32 10, i32 26, i32 12, i32 28, i32 14, i32 30>
; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %trn2v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 16, i32 0, i32 18, i32 2, i32 20, i32 4, i32 22, i32 6, i32 24, i32 8, i32 26, i32 10, i32 28, i32 12, i32 30, i32 14>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %trn3v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 1, i32 17, i32 3, i32 19, i32 5, i32 21, i32 7, i32 23, i32 9, i32 25, i32 11, i32 27, i32 13, i32 29, i32 15, i32 31>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn3v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 1, i32 17, i32 3, i32 19, i32 5, i32 21, i32 7, i32 23, i32 9, i32 25, i32 11, i32 27, i32 13, i32 29, i32 15, i32 31>
; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %trn4v16i8 = shufflevector <16 x i8> poison, <16 x i8> poison, <16 x i32> <i32 17, i32 1, i32 19, i32 3, i32 21, i32 5, i32 23, i32 7, i32 25, i32 9, i32 27, i32 11, i32 29, i32 13, i32 31, i32 15>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %trn1v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 0, i32 4, i32 2, i32 6>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn1v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 0, i32 4, i32 2, i32 6>
; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %trn2v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 4, i32 0, i32 6, i32 2>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %trn3v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 1, i32 5, i32 3, i32 7>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn3v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 1, i32 5, i32 3, i32 7>
; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %trn4v4i16 = shufflevector <4 x i16> poison, <4 x i16> poison, <4 x i32> <i32 5, i32 1, i32 7, i32 3>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %trn1v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 0, i32 8, i32 2, i32 10, i32 4, i32 12, i32 6, i32 14>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn1v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 0, i32 8, i32 2, i32 10, i32 4, i32 12, i32 6, i32 14>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %trn2v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 8, i32 0, i32 10, i32 2, i32 12, i32 4, i32 14, i32 6>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %trn3v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 1, i32 9, i32 3, i32 11, i32 5, i32 13, i32 7, i32 15>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn3v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 1, i32 9, i32 3, i32 11, i32 5, i32 13, i32 7, i32 15>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %trn4v8i16 = shufflevector <8 x i16> poison, <8 x i16> poison, <8 x i32> <i32 9, i32 1, i32 11, i32 3, i32 13, i32 5, i32 15, i32 7>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %trn1v2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <2 x i32> <i32 0, i32 2>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn1v2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <2 x i32> <i32 0, i32 2>
; CHECK-NEON-NEXT: Cost Model: Found costs of 6 for: %trn2v2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <2 x i32> <i32 2, i32 0>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %trn3v2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <2 x i32> <i32 1, i32 3>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn3v2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <2 x i32> <i32 1, i32 3>
; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %trn4v2i32 = shufflevector <2 x i32> poison, <2 x i32> poison, <2 x i32> <i32 3, i32 1>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %trn1v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 0, i32 4, i32 2, i32 6>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn1v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 0, i32 4, i32 2, i32 6>
; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %trn2v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 4, i32 0, i32 6, i32 2>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %trn3v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 1, i32 5, i32 3, i32 7>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn3v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 1, i32 5, i32 3, i32 7>
; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %trn4v4i32 = shufflevector <4 x i32> poison, <4 x i32> poison, <4 x i32> <i32 5, i32 1, i32 7, i32 3>
; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %trn1v2i64 = shufflevector <2 x i64> poison, <2 x i64> poison, <2 x i32> <i32 0, i32 2>
; CHECK-NEON-NEXT: Cost Model: Found costs of 6 for: %trn2v2i64 = shufflevector <2 x i64> poison, <2 x i64> poison, <2 x i32> <i32 2, i32 0>
; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %trn3v2i64 = shufflevector <2 x i64> poison, <2 x i64> poison, <2 x i32> <i32 1, i32 3>
; CHECK-NEON-NEXT: Cost Model: Found costs of 12 for: %trn4v2i64 = shufflevector <2 x i64> poison, <2 x i64> poison, <2 x i32> <i32 3, i32 1>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 10 for: %trn1v2f32 = shufflevector <2 x float> poison, <2 x float> poison, <2 x i32> <i32 0, i32 2>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn1v2f32 = shufflevector <2 x float> poison, <2 x float> poison, <2 x i32> <i32 0, i32 2>
; CHECK-NEON-NEXT: Cost Model: Found costs of 5 for: %trn2v2f32 = shufflevector <2 x float> poison, <2 x float> poison, <2 x i32> <i32 2, i32 0>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 10 for: %trn3v2f32 = shufflevector <2 x float> poison, <2 x float> poison, <2 x i32> <i32 1, i32 3>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn3v2f32 = shufflevector <2 x float> poison, <2 x float> poison, <2 x i32> <i32 1, i32 3>
; CHECK-NEON-NEXT: Cost Model: Found costs of 10 for: %trn4v2f32 = shufflevector <2 x float> poison, <2 x float> poison, <2 x i32> <i32 3, i32 1>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 20 for: %trn1v4f32 = shufflevector <4 x float> poison, <4 x float> poison, <4 x i32> <i32 0, i32 4, i32 2, i32 6>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn1v4f32 = shufflevector <4 x float> poison, <4 x float> poison, <4 x i32> <i32 0, i32 4, i32 2, i32 6>
; CHECK-NEON-NEXT: Cost Model: Found costs of 20 for: %trn2v4f32 = shufflevector <4 x float> poison, <4 x float> poison, <4 x i32> <i32 4, i32 0, i32 6, i32 2>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 20 for: %trn3v4f32 = shufflevector <4 x float> poison, <4 x float> poison, <4 x i32> <i32 1, i32 5, i32 3, i32 7>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %trn3v4f32 = shufflevector <4 x float> poison, <4 x float> poison, <4 x i32> <i32 1, i32 5, i32 3, i32 7>
; CHECK-NEON-NEXT: Cost Model: Found costs of 20 for: %trn4v4f32 = shufflevector <4 x float> poison, <4 x float> poison, <4 x i32> <i32 5, i32 1, i32 7, i32 3>
; CHECK-NEON-NEXT: Cost Model: Found costs of 4 for: %trn1v2f64 = shufflevector <2 x double> poison, <2 x double> poison, <2 x i32> <i32 0, i32 2>
; CHECK-NEON-NEXT: Cost Model: Found costs of 2 for: %trn2v2f64 = shufflevector <2 x double> poison, <2 x double> poison, <2 x i32> <i32 2, i32 0>
More information about the llvm-commits
mailing list