[llvm] [ARM] Add vtrn/vzip/vuzp shuffle costs. (PR #223611)
David Green via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 15 00:14:02 PDT 2026
https://github.com/davemgreen created https://github.com/llvm/llvm-project/pull/223611
This is an extension to #223310 to add other shuffle cost kinds for the Neon Arm cost model.
Fixes the cost model of #222833
>From 65bb70c163a8765f92dc4aa06403bb7b5d2a04c3 Mon Sep 17 00:00:00 2001
From: David Green <david.green at arm.com>
Date: Mon, 14 Sep 2026 07:10:39 +0100
Subject: [PATCH 1/2] [ARM] Add Neon costs for vrev shuffles
This, like for AArch64 and MVE, allows some of the legal vrev shuffle masks to
be costed as if they are a single instruction, which can help prevent the mid
end from deoptimizing the code.
---
.../lib/Target/ARM/ARMTargetTransformInfo.cpp | 11 ++++++++++
llvm/test/Analysis/CostModel/ARM/shuffle.ll | 20 +++++++++----------
2 files changed, 21 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp b/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
index 7a2642c78e466a..d8ba99a4da0231 100644
--- a/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
+++ b/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
@@ -1308,6 +1308,17 @@ InstructionCost ARMTTIImpl::getShuffleCost(TTI::ShuffleKind Kind,
ISD::VECTOR_SHUFFLE, LT.second))
return LT.first * Entry->Cost;
}
+
+ // Check for other shuffles that are not SK_ kinds but we have native
+ // instructions for, for example REV.
+ if (!Mask.empty()) {
+ std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(SrcTy);
+ if (LT.second.isVector() &&
+ Mask.size() <= LT.second.getVectorNumElements() &&
+ (isVREVMask(Mask, LT.second, 16) || isVREVMask(Mask, LT.second, 32) ||
+ isVREVMask(Mask, LT.second, 64)))
+ return LT.first;
+ }
}
if (ST->hasMVEIntegerOps()) {
if (Kind == TTI::SK_Broadcast) {
diff --git a/llvm/test/Analysis/CostModel/ARM/shuffle.ll b/llvm/test/Analysis/CostModel/ARM/shuffle.ll
index 0fd14566e23def..92bf4907a134f9 100644
--- a/llvm/test/Analysis/CostModel/ARM/shuffle.ll
+++ b/llvm/test/Analysis/CostModel/ARM/shuffle.ll
@@ -342,19 +342,19 @@ define void @vrev2() {
; CHECK-MVE-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; CHECK-NEON-LABEL: 'vrev2'
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %v4i8 = shufflevector <4 x i8> undef, <4 x i8> undef, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %v8i8 = shufflevector <8 x i8> undef, <8 x i8> undef, <8 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %v16i8 = shufflevector <16 x i8> undef, <16 x i8> undef, <16 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6, i32 9, i32 8, i32 11, i32 10, i32 13, i32 12, i32 15, i32 14>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %v4i16 = shufflevector <4 x i16> undef, <4 x i16> undef, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %v8i16 = shufflevector <8 x i16> undef, <8 x i16> undef, <8 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %v4i8 = shufflevector <4 x i8> undef, <4 x i8> undef, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %v8i8 = shufflevector <8 x i8> undef, <8 x i8> undef, <8 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %v16i8 = shufflevector <16 x i8> undef, <16 x i8> undef, <16 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6, i32 9, i32 8, i32 11, i32 10, i32 13, i32 12, i32 15, i32 14>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %v4i16 = shufflevector <4 x i16> undef, <4 x i16> undef, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %v8i16 = shufflevector <8 x i16> undef, <8 x i16> undef, <8 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6>
; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %v16i16 = shufflevector <16 x i16> undef, <16 x i16> undef, <16 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6, i32 9, i32 8, i32 11, i32 10, i32 13, i32 12, i32 15, i32 14>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %v4i32 = shufflevector <4 x i32> undef, <4 x i32> undef, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %v4i32 = shufflevector <4 x i32> undef, <4 x i32> undef, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %v8i32 = shufflevector <8 x i32> undef, <8 x i32> undef, <8 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6>
; CHECK-NEON-NEXT: Cost Model: Found costs of 24 for: %v4i64 = shufflevector <4 x i64> undef, <4 x i64> undef, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
; CHECK-NEON-NEXT: Cost Model: Found costs of 20 for: %v4f16 = shufflevector <4 x half> undef, <4 x half> undef, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
; CHECK-NEON-NEXT: Cost Model: Found costs of 40 for: %v8f16 = shufflevector <8 x half> undef, <8 x half> undef, <8 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6>
; CHECK-NEON-NEXT: Cost Model: Found costs of 80 for: %v16f16 = shufflevector <16 x half> undef, <16 x half> undef, <16 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6, i32 9, i32 8, i32 11, i32 10, i32 13, i32 12, i32 15, i32 14>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 20 for: %v4f32 = shufflevector <4 x float> undef, <4 x float> undef, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %v4f32 = shufflevector <4 x float> undef, <4 x float> undef, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
; CHECK-NEON-NEXT: Cost Model: Found costs of 40 for: %v8f32 = shufflevector <8 x float> undef, <8 x float> undef, <8 x i32> <i32 1, i32 0, i32 3, i32 2, i32 5, i32 4, i32 7, i32 6>
; CHECK-NEON-NEXT: Cost Model: Found costs of 8 for: %v4f64 = shufflevector <4 x double> undef, <4 x double> undef, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
; CHECK-NEON-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
@@ -397,9 +397,9 @@ define void @vrev4() {
; CHECK-MVE-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; CHECK-NEON-LABEL: 'vrev4'
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %v8i8 = shufflevector <8 x i8> undef, <8 x i8> undef, <8 x i32> <i32 3, i32 2, i32 1, i32 0, i32 7, i32 6, i32 5, i32 4>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %v16i8 = shufflevector <16 x i8> undef, <16 x i8> undef, <16 x i32> <i32 3, i32 2, i32 1, i32 0, i32 7, i32 6, i32 5, i32 4, i32 11, i32 10, i32 9, i32 8, i32 15, i32 14, i32 13, i32 12>
-; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %v8i16 = shufflevector <8 x i16> undef, <8 x i16> undef, <8 x i32> <i32 3, i32 2, i32 1, i32 0, i32 7, i32 6, i32 5, i32 4>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %v8i8 = shufflevector <8 x i8> undef, <8 x i8> undef, <8 x i32> <i32 3, i32 2, i32 1, i32 0, i32 7, i32 6, i32 5, i32 4>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %v16i8 = shufflevector <16 x i8> undef, <16 x i8> undef, <16 x i32> <i32 3, i32 2, i32 1, i32 0, i32 7, i32 6, i32 5, i32 4, i32 11, i32 10, i32 9, i32 8, i32 15, i32 14, i32 13, i32 12>
+; CHECK-NEON-NEXT: Cost Model: Found costs of 1 for: %v8i16 = shufflevector <8 x i16> undef, <8 x i16> undef, <8 x i32> <i32 3, i32 2, i32 1, i32 0, i32 7, i32 6, i32 5, i32 4>
; CHECK-NEON-NEXT: Cost Model: Found costs of 96 for: %v16i16 = shufflevector <16 x i16> undef, <16 x i16> undef, <16 x i32> <i32 3, i32 2, i32 1, i32 0, i32 7, i32 6, i32 5, i32 4, i32 11, i32 10, i32 9, i32 8, i32 15, i32 14, i32 13, i32 12>
; CHECK-NEON-NEXT: Cost Model: Found costs of 48 for: %v8i32 = shufflevector <8 x i32> undef, <8 x i32> undef, <8 x i32> <i32 3, i32 2, i32 1, i32 0, i32 7, i32 6, i32 5, i32 4>
; CHECK-NEON-NEXT: Cost Model: Found costs of 40 for: %v8f16 = shufflevector <8 x half> undef, <8 x half> undef, <8 x i32> <i32 3, i32 2, i32 1, i32 0, i32 7, i32 6, i32 5, i32 4>
>From bbc90d71e49597211c99178bcb06607a888aa481 Mon Sep 17 00:00:00 2001
From: David Green <david.green at arm.com>
Date: Tue, 15 Sep 2026 07:52:23 +0100
Subject: [PATCH 2/2] [ARM] Add vtrn/vzip/vuzp shuffle costs.
---
llvm/lib/Target/ARM/ARMISelLowering.cpp | 226 -----------------
.../lib/Target/ARM/ARMTargetTransformInfo.cpp | 9 +-
llvm/lib/Target/ARM/ARMTargetTransformInfo.h | 230 ++++++++++++++++++
3 files changed, 238 insertions(+), 227 deletions(-)
diff --git a/llvm/lib/Target/ARM/ARMISelLowering.cpp b/llvm/lib/Target/ARM/ARMISelLowering.cpp
index 760ae18be94873..9379f2f8dd12e6 100644
--- a/llvm/lib/Target/ARM/ARMISelLowering.cpp
+++ b/llvm/lib/Target/ARM/ARMISelLowering.cpp
@@ -7155,232 +7155,6 @@ static bool isVTBLMask(ArrayRef<int> M, EVT VT) {
return VT == MVT::v8i8 && M.size() == 8;
}
-static unsigned SelectPairHalf(unsigned Elements, ArrayRef<int> Mask,
- unsigned Index) {
- if (Mask.size() == Elements * 2)
- return Index / Elements;
- return Mask[Index] == 0 ? 0 : 1;
-}
-
-// Checks whether the shuffle mask represents a vector transpose (VTRN) by
-// checking that pairs of elements in the shuffle mask represent the same index
-// in each vector, incrementing the expected index by 2 at each step.
-// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 2, 6]
-// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,c,g}
-// v2={e,f,g,h}
-// WhichResult gives the offset for each element in the mask based on which
-// of the two results it belongs to.
-//
-// The transpose can be represented either as:
-// result1 = shufflevector v1, v2, result1_shuffle_mask
-// result2 = shufflevector v1, v2, result2_shuffle_mask
-// where v1/v2 and the shuffle masks have the same number of elements
-// (here WhichResult (see below) indicates which result is being checked)
-//
-// or as:
-// results = shufflevector v1, v2, shuffle_mask
-// where both results are returned in one vector and the shuffle mask has twice
-// as many elements as v1/v2 (here WhichResult will always be 0 if true) here we
-// want to check the low half and high half of the shuffle mask as if it were
-// the other case
-static bool isVTRNMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
- return false;
-
- // If the mask is twice as long as the input vector then we need to check the
- // upper and lower parts of the mask with a matching value for WhichResult
- // FIXME: A mask with only even values will be rejected in case the first
- // element is undefined, e.g. [-1, 4, 2, 6] will be rejected, because only
- // M[0] is used to determine WhichResult
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- for (unsigned j = 0; j < NumElts; j += 2) {
- if ((M[i+j] >= 0 && (unsigned) M[i+j] != j + WhichResult) ||
- (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != j + NumElts + WhichResult))
- return false;
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- return true;
-}
-
-/// isVTRN_v_undef_Mask - Special case of isVTRNMask for canonical form of
-/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
-/// Mask is e.g., <0, 0, 2, 2> instead of <0, 4, 2, 6>.
-static bool isVTRN_v_undef_Mask(ArrayRef<int> M, EVT VT, unsigned &WhichResult){
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
- return false;
-
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- for (unsigned j = 0; j < NumElts; j += 2) {
- if ((M[i+j] >= 0 && (unsigned) M[i+j] != j + WhichResult) ||
- (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != j + WhichResult))
- return false;
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- return true;
-}
-
-// Checks whether the shuffle mask represents a vector unzip (VUZP) by checking
-// that the mask elements are either all even and in steps of size 2 or all odd
-// and in steps of size 2.
-// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 2, 4, 6]
-// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,c,e,g}
-// v2={e,f,g,h}
-// Requires similar checks to that of isVTRNMask with
-// respect the how results are returned.
-static bool isVUZPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if (M.size() != NumElts && M.size() != NumElts*2)
- return false;
-
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- for (unsigned j = 0; j < NumElts; ++j) {
- if (M[i+j] >= 0 && (unsigned) M[i+j] != 2 * j + WhichResult)
- return false;
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
- if (VT.is64BitVector() && EltSz == 32)
- return false;
-
- return true;
-}
-
-/// isVUZP_v_undef_Mask - Special case of isVUZPMask for canonical form of
-/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
-/// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>,
-static bool isVUZP_v_undef_Mask(ArrayRef<int> M, EVT VT, unsigned &WhichResult){
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if (M.size() != NumElts && M.size() != NumElts*2)
- return false;
-
- unsigned Half = NumElts / 2;
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- for (unsigned j = 0; j < NumElts; j += Half) {
- unsigned Idx = WhichResult;
- for (unsigned k = 0; k < Half; ++k) {
- int MIdx = M[i + j + k];
- if (MIdx >= 0 && (unsigned) MIdx != Idx)
- return false;
- Idx += 2;
- }
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
- if (VT.is64BitVector() && EltSz == 32)
- return false;
-
- return true;
-}
-
-// Checks whether the shuffle mask represents a vector zip (VZIP) by checking
-// that pairs of elements of the shufflemask represent the same index in each
-// vector incrementing sequentially through the vectors.
-// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 1, 5]
-// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,b,f}
-// v2={e,f,g,h}
-// Requires similar checks to that of isVTRNMask with respect the how results
-// are returned.
-static bool isVZIPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
- return false;
-
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- unsigned Idx = WhichResult * NumElts / 2;
- for (unsigned j = 0; j < NumElts; j += 2) {
- if ((M[i+j] >= 0 && (unsigned) M[i+j] != Idx) ||
- (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != Idx + NumElts))
- return false;
- Idx += 1;
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
- if (VT.is64BitVector() && EltSz == 32)
- return false;
-
- return true;
-}
-
-/// isVZIP_v_undef_Mask - Special case of isVZIPMask for canonical form of
-/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
-/// Mask is e.g., <0, 0, 1, 1> instead of <0, 4, 1, 5>.
-static bool isVZIP_v_undef_Mask(ArrayRef<int> M, EVT VT, unsigned &WhichResult){
- unsigned EltSz = VT.getScalarSizeInBits();
- if (EltSz == 64)
- return false;
-
- unsigned NumElts = VT.getVectorNumElements();
- if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
- return false;
-
- for (unsigned i = 0; i < M.size(); i += NumElts) {
- WhichResult = SelectPairHalf(NumElts, M, i);
- unsigned Idx = WhichResult * NumElts / 2;
- for (unsigned j = 0; j < NumElts; j += 2) {
- if ((M[i+j] >= 0 && (unsigned) M[i+j] != Idx) ||
- (M[i+j+1] >= 0 && (unsigned) M[i+j+1] != Idx))
- return false;
- Idx += 1;
- }
- }
-
- if (M.size() == NumElts*2)
- WhichResult = 0;
-
- // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
- if (VT.is64BitVector() && EltSz == 32)
- return false;
-
- return true;
-}
-
/// Check if \p ShuffleMask is a NEON two-result shuffle (VZIP, VUZP, VTRN),
/// and return the corresponding ARMISD opcode if it is, or 0 if it isn't.
static unsigned isNEONTwoResultShuffleMask(ArrayRef<int> ShuffleMask, EVT VT,
diff --git a/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp b/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
index d8ba99a4da0231..26d53b97d5d31e 100644
--- a/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
+++ b/llvm/lib/Target/ARM/ARMTargetTransformInfo.cpp
@@ -1313,10 +1313,17 @@ InstructionCost ARMTTIImpl::getShuffleCost(TTI::ShuffleKind Kind,
// instructions for, for example REV.
if (!Mask.empty()) {
std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(SrcTy);
+ unsigned Unused;
if (LT.second.isVector() &&
Mask.size() <= LT.second.getVectorNumElements() &&
(isVREVMask(Mask, LT.second, 16) || isVREVMask(Mask, LT.second, 32) ||
- isVREVMask(Mask, LT.second, 64)))
+ isVREVMask(Mask, LT.second, 64) ||
+ isVTRNMask(Mask, LT.second, Unused) ||
+ isVTRN_v_undef_Mask(Mask, LT.second, Unused) ||
+ isVZIPMask(Mask, LT.second, Unused) ||
+ isVZIP_v_undef_Mask(Mask, LT.second, Unused) ||
+ isVUZPMask(Mask, LT.second, Unused) ||
+ isVUZP_v_undef_Mask(Mask, LT.second, Unused)))
return LT.first;
}
}
diff --git a/llvm/lib/Target/ARM/ARMTargetTransformInfo.h b/llvm/lib/Target/ARM/ARMTargetTransformInfo.h
index 3e6ad7f13b4e70..3ef3f25831a9b2 100644
--- a/llvm/lib/Target/ARM/ARMTargetTransformInfo.h
+++ b/llvm/lib/Target/ARM/ARMTargetTransformInfo.h
@@ -364,6 +364,236 @@ inline bool isVREVMask(ArrayRef<int> M, EVT VT, unsigned BlockSize) {
return true;
}
+inline unsigned SelectPairHalf(unsigned Elements, ArrayRef<int> Mask,
+ unsigned Index) {
+ if (Mask.size() == Elements * 2)
+ return Index / Elements;
+ return Mask[Index] == 0 ? 0 : 1;
+}
+
+// Checks whether the shuffle mask represents a vector transpose (VTRN) by
+// checking that pairs of elements in the shuffle mask represent the same index
+// in each vector, incrementing the expected index by 2 at each step.
+// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 2, 6]
+// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,c,g}
+// v2={e,f,g,h}
+// WhichResult gives the offset for each element in the mask based on which
+// of the two results it belongs to.
+//
+// The transpose can be represented either as:
+// result1 = shufflevector v1, v2, result1_shuffle_mask
+// result2 = shufflevector v1, v2, result2_shuffle_mask
+// where v1/v2 and the shuffle masks have the same number of elements
+// (here WhichResult (see below) indicates which result is being checked)
+//
+// or as:
+// results = shufflevector v1, v2, shuffle_mask
+// where both results are returned in one vector and the shuffle mask has twice
+// as many elements as v1/v2 (here WhichResult will always be 0 if true) here we
+// want to check the low half and high half of the shuffle mask as if it were
+// the other case
+inline bool isVTRNMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
+ return false;
+
+ // If the mask is twice as long as the input vector then we need to check the
+ // upper and lower parts of the mask with a matching value for WhichResult
+ // FIXME: A mask with only even values will be rejected in case the first
+ // element is undefined, e.g. [-1, 4, 2, 6] will be rejected, because only
+ // M[0] is used to determine WhichResult
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ for (unsigned j = 0; j < NumElts; j += 2) {
+ if ((M[i + j] >= 0 && (unsigned)M[i + j] != j + WhichResult) ||
+ (M[i + j + 1] >= 0 &&
+ (unsigned)M[i + j + 1] != j + NumElts + WhichResult))
+ return false;
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ return true;
+}
+
+/// isVTRN_v_undef_Mask - Special case of isVTRNMask for canonical form of
+/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
+/// Mask is e.g., <0, 0, 2, 2> instead of <0, 4, 2, 6>.
+inline bool isVTRN_v_undef_Mask(ArrayRef<int> M, EVT VT,
+ unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
+ return false;
+
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ for (unsigned j = 0; j < NumElts; j += 2) {
+ if ((M[i + j] >= 0 && (unsigned)M[i + j] != j + WhichResult) ||
+ (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != j + WhichResult))
+ return false;
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ return true;
+}
+
+// Checks whether the shuffle mask represents a vector unzip (VUZP) by checking
+// that the mask elements are either all even and in steps of size 2 or all odd
+// and in steps of size 2.
+// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 2, 4, 6]
+// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,c,e,g}
+// v2={e,f,g,h}
+// Requires similar checks to that of isVTRNMask with
+// respect the how results are returned.
+inline bool isVUZPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if (M.size() != NumElts && M.size() != NumElts * 2)
+ return false;
+
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ for (unsigned j = 0; j < NumElts; ++j) {
+ if (M[i + j] >= 0 && (unsigned)M[i + j] != 2 * j + WhichResult)
+ return false;
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
+ if (VT.is64BitVector() && EltSz == 32)
+ return false;
+
+ return true;
+}
+
+/// isVUZP_v_undef_Mask - Special case of isVUZPMask for canonical form of
+/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
+/// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>,
+inline bool isVUZP_v_undef_Mask(ArrayRef<int> M, EVT VT,
+ unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if (M.size() != NumElts && M.size() != NumElts * 2)
+ return false;
+
+ unsigned Half = NumElts / 2;
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ for (unsigned j = 0; j < NumElts; j += Half) {
+ unsigned Idx = WhichResult;
+ for (unsigned k = 0; k < Half; ++k) {
+ int MIdx = M[i + j + k];
+ if (MIdx >= 0 && (unsigned)MIdx != Idx)
+ return false;
+ Idx += 2;
+ }
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
+ if (VT.is64BitVector() && EltSz == 32)
+ return false;
+
+ return true;
+}
+
+// Checks whether the shuffle mask represents a vector zip (VZIP) by checking
+// that pairs of elements of the shufflemask represent the same index in each
+// vector incrementing sequentially through the vectors.
+// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 1, 5]
+// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,b,f}
+// v2={e,f,g,h}
+// Requires similar checks to that of isVTRNMask with respect the how results
+// are returned.
+inline bool isVZIPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
+ return false;
+
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ unsigned Idx = WhichResult * NumElts / 2;
+ for (unsigned j = 0; j < NumElts; j += 2) {
+ if ((M[i + j] >= 0 && (unsigned)M[i + j] != Idx) ||
+ (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != Idx + NumElts))
+ return false;
+ Idx += 1;
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
+ if (VT.is64BitVector() && EltSz == 32)
+ return false;
+
+ return true;
+}
+
+/// isVZIP_v_undef_Mask - Special case of isVZIPMask for canonical form of
+/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
+/// Mask is e.g., <0, 0, 1, 1> instead of <0, 4, 1, 5>.
+inline bool isVZIP_v_undef_Mask(ArrayRef<int> M, EVT VT,
+ unsigned &WhichResult) {
+ unsigned EltSz = VT.getScalarSizeInBits();
+ if (EltSz == 64)
+ return false;
+
+ unsigned NumElts = VT.getVectorNumElements();
+ if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
+ return false;
+
+ for (unsigned i = 0; i < M.size(); i += NumElts) {
+ WhichResult = SelectPairHalf(NumElts, M, i);
+ unsigned Idx = WhichResult * NumElts / 2;
+ for (unsigned j = 0; j < NumElts; j += 2) {
+ if ((M[i + j] >= 0 && (unsigned)M[i + j] != Idx) ||
+ (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != Idx))
+ return false;
+ Idx += 1;
+ }
+ }
+
+ if (M.size() == NumElts * 2)
+ WhichResult = 0;
+
+ // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
+ if (VT.is64BitVector() && EltSz == 32)
+ return false;
+
+ return true;
+}
+
} // end namespace llvm
#endif // LLVM_LIB_TARGET_ARM_ARMTARGETTRANSFORMINFO_H
More information about the llvm-commits
mailing list