[llvm] [AArch64][SDAG] Enable +sme2p2/+sve2p2 lowering for compressstore (PR #218624)
Jack Styles via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 03:50:57 PDT 2026
https://github.com/Stylie777 updated https://github.com/llvm/llvm-project/pull/218624
>From cb617e2f9f6a33d7138787d374a2da7f0bffe9ba Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 24 Aug 2026 09:44:28 +0100
Subject: [PATCH 01/12] [AArch64][SDAG][NFC] Use CNTP intrinsic for MSTORE
lowering
Currently, when lowering MSTORE the process is to Zero Extend,
VECREDUCE_ADD and then Zero Extend the result for types that are
not i64. This for i8/i16 types does not work as AArch64 cannot
zero extend these types to i64. Instead, use the CNTP intrinsic
to combine the values before storing them.
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 13 ++++---------
1 file changed, 4 insertions(+), 9 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 36a86241fe3b2..c1ca0f789e534 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -33922,16 +33922,11 @@ SDValue AArch64TargetLowering::LowerMSTORE(SDValue Op,
return SDValue();
EVT MaskVT = Store->getMask().getValueType();
- EVT MaskExtVT = getPromotedVTForPredicate(MaskVT);
- EVT MaskReduceVT = MaskExtVT.getScalarType();
SDValue Zero = DAG.getConstant(0, DL, MVT::i64);
-
- SDValue MaskExt =
- DAG.getNode(ISD::ZERO_EXTEND, DL, MaskExtVT, Store->getMask());
- SDValue CntActive =
- DAG.getNode(ISD::VECREDUCE_ADD, DL, MaskReduceVT, MaskExt);
- if (MaskReduceVT != MVT::i64)
- CntActive = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i64, CntActive);
+ SDValue CntActive = DAG.getNode(
+ ISD::INTRINSIC_WO_CHAIN, DL, MVT::i64,
+ DAG.getTargetConstant(Intrinsic::aarch64_sve_cntp, DL, MVT::i64),
+ Store->getMask(), Store->getMask());
SDValue CompressedValue =
DAG.getNode(ISD::VECTOR_COMPRESS, DL, VT, Store->getValue(),
>From 785b0aa4f2e22c887458160186ad84bcae3befa0 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 20 Aug 2026 13:52:13 +0100
Subject: [PATCH 02/12] [AArch64][SDAG] Enable +sme2p2/+sve2p2 lowering for
MSTORE
Following on from #215219 it will now be possible to lower
llvm.masked.compressstore for sve2p2/sme2p2 using compact
for i8/i16/f16/bf16 types.
---
.../Target/AArch64/AArch64ISelLowering.cpp | 6 +-
.../AArch64/AArch64TargetTransformInfo.h | 5 ++
.../CostModel/AArch64/masked_compress_load.ll | 74 +++++++++----------
.../sve-masked-compressstore-sve2p2.ll | 48 ++++++++++--
4 files changed, 88 insertions(+), 45 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index c1ca0f789e534..6a08bf6bc0e38 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2283,6 +2283,11 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
// lowering for non-compressing masked stores.
setOperationAction(ISD::MSTORE, VT, Custom);
}
+ if (Subtarget->hasSVE2p2() || Subtarget->hasSME2p2()) {
+ // With +sve2p2/+sme2p2 the full range of vector types are supported.
+ for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
+ setOperationAction(ISD::MSTORE, VT, Custom);
+ }
// Histcnt is SVE2 only
if (Subtarget->hasSVE2()) {
@@ -33927,7 +33932,6 @@ SDValue AArch64TargetLowering::LowerMSTORE(SDValue Op,
ISD::INTRINSIC_WO_CHAIN, DL, MVT::i64,
DAG.getTargetConstant(Intrinsic::aarch64_sve_cntp, DL, MVT::i64),
Store->getMask(), Store->getMask());
-
SDValue CompressedValue =
DAG.getNode(ISD::VECTOR_COMPRESS, DL, VT, Store->getValue(),
Store->getMask(), DAG.getPOISON(VT));
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index 19de90427cb74..64337ff4d5db7 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -334,6 +334,11 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
}
bool isElementTypeLegalForCompressStore(Type *Ty) const {
+ if ((ST->hasSVE2p2() || ST->hasSME2p2()) &&
+ ((Ty->isIntegerTy(8) || Ty->isIntegerTy(16)) ||
+ ((Ty->isHalfTy() || Ty->isBFloatTy()) &&
+ Ty->getScalarSizeInBits() == 16)))
+ return true;
return Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
Ty->isIntegerTy(64);
}
diff --git a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
index cb59f1016a4d1..7cc880ce5b66d 100644
--- a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
@@ -95,21 +95,21 @@ define void @fixed() {
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i8.p0(<2 x i8> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i8.p0(<4 x i8> poison, ptr poison, <4 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8i8.p0(<8 x i8> poison, ptr poison, <8 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v16i8.p0(<16 x i8> poison, ptr poison, <16 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v16i8.p0(<16 x i8> poison, ptr poison, <16 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i16.p0(<2 x i16> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i16.p0(<4 x i16> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8i16.p0(<8 x i16> poison, ptr poison, <8 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8i16.p0(<8 x i16> poison, ptr poison, <8 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i32.p0(<2 x i32> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v4i32.p0(<4 x i32> poison, ptr poison, <4 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v2i64.p0(<2 x i64> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f16.p0(<2 x half> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4f16.p0(<4 x half> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8f16.p0(<8 x half> poison, ptr poison, <8 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8f16.p0(<8 x half> poison, ptr poison, <8 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f32.p0(<2 x float> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v4f32.p0(<4 x float> poison, ptr poison, <4 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v2f64.p0(<2 x double> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v4i64.p0(<4 x i64> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v32f16.p0(<32 x half> poison, ptr poison, <32 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.v32f16.p0(<32 x half> poison, ptr poison, <32 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
entry:
@@ -142,25 +142,25 @@ entry:
define void @scalable() {
; SVE2p2-SME2p2-NON-STREAMING-LABEL: 'scalable'
-; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv1i64.p0(<vscale x 1 x i64> poison, ptr poison, <vscale x 1 x i1> poison)
; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; SVE2p2-SME2p2-STREAMING-LABEL: 'scalable'
@@ -186,25 +186,25 @@ define void @scalable() {
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; SVE2p2-SME2p2-LABEL: 'scalable'
-; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
-; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
+; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv1i64.p0(<vscale x 1 x i64> poison, ptr poison, <vscale x 1 x i1> poison)
; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
+; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
; SVE2p2-SME2p2-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; SVE-ONLY-LABEL: 'scalable'
@@ -230,25 +230,25 @@ define void @scalable() {
; SVE-ONLY-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; SVE2p2-SME2p2-SVE256-LABEL: 'scalable'
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv1i64.p0(<vscale x 1 x i64> poison, ptr poison, <vscale x 1 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
; SVE2p2-SME2p2-SVE256-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
entry:
diff --git a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
index 92ecc3c83e2c5..5655d53b5ddaa 100644
--- a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
+++ b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
@@ -1,17 +1,51 @@
-; RUN: llc -mtriple=aarch64 -mattr=+sve2p2 < %s
-
-;; These masked.compressstore operations could be natively supported with +sve2p2
-;; (or by promoting to 32/64 bit elements + a truncstore), but currently are not
-;; supported.
-
-; XFAIL: *
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=aarch64 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BASE --allow-unused-prefixes
+; RUN: llc -mtriple=aarch64 -aarch64-sve-vector-bits-min=256 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-VL256 --allow-unused-prefixes
define void @test_compressstore_nxv8i16(ptr %p, <vscale x 8 x i16> %vec, <vscale x 8 x i1> %mask) {
+; CHECK-LABEL: test_compressstore_nxv8i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: cntp x8, p0, p0.h
+; CHECK-NEXT: compact z0.h, p0, z0.h
+; CHECK-NEXT: whilelo p1.h, xzr, x8
+; CHECK-NEXT: st1h { z0.h }, p1, [x0]
+; CHECK-NEXT: ret
tail call void @llvm.masked.compressstore.nxv8i16(<vscale x 8 x i16> %vec, ptr align 2 %p, <vscale x 8 x i1> %mask)
ret void
}
define void @test_compressstore_nxv16i8(ptr %p, <vscale x 16 x i8> %vec, <vscale x 16 x i1> %mask) {
+; CHECK-LABEL: test_compressstore_nxv16i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: cntp x8, p0, p0.b
+; CHECK-NEXT: compact z0.b, p0, z0.b
+; CHECK-NEXT: whilelo p1.b, xzr, x8
+; CHECK-NEXT: st1b { z0.b }, p1, [x0]
+; CHECK-NEXT: ret
tail call void @llvm.masked.compressstore.nxv16i8(<vscale x 16 x i8> %vec, ptr align 1 %p, <vscale x 16 x i1> %mask)
ret void
}
+
+define void @test_compressstore_nxv8f16(ptr %p, <vscale x 8 x half> %vec, <vscale x 8 x i1> %mask) {
+; CHECK-LABEL: test_compressstore_nxv8f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: cntp x8, p0, p0.h
+; CHECK-NEXT: compact z0.h, p0, z0.h
+; CHECK-NEXT: whilelo p1.h, xzr, x8
+; CHECK-NEXT: st1h { z0.h }, p1, [x0]
+; CHECK-NEXT: ret
+ tail call void @llvm.masked.compressstore.nxv8f16(<vscale x 8 x half> %vec, ptr align 1 %p, <vscale x 8 x i1> %mask)
+ ret void
+}
+
+define void @test_compressstore_nxv8bf16(ptr %p, <vscale x 8 x bfloat> %vec, <vscale x 8 x i1> %mask) {
+; CHECK-LABEL: test_compressstore_nxv8bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: cntp x8, p0, p0.h
+; CHECK-NEXT: compact z0.h, p0, z0.h
+; CHECK-NEXT: whilelo p1.h, xzr, x8
+; CHECK-NEXT: st1h { z0.h }, p1, [x0]
+; CHECK-NEXT: ret
+ tail call void @llvm.masked.compressstore.nxv8bf16(<vscale x 8 x bfloat> %vec, ptr align 1 %p, <vscale x 8 x i1> %mask)
+ ret void
+}
>From f8a722e61b99837d18fda72e3460662e8d5f8292 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 24 Aug 2026 10:03:30 +0100
Subject: [PATCH 03/12] Remove whitespace change
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 1 +
1 file changed, 1 insertion(+)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 6a08bf6bc0e38..07b28395c0dfe 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -33932,6 +33932,7 @@ SDValue AArch64TargetLowering::LowerMSTORE(SDValue Op,
ISD::INTRINSIC_WO_CHAIN, DL, MVT::i64,
DAG.getTargetConstant(Intrinsic::aarch64_sve_cntp, DL, MVT::i64),
Store->getMask(), Store->getMask());
+
SDValue CompressedValue =
DAG.getNode(ISD::VECTOR_COMPRESS, DL, VT, Store->getValue(),
Store->getMask(), DAG.getPOISON(VT));
>From 8e09f73faf8315ede14485f2797935482a9a3452 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 24 Aug 2026 10:48:06 +0100
Subject: [PATCH 04/12] Enable Streaming SME2p2
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 11 ++++++-----
.../AArch64/sve-masked-compressstore-sve2p2.ll | 3 +++
2 files changed, 9 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 07b28395c0dfe..639d5b079f734 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2283,11 +2283,6 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
// lowering for non-compressing masked stores.
setOperationAction(ISD::MSTORE, VT, Custom);
}
- if (Subtarget->hasSVE2p2() || Subtarget->hasSME2p2()) {
- // With +sve2p2/+sme2p2 the full range of vector types are supported.
- for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
- setOperationAction(ISD::MSTORE, VT, Custom);
- }
// Histcnt is SVE2 only
if (Subtarget->hasSVE2()) {
@@ -2308,6 +2303,12 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
}
}
+ if ((Subtarget->isSVEAvailable() && Subtarget->hasSVE2p2()) || (Subtarget->isSVEorStreamingSVEAvailable() && Subtarget->hasSME2p2())) {
+ // With +sve2p2/+sme2p2 the full range of vector types are supported.
+ for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
+ setOperationAction(ISD::MSTORE, VT, Custom);
+ }
+
if (Subtarget->hasMOPS() && Subtarget->hasMTE()) {
// Only required for llvm.aarch64.mops.memset.tag
setOperationAction(ISD::INTRINSIC_W_CHAIN, MVT::i8, Custom);
diff --git a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
index 5655d53b5ddaa..4fca173a712ba 100644
--- a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
+++ b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
@@ -1,6 +1,9 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
; RUN: llc -mtriple=aarch64 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BASE --allow-unused-prefixes
; RUN: llc -mtriple=aarch64 -aarch64-sve-vector-bits-min=256 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-VL256 --allow-unused-prefixes
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve,+sme2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SVE2p2 --allow-unused-prefixes
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2 -force-streaming < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SME2p2-STREAMING --allow-unused-prefixes
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2,+sve2p2 -force-streaming-compatible < %s | FileCheck %s --check-prefixes=CHECK,CHECK-STREAMING-COMPAT --allow-unused-prefixes
define void @test_compressstore_nxv8i16(ptr %p, <vscale x 8 x i16> %vec, <vscale x 8 x i1> %mask) {
; CHECK-LABEL: test_compressstore_nxv8i16:
>From 3865d30d79af0986c28e23b2f713e22d70ebd905 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 24 Aug 2026 10:50:40 +0100
Subject: [PATCH 05/12] format
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 3 ++-
1 file changed, 2 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 639d5b079f734..a0fb5365cd843 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2303,7 +2303,8 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
}
}
- if ((Subtarget->isSVEAvailable() && Subtarget->hasSVE2p2()) || (Subtarget->isSVEorStreamingSVEAvailable() && Subtarget->hasSME2p2())) {
+ if ((Subtarget->isSVEAvailable() && Subtarget->hasSVE2p2()) ||
+ (Subtarget->isSVEorStreamingSVEAvailable() && Subtarget->hasSME2p2())) {
// With +sve2p2/+sme2p2 the full range of vector types are supported.
for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
setOperationAction(ISD::MSTORE, VT, Custom);
>From 929e6c0139c46fea7227a2185de5a0ef4987ba73 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 26 Aug 2026 14:38:12 +0100
Subject: [PATCH 06/12] Respond to review comments
---
.../Target/AArch64/AArch64ISelLowering.cpp | 11 +-
.../AArch64/AArch64TargetTransformInfo.h | 8 +-
.../sve-masked-compressstore-sve2p2.ll | 289 +++++++++++++++++-
3 files changed, 290 insertions(+), 18 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 555a3582c37ee..cdbf7fed84e10 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2227,8 +2227,10 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
if (Subtarget->hasSVE2p2() || Subtarget->hasSME2p2()) {
// With +sve2p2/+sme2p2 the full range of vector types are supported.
- for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
+ for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16}) {
+ setOperationAction(ISD::MSTORE, VT, Custom);
setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
+ }
for (auto VT : {MVT::v8i8, MVT::v16i8, MVT::v4i16, MVT::v8i16, MVT::v4f16,
MVT::v8f16, MVT::v4bf16, MVT::v8bf16})
@@ -2303,13 +2305,6 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
}
}
- if ((Subtarget->isSVEAvailable() && Subtarget->hasSVE2p2()) ||
- (Subtarget->isSVEorStreamingSVEAvailable() && Subtarget->hasSME2p2())) {
- // With +sve2p2/+sme2p2 the full range of vector types are supported.
- for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
- setOperationAction(ISD::MSTORE, VT, Custom);
- }
-
if (Subtarget->hasMOPS() && Subtarget->hasMTE()) {
// Only required for llvm.aarch64.mops.memset.tag
setOperationAction(ISD::INTRINSIC_W_CHAIN, MVT::i8, Custom);
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index a6cb5bdebcfab..5c7b2703b4257 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -334,10 +334,8 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
}
bool isElementTypeLegalForCompressStore(Type *Ty) const {
- if ((ST->hasSVE2p2() || ST->hasSME2p2()) &&
- ((Ty->isIntegerTy(8) || Ty->isIntegerTy(16)) ||
- ((Ty->isHalfTy() || Ty->isBFloatTy()) &&
- Ty->getScalarSizeInBits() == 16)))
+ if ((ST->hasSVE2p2() || (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())) &&
+ (Ty->isIntegerTy(8) || Ty->isIntegerTy(16) || Ty->getScalarSizeInBits() == 16))
return true;
return Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
Ty->isIntegerTy(64);
@@ -345,7 +343,7 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
bool isLegalMaskedCompressStore(Type *DataType,
Align Alignment) const override {
- if (!ST->isSVEAvailable())
+ if (!ST->isSVEorStreamingSVEAvailable())
return false;
if (isa<FixedVectorType>(DataType) &&
diff --git a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
index 4fca173a712ba..42cef84285a0f 100644
--- a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
+++ b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
@@ -1,9 +1,9 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -mtriple=aarch64 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BASE --allow-unused-prefixes
-; RUN: llc -mtriple=aarch64 -aarch64-sve-vector-bits-min=256 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-VL256 --allow-unused-prefixes
-; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve,+sme2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SVE2p2 --allow-unused-prefixes
-; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2 -force-streaming < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SME2p2-STREAMING --allow-unused-prefixes
-; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2,+sve2p2 -force-streaming-compatible < %s | FileCheck %s --check-prefixes=CHECK,CHECK-STREAMING-COMPAT --allow-unused-prefixes
+; RUN: llc -mtriple=aarch64 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BASE
+; RUN: llc -mtriple=aarch64 -aarch64-sve-vector-bits-min=256 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-VL256
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve,+sme2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SVE2p2
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2 -force-streaming < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SME2p2-STREAMING
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2,+sve2p2 -force-streaming-compatible < %s | FileCheck %s --check-prefixes=CHECK,CHECK-STREAMING-COMPAT
define void @test_compressstore_nxv8i16(ptr %p, <vscale x 8 x i16> %vec, <vscale x 8 x i1> %mask) {
; CHECK-LABEL: test_compressstore_nxv8i16:
@@ -52,3 +52,282 @@ define void @test_compressstore_nxv8bf16(ptr %p, <vscale x 8 x bfloat> %vec, <vs
tail call void @llvm.masked.compressstore.nxv8bf16(<vscale x 8 x bfloat> %vec, ptr align 1 %p, <vscale x 8 x i1> %mask)
ret void
}
+
+define void @test_compressstore_v8i16(ptr %p, <8 x i16> %vec, <8 x i1> %mask) {
+; CHECK-BASE-LABEL: test_compressstore_v8i16:
+; CHECK-BASE: // %bb.0:
+; CHECK-BASE-NEXT: ushll v1.8h, v1.8b, #0
+; CHECK-BASE-NEXT: ptrue p0.h, vl8
+; CHECK-BASE-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT: shl v1.8h, v1.8h, #15
+; CHECK-BASE-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-BASE-NEXT: cntp x8, p1, p1.h
+; CHECK-BASE-NEXT: compact z0.h, p1, z0.h
+; CHECK-BASE-NEXT: whilelo p0.h, xzr, x8
+; CHECK-BASE-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-BASE-NEXT: ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v8i16:
+; CHECK-VL256: // %bb.0:
+; CHECK-VL256-NEXT: ushll v1.8h, v1.8b, #0
+; CHECK-VL256-NEXT: ptrue p0.h, vl8
+; CHECK-VL256-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT: shl v1.8h, v1.8h, #15
+; CHECK-VL256-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-VL256-NEXT: cntp x8, p1, p1.h
+; CHECK-VL256-NEXT: compact z0.h, p1, z0.h
+; CHECK-VL256-NEXT: whilelo p0.h, xzr, x8
+; CHECK-VL256-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-VL256-NEXT: ret
+;
+; CHECK-SVE2p2-LABEL: test_compressstore_v8i16:
+; CHECK-SVE2p2: // %bb.0:
+; CHECK-SVE2p2-NEXT: ushll v1.8h, v1.8b, #0
+; CHECK-SVE2p2-NEXT: ptrue p0.h, vl8
+; CHECK-SVE2p2-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-SVE2p2-NEXT: shl v1.8h, v1.8h, #15
+; CHECK-SVE2p2-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SVE2p2-NEXT: cntp x8, p1, p1.h
+; CHECK-SVE2p2-NEXT: compact z0.h, p1, z0.h
+; CHECK-SVE2p2-NEXT: whilelo p0.h, xzr, x8
+; CHECK-SVE2p2-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-SVE2p2-NEXT: ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v8i16:
+; CHECK-SME2p2-STREAMING: // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT: uunpklo z1.h, z1.b
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p0.h, vl8
+; CHECK-SME2p2-STREAMING-NEXT: lsl z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT: asr z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SME2p2-STREAMING-NEXT: cntp x8, p1, p1.h
+; CHECK-SME2p2-STREAMING-NEXT: compact z0.h, p1, z0.h
+; CHECK-SME2p2-STREAMING-NEXT: whilelo p0.h, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT: ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v8i16:
+; CHECK-STREAMING-COMPAT: // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p0.h, vl8
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT: uunpklo z1.h, z1.b
+; CHECK-STREAMING-COMPAT-NEXT: lsl z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT: asr z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-STREAMING-COMPAT-NEXT: cntp x8, p1, p1.h
+; CHECK-STREAMING-COMPAT-NEXT: compact z0.h, p1, z0.h
+; CHECK-STREAMING-COMPAT-NEXT: whilelo p0.h, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT: ret
+ tail call void @llvm.masked.compressstore.v8i16(<8 x i16> %vec, ptr align 2 %p, <8 x i1> %mask)
+ ret void
+}
+
+define void @test_compressstore_v16i8(ptr %p, <16 x i8> %vec, <16 x i1> %mask) {
+; CHECK-BASE-LABEL: test_compressstore_v16i8:
+; CHECK-BASE: // %bb.0:
+; CHECK-BASE-NEXT: shl v1.16b, v1.16b, #7
+; CHECK-BASE-NEXT: ptrue p0.b, vl16
+; CHECK-BASE-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT: cmpne p1.b, p0/z, z1.b, #0
+; CHECK-BASE-NEXT: cntp x8, p1, p1.b
+; CHECK-BASE-NEXT: compact z0.b, p1, z0.b
+; CHECK-BASE-NEXT: whilelo p0.b, xzr, x8
+; CHECK-BASE-NEXT: st1b { z0.b }, p0, [x0]
+; CHECK-BASE-NEXT: ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v16i8:
+; CHECK-VL256: // %bb.0:
+; CHECK-VL256-NEXT: shl v1.16b, v1.16b, #7
+; CHECK-VL256-NEXT: ptrue p0.b, vl16
+; CHECK-VL256-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT: cmpne p1.b, p0/z, z1.b, #0
+; CHECK-VL256-NEXT: cntp x8, p1, p1.b
+; CHECK-VL256-NEXT: compact z0.b, p1, z0.b
+; CHECK-VL256-NEXT: whilelo p0.b, xzr, x8
+; CHECK-VL256-NEXT: st1b { z0.b }, p0, [x0]
+; CHECK-VL256-NEXT: ret
+;
+; CHECK-SVE2p2-LABEL: test_compressstore_v16i8:
+; CHECK-SVE2p2: // %bb.0:
+; CHECK-SVE2p2-NEXT: shl v1.16b, v1.16b, #7
+; CHECK-SVE2p2-NEXT: ptrue p0.b, vl16
+; CHECK-SVE2p2-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-SVE2p2-NEXT: cmpne p1.b, p0/z, z1.b, #0
+; CHECK-SVE2p2-NEXT: cntp x8, p1, p1.b
+; CHECK-SVE2p2-NEXT: compact z0.b, p1, z0.b
+; CHECK-SVE2p2-NEXT: whilelo p0.b, xzr, x8
+; CHECK-SVE2p2-NEXT: st1b { z0.b }, p0, [x0]
+; CHECK-SVE2p2-NEXT: ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v16i8:
+; CHECK-SME2p2-STREAMING: // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT: lsl z1.b, z1.b, #7
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p0.b, vl16
+; CHECK-SME2p2-STREAMING-NEXT: asr z1.b, z1.b, #7
+; CHECK-SME2p2-STREAMING-NEXT: cmpne p1.b, p0/z, z1.b, #0
+; CHECK-SME2p2-STREAMING-NEXT: cntp x8, p1, p1.b
+; CHECK-SME2p2-STREAMING-NEXT: compact z0.b, p1, z0.b
+; CHECK-SME2p2-STREAMING-NEXT: whilelo p0.b, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT: st1b { z0.b }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT: ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v16i8:
+; CHECK-STREAMING-COMPAT: // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q1 killed $q1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p0.b, vl16
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT: lsl z1.b, z1.b, #7
+; CHECK-STREAMING-COMPAT-NEXT: asr z1.b, z1.b, #7
+; CHECK-STREAMING-COMPAT-NEXT: cmpne p1.b, p0/z, z1.b, #0
+; CHECK-STREAMING-COMPAT-NEXT: cntp x8, p1, p1.b
+; CHECK-STREAMING-COMPAT-NEXT: compact z0.b, p1, z0.b
+; CHECK-STREAMING-COMPAT-NEXT: whilelo p0.b, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT: st1b { z0.b }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT: ret
+ tail call void @llvm.masked.compressstore.v16i8(<16 x i8> %vec, ptr align 1 %p, <16 x i1> %mask)
+ ret void
+}
+
+define void @test_compressstore_v8f16(ptr %p, <8 x half> %vec, <8 x i1> %mask) {
+; CHECK-BASE-LABEL: test_compressstore_v8f16:
+; CHECK-BASE: // %bb.0:
+; CHECK-BASE-NEXT: ushll v1.8h, v1.8b, #0
+; CHECK-BASE-NEXT: ptrue p0.h, vl8
+; CHECK-BASE-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT: shl v1.8h, v1.8h, #15
+; CHECK-BASE-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-BASE-NEXT: cntp x8, p1, p1.h
+; CHECK-BASE-NEXT: compact z0.h, p1, z0.h
+; CHECK-BASE-NEXT: whilelo p0.h, xzr, x8
+; CHECK-BASE-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-BASE-NEXT: ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v8f16:
+; CHECK-VL256: // %bb.0:
+; CHECK-VL256-NEXT: ushll v1.8h, v1.8b, #0
+; CHECK-VL256-NEXT: ptrue p0.h, vl8
+; CHECK-VL256-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT: shl v1.8h, v1.8h, #15
+; CHECK-VL256-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-VL256-NEXT: cntp x8, p1, p1.h
+; CHECK-VL256-NEXT: compact z0.h, p1, z0.h
+; CHECK-VL256-NEXT: whilelo p0.h, xzr, x8
+; CHECK-VL256-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-VL256-NEXT: ret
+;
+; CHECK-SVE2p2-LABEL: test_compressstore_v8f16:
+; CHECK-SVE2p2: // %bb.0:
+; CHECK-SVE2p2-NEXT: ushll v1.8h, v1.8b, #0
+; CHECK-SVE2p2-NEXT: ptrue p0.h, vl8
+; CHECK-SVE2p2-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-SVE2p2-NEXT: shl v1.8h, v1.8h, #15
+; CHECK-SVE2p2-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SVE2p2-NEXT: cntp x8, p1, p1.h
+; CHECK-SVE2p2-NEXT: compact z0.h, p1, z0.h
+; CHECK-SVE2p2-NEXT: whilelo p0.h, xzr, x8
+; CHECK-SVE2p2-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-SVE2p2-NEXT: ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v8f16:
+; CHECK-SME2p2-STREAMING: // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT: uunpklo z1.h, z1.b
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p0.h, vl8
+; CHECK-SME2p2-STREAMING-NEXT: lsl z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT: asr z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SME2p2-STREAMING-NEXT: cntp x8, p1, p1.h
+; CHECK-SME2p2-STREAMING-NEXT: compact z0.h, p1, z0.h
+; CHECK-SME2p2-STREAMING-NEXT: whilelo p0.h, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT: ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v8f16:
+; CHECK-STREAMING-COMPAT: // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p0.h, vl8
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT: uunpklo z1.h, z1.b
+; CHECK-STREAMING-COMPAT-NEXT: lsl z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT: asr z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-STREAMING-COMPAT-NEXT: cntp x8, p1, p1.h
+; CHECK-STREAMING-COMPAT-NEXT: compact z0.h, p1, z0.h
+; CHECK-STREAMING-COMPAT-NEXT: whilelo p0.h, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT: ret
+ tail call void @llvm.masked.compressstore.v8f16(<8 x half> %vec, ptr align 1 %p, <8 x i1> %mask)
+ ret void
+}
+
+define void @test_compressstore_v8bf16(ptr %p, <8 x bfloat> %vec, <8 x i1> %mask) {
+; CHECK-BASE-LABEL: test_compressstore_v8bf16:
+; CHECK-BASE: // %bb.0:
+; CHECK-BASE-NEXT: ushll v1.8h, v1.8b, #0
+; CHECK-BASE-NEXT: ptrue p0.h, vl8
+; CHECK-BASE-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT: shl v1.8h, v1.8h, #15
+; CHECK-BASE-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-BASE-NEXT: cntp x8, p1, p1.h
+; CHECK-BASE-NEXT: compact z0.h, p1, z0.h
+; CHECK-BASE-NEXT: whilelo p0.h, xzr, x8
+; CHECK-BASE-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-BASE-NEXT: ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v8bf16:
+; CHECK-VL256: // %bb.0:
+; CHECK-VL256-NEXT: ushll v1.8h, v1.8b, #0
+; CHECK-VL256-NEXT: ptrue p0.h, vl8
+; CHECK-VL256-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT: shl v1.8h, v1.8h, #15
+; CHECK-VL256-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-VL256-NEXT: cntp x8, p1, p1.h
+; CHECK-VL256-NEXT: compact z0.h, p1, z0.h
+; CHECK-VL256-NEXT: whilelo p0.h, xzr, x8
+; CHECK-VL256-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-VL256-NEXT: ret
+;
+; CHECK-SVE2p2-LABEL: test_compressstore_v8bf16:
+; CHECK-SVE2p2: // %bb.0:
+; CHECK-SVE2p2-NEXT: ushll v1.8h, v1.8b, #0
+; CHECK-SVE2p2-NEXT: ptrue p0.h, vl8
+; CHECK-SVE2p2-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-SVE2p2-NEXT: shl v1.8h, v1.8h, #15
+; CHECK-SVE2p2-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SVE2p2-NEXT: cntp x8, p1, p1.h
+; CHECK-SVE2p2-NEXT: compact z0.h, p1, z0.h
+; CHECK-SVE2p2-NEXT: whilelo p0.h, xzr, x8
+; CHECK-SVE2p2-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-SVE2p2-NEXT: ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v8bf16:
+; CHECK-SME2p2-STREAMING: // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT: uunpklo z1.h, z1.b
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p0.h, vl8
+; CHECK-SME2p2-STREAMING-NEXT: lsl z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT: asr z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SME2p2-STREAMING-NEXT: cntp x8, p1, p1.h
+; CHECK-SME2p2-STREAMING-NEXT: compact z0.h, p1, z0.h
+; CHECK-SME2p2-STREAMING-NEXT: whilelo p0.h, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT: ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v8bf16:
+; CHECK-STREAMING-COMPAT: // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p0.h, vl8
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT: uunpklo z1.h, z1.b
+; CHECK-STREAMING-COMPAT-NEXT: lsl z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT: asr z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT: cmpne p1.h, p0/z, z1.h, #0
+; CHECK-STREAMING-COMPAT-NEXT: cntp x8, p1, p1.h
+; CHECK-STREAMING-COMPAT-NEXT: compact z0.h, p1, z0.h
+; CHECK-STREAMING-COMPAT-NEXT: whilelo p0.h, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT: st1h { z0.h }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT: ret
+ tail call void @llvm.masked.compressstore.v8bf16(<8 x bfloat> %vec, ptr align 1 %p, <8 x i1> %mask)
+ ret void
+}
>From e692e9928804b12a995bc31772aabd201a0dbb95 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 26 Aug 2026 14:45:32 +0100
Subject: [PATCH 07/12] Simplify Target Hook Check & Format
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 3 ++-
llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h | 5 +++--
2 files changed, 5 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index cdbf7fed84e10..4439d6b826ebb 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2227,7 +2227,8 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
if (Subtarget->hasSVE2p2() || Subtarget->hasSME2p2()) {
// With +sve2p2/+sme2p2 the full range of vector types are supported.
- for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16}) {
+ for (auto VT :
+ {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16}) {
setOperationAction(ISD::MSTORE, VT, Custom);
setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
}
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index 5c7b2703b4257..71878619b6570 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -334,8 +334,9 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
}
bool isElementTypeLegalForCompressStore(Type *Ty) const {
- if ((ST->hasSVE2p2() || (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())) &&
- (Ty->isIntegerTy(8) || Ty->isIntegerTy(16) || Ty->getScalarSizeInBits() == 16))
+ if ((ST->hasSVE2p2() ||
+ (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())) &&
+ (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16))
return true;
return Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
Ty->isIntegerTy(64);
>From 8c824d4675bfc1aa37ecaf127e1978d405b66b71 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 27 Aug 2026 10:05:30 +0100
Subject: [PATCH 08/12] Allow i8/i16 types in SME2p2 Streaming Mode
---
.../AArch64/AArch64TargetTransformInfo.h | 12 ++++----
.../CostModel/AArch64/masked_compress_load.ll | 30 +++++++++----------
2 files changed, 21 insertions(+), 21 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index 71878619b6570..e11b4488d3e60 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -334,17 +334,17 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
}
bool isElementTypeLegalForCompressStore(Type *Ty) const {
- if ((ST->hasSVE2p2() ||
- (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())) &&
- (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16))
- return true;
- return Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
+ // For streaming SME2p2, only consider 8bit or 16bit Scalar types.
+ if ((ST->isStreamingSVEAvailable() && ST->hasSME2p2()))
+ return Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16;
+
+ return ((ST->hasSVE2p2() || ST->hasSME2p2()) && (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16)) || Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
Ty->isIntegerTy(64);
}
bool isLegalMaskedCompressStore(Type *DataType,
Align Alignment) const override {
- if (!ST->isSVEorStreamingSVEAvailable())
+ if (!(ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())))
return false;
if (isa<FixedVectorType>(DataType) &&
diff --git a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
index 7cc880ce5b66d..69a3edb499b31 100644
--- a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
@@ -32,21 +32,21 @@ define void @fixed() {
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i8.p0(<2 x i8> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i8.p0(<4 x i8> poison, ptr poison, <4 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8i8.p0(<8 x i8> poison, ptr poison, <8 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v16i8.p0(<16 x i8> poison, ptr poison, <16 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v16i8.p0(<16 x i8> poison, ptr poison, <16 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i16.p0(<2 x i16> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i16.p0(<4 x i16> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8i16.p0(<8 x i16> poison, ptr poison, <8 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8i16.p0(<8 x i16> poison, ptr poison, <8 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i32.p0(<2 x i32> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i32.p0(<4 x i32> poison, ptr poison, <4 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i64.p0(<2 x i64> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f16.p0(<2 x half> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4f16.p0(<4 x half> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8f16.p0(<8 x half> poison, ptr poison, <8 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8f16.p0(<8 x half> poison, ptr poison, <8 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f32.p0(<2 x float> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4f32.p0(<4 x float> poison, ptr poison, <4 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f64.p0(<2 x double> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i64.p0(<4 x i64> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v32f16.p0(<32 x half> poison, ptr poison, <32 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.v32f16.p0(<32 x half> poison, ptr poison, <32 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; SVE2p2-SME2p2-LABEL: 'fixed'
@@ -164,25 +164,25 @@ define void @scalable() {
; SVE2p2-SME2p2-NON-STREAMING-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; SVE2p2-SME2p2-STREAMING-LABEL: 'scalable'
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv1i64.p0(<vscale x 1 x i64> poison, ptr poison, <vscale x 1 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; SVE2p2-SME2p2-LABEL: 'scalable'
>From 3248b1571b3c524a277e2c329220a7da8c7fde18 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 27 Aug 2026 10:06:36 +0100
Subject: [PATCH 09/12] format
---
llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h | 7 +++++--
1 file changed, 5 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index e11b4488d3e60..2420e01eb3b3d 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -338,13 +338,16 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
if ((ST->isStreamingSVEAvailable() && ST->hasSME2p2()))
return Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16;
- return ((ST->hasSVE2p2() || ST->hasSME2p2()) && (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16)) || Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
+ return ((ST->hasSVE2p2() || ST->hasSME2p2()) &&
+ (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16)) ||
+ Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
Ty->isIntegerTy(64);
}
bool isLegalMaskedCompressStore(Type *DataType,
Align Alignment) const override {
- if (!(ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())))
+ if (!(ST->isSVEAvailable() ||
+ (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())))
return false;
if (isa<FixedVectorType>(DataType) &&
>From 20e2f7cb64997577a71d8d4919e476ca411e9b24 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 27 Aug 2026 10:09:27 +0100
Subject: [PATCH 10/12] Add lowering comment
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 1 +
1 file changed, 1 insertion(+)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 4439d6b826ebb..8f68e4af04207 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2229,6 +2229,7 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
// With +sve2p2/+sme2p2 the full range of vector types are supported.
for (auto VT :
{MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16}) {
+ // Use custom lowering for MSTORE so we can handle compressstore (using VECTOR_COMPRESS).
setOperationAction(ISD::MSTORE, VT, Custom);
setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
}
>From 4550b93b425614f5903fb8c624fab1cdbae3401f Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 27 Aug 2026 10:10:33 +0100
Subject: [PATCH 11/12] format
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 3 ++-
1 file changed, 2 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 8f68e4af04207..83c07a19a55e8 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2229,7 +2229,8 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
// With +sve2p2/+sme2p2 the full range of vector types are supported.
for (auto VT :
{MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16}) {
- // Use custom lowering for MSTORE so we can handle compressstore (using VECTOR_COMPRESS).
+ // Use custom lowering for MSTORE so we can handle compressstore (using
+ // VECTOR_COMPRESS).
setOperationAction(ISD::MSTORE, VT, Custom);
setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
}
>From b881222cc321c29b7af7c122fc72d3af3939e3d4 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 27 Aug 2026 11:49:57 +0100
Subject: [PATCH 12/12] Fix SME2p2 lowering to allow 32/64bit types
---
.../Target/AArch64/AArch64ISelLowering.cpp | 21 +-
.../AArch64/AArch64TargetTransformInfo.h | 17 +-
.../CostModel/AArch64/masked_compress_load.ll | 24 +-
.../AArch64/sve-masked-compressstore.ll | 490 ++++++++++++++++--
4 files changed, 485 insertions(+), 67 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 83c07a19a55e8..2bf587fd9136e 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2218,12 +2218,22 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
for (auto VT :
{MVT::nxv4i32, MVT::nxv2i64, MVT::nxv2f32, MVT::nxv4f32, MVT::nxv2f64})
setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
+ for (auto VT : {MVT::nxv2i8, MVT::nxv2i16, MVT::nxv2i32, MVT::nxv2i64,
+ MVT::nxv2f32, MVT::nxv2f64, MVT::nxv4i8, MVT::nxv4i16,
+ MVT::nxv4i32, MVT::nxv4f32}) {
+ // Use a custom lowering for masked stores that could be a supported
+ // compressing store. Note: These types still use the normal (Legal)
+ // lowering for non-compressing masked stores.
+ setOperationAction(ISD::MSTORE, VT, Custom);
+ }
// If we have SVE, we can use SVE logic for legal NEON vectors in the lowest
// bits of the SVE register.
for (auto VT : {MVT::v2i32, MVT::v4i32, MVT::v2i64, MVT::v2f32, MVT::v4f32,
- MVT::v2f64})
+ MVT::v2f64}) {
+ setOperationAction(ISD::MSTORE, VT, Custom);
setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
+ }
if (Subtarget->hasSVE2p2() || Subtarget->hasSME2p2()) {
// With +sve2p2/+sme2p2 the full range of vector types are supported.
@@ -2280,15 +2290,6 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
MVT::v2f32, MVT::v4f32, MVT::v2f64})
setOperationAction(ISD::VECREDUCE_SEQ_FADD, VT, Custom);
- for (auto VT : {MVT::nxv2i8, MVT::nxv2i16, MVT::nxv2i32, MVT::nxv2i64,
- MVT::nxv2f32, MVT::nxv2f64, MVT::nxv4i8, MVT::nxv4i16,
- MVT::nxv4i32, MVT::nxv4f32}) {
- // Use a custom lowering for masked stores that could be a supported
- // compressing store. Note: These types still use the normal (Legal)
- // lowering for non-compressing masked stores.
- setOperationAction(ISD::MSTORE, VT, Custom);
- }
-
// Histcnt is SVE2 only
if (Subtarget->hasSVE2()) {
setOperationAction(ISD::EXPERIMENTAL_VECTOR_HISTOGRAM, MVT::nxv4i32,
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index 2420e01eb3b3d..3038978be259e 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -334,14 +334,15 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
}
bool isElementTypeLegalForCompressStore(Type *Ty) const {
- // For streaming SME2p2, only consider 8bit or 16bit Scalar types.
- if ((ST->isStreamingSVEAvailable() && ST->hasSME2p2()))
- return Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16;
-
- return ((ST->hasSVE2p2() || ST->hasSME2p2()) &&
- (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16)) ||
- Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
- Ty->isIntegerTy(64);
+ // 32-bit and 64-bit element types are legal if we have SVE.
+ if (Ty->getScalarSizeInBits() == 32 || Ty->getScalarSizeInBits() == 64)
+ return true;
+
+ // 8-bit and 16-bit types require +sve2p2 or +sme2p2.
+ if (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16)
+ return ST->hasSVE2p2() || ST->hasSME2p2();
+
+ return false;
}
bool isLegalMaskedCompressStore(Type *DataType,
diff --git a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
index 69a3edb499b31..54396e133d800 100644
--- a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
@@ -37,15 +37,15 @@ define void @fixed() {
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i16.p0(<4 x i16> poison, ptr poison, <4 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8i16.p0(<8 x i16> poison, ptr poison, <8 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i32.p0(<2 x i32> poison, ptr poison, <2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i32.p0(<4 x i32> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i64.p0(<2 x i64> poison, ptr poison, <2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v4i32.p0(<4 x i32> poison, ptr poison, <4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v2i64.p0(<2 x i64> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f16.p0(<2 x half> poison, ptr poison, <2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4f16.p0(<4 x half> poison, ptr poison, <4 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8f16.p0(<8 x half> poison, ptr poison, <8 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f32.p0(<2 x float> poison, ptr poison, <2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4f32.p0(<4 x float> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f64.p0(<2 x double> poison, ptr poison, <2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i64.p0(<4 x i64> poison, ptr poison, <4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v4f32.p0(<4 x float> poison, ptr poison, <4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v2f64.p0(<2 x double> poison, ptr poison, <2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.v4i64.p0(<4 x i64> poison, ptr poison, <4 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.v32f16.p0(<32 x half> poison, ptr poison, <32 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
@@ -171,17 +171,17 @@ define void @scalable() {
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv1i64.p0(<vscale x 1 x i64> poison, ptr poison, <vscale x 1 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
; SVE2p2-SME2p2-STREAMING-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
diff --git a/llvm/test/CodeGen/AArch64/sve-masked-compressstore.ll b/llvm/test/CodeGen/AArch64/sve-masked-compressstore.ll
index df449f79cc9b5..b6fa9ea943c75 100644
--- a/llvm/test/CodeGen/AArch64/sve-masked-compressstore.ll
+++ b/llvm/test/CodeGen/AArch64/sve-masked-compressstore.ll
@@ -1,7 +1,9 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
; RUN: llc -mtriple=aarch64 -mattr=+sve < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BASE
; RUN: llc -mtriple=aarch64 -aarch64-sve-vector-bits-min=256 -mattr=+sve < %s | FileCheck %s --check-prefixes=CHECK,CHECK-VL256
-
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve,+sme2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SME2p2
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2 -force-streaming < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SME2p2-STREAMING
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2,+sve2p2 -force-streaming-compatible < %s | FileCheck %s --check-prefixes=CHECK,CHECK-STREAMING-COMPAT
;; Full SVE vectors (supported with +sve)
define void @test_compressstore_nxv4i32(ptr %p, <vscale x 4 x i32> %vec, <vscale x 4 x i1> %mask) {
@@ -115,52 +117,214 @@ define void @test_compressstore_nxv4i16(ptr %p, <vscale x 4 x i16> %vec, <vscale
;; NEON vector types (promoted to SVE)
define void @test_compressstore_v2f64(ptr %p, <2 x double> %vec, <2 x i1> %mask) {
-; CHECK-LABEL: test_compressstore_v2f64:
-; CHECK: // %bb.0:
-; CHECK-NEXT: ushll v1.2d, v1.2s, #0
-; CHECK-NEXT: ptrue p0.d, vl2
-; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
-; CHECK-NEXT: shl v1.2d, v1.2d, #63
-; CHECK-NEXT: cmpne p1.d, p0/z, z1.d, #0
-; CHECK-NEXT: cntp x8, p1, p1.d
-; CHECK-NEXT: compact z0.d, p1, z0.d
-; CHECK-NEXT: whilelo p0.d, xzr, x8
-; CHECK-NEXT: st1d { z0.d }, p0, [x0]
-; CHECK-NEXT: ret
+; CHECK-BASE-LABEL: test_compressstore_v2f64:
+; CHECK-BASE: // %bb.0:
+; CHECK-BASE-NEXT: ushll v1.2d, v1.2s, #0
+; CHECK-BASE-NEXT: ptrue p0.d, vl2
+; CHECK-BASE-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT: shl v1.2d, v1.2d, #63
+; CHECK-BASE-NEXT: cmpne p1.d, p0/z, z1.d, #0
+; CHECK-BASE-NEXT: cntp x8, p1, p1.d
+; CHECK-BASE-NEXT: compact z0.d, p1, z0.d
+; CHECK-BASE-NEXT: whilelo p0.d, xzr, x8
+; CHECK-BASE-NEXT: st1d { z0.d }, p0, [x0]
+; CHECK-BASE-NEXT: ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v2f64:
+; CHECK-VL256: // %bb.0:
+; CHECK-VL256-NEXT: ushll v1.2d, v1.2s, #0
+; CHECK-VL256-NEXT: ptrue p0.d, vl2
+; CHECK-VL256-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT: shl v1.2d, v1.2d, #63
+; CHECK-VL256-NEXT: cmpne p1.d, p0/z, z1.d, #0
+; CHECK-VL256-NEXT: cntp x8, p1, p1.d
+; CHECK-VL256-NEXT: compact z0.d, p1, z0.d
+; CHECK-VL256-NEXT: whilelo p0.d, xzr, x8
+; CHECK-VL256-NEXT: st1d { z0.d }, p0, [x0]
+; CHECK-VL256-NEXT: ret
+;
+; CHECK-SME2p2-LABEL: test_compressstore_v2f64:
+; CHECK-SME2p2: // %bb.0:
+; CHECK-SME2p2-NEXT: ushll v1.2d, v1.2s, #0
+; CHECK-SME2p2-NEXT: ptrue p0.d, vl2
+; CHECK-SME2p2-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-SME2p2-NEXT: shl v1.2d, v1.2d, #63
+; CHECK-SME2p2-NEXT: cmpne p1.d, p0/z, z1.d, #0
+; CHECK-SME2p2-NEXT: cntp x8, p1, p1.d
+; CHECK-SME2p2-NEXT: compact z0.d, p1, z0.d
+; CHECK-SME2p2-NEXT: whilelo p0.d, xzr, x8
+; CHECK-SME2p2-NEXT: st1d { z0.d }, p0, [x0]
+; CHECK-SME2p2-NEXT: ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v2f64:
+; CHECK-SME2p2-STREAMING: // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT: uunpklo z1.d, z1.s
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p0.d, vl2
+; CHECK-SME2p2-STREAMING-NEXT: lsl z1.d, z1.d, #63
+; CHECK-SME2p2-STREAMING-NEXT: asr z1.d, z1.d, #63
+; CHECK-SME2p2-STREAMING-NEXT: cmpne p1.d, p0/z, z1.d, #0
+; CHECK-SME2p2-STREAMING-NEXT: cntp x8, p1, p1.d
+; CHECK-SME2p2-STREAMING-NEXT: compact z0.d, p1, z0.d
+; CHECK-SME2p2-STREAMING-NEXT: whilelo p0.d, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT: st1d { z0.d }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT: ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v2f64:
+; CHECK-STREAMING-COMPAT: // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p0.d, vl2
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT: uunpklo z1.d, z1.s
+; CHECK-STREAMING-COMPAT-NEXT: lsl z1.d, z1.d, #63
+; CHECK-STREAMING-COMPAT-NEXT: asr z1.d, z1.d, #63
+; CHECK-STREAMING-COMPAT-NEXT: cmpne p1.d, p0/z, z1.d, #0
+; CHECK-STREAMING-COMPAT-NEXT: cntp x8, p1, p1.d
+; CHECK-STREAMING-COMPAT-NEXT: compact z0.d, p1, z0.d
+; CHECK-STREAMING-COMPAT-NEXT: whilelo p0.d, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT: st1d { z0.d }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT: ret
tail call void @llvm.masked.compressstore.v2f64(<2 x double> %vec, ptr align 8 %p, <2 x i1> %mask)
ret void
}
define void @test_compressstore_v4i32(ptr %p, <4 x i32> %vec, <4 x i1> %mask) {
-; CHECK-LABEL: test_compressstore_v4i32:
-; CHECK: // %bb.0:
-; CHECK-NEXT: ushll v1.4s, v1.4h, #0
-; CHECK-NEXT: ptrue p0.s, vl4
-; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
-; CHECK-NEXT: shl v1.4s, v1.4s, #31
-; CHECK-NEXT: cmpne p1.s, p0/z, z1.s, #0
-; CHECK-NEXT: cntp x8, p1, p1.s
-; CHECK-NEXT: compact z0.s, p1, z0.s
-; CHECK-NEXT: whilelo p0.s, xzr, x8
-; CHECK-NEXT: st1w { z0.s }, p0, [x0]
-; CHECK-NEXT: ret
+; CHECK-BASE-LABEL: test_compressstore_v4i32:
+; CHECK-BASE: // %bb.0:
+; CHECK-BASE-NEXT: ushll v1.4s, v1.4h, #0
+; CHECK-BASE-NEXT: ptrue p0.s, vl4
+; CHECK-BASE-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT: shl v1.4s, v1.4s, #31
+; CHECK-BASE-NEXT: cmpne p1.s, p0/z, z1.s, #0
+; CHECK-BASE-NEXT: cntp x8, p1, p1.s
+; CHECK-BASE-NEXT: compact z0.s, p1, z0.s
+; CHECK-BASE-NEXT: whilelo p0.s, xzr, x8
+; CHECK-BASE-NEXT: st1w { z0.s }, p0, [x0]
+; CHECK-BASE-NEXT: ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v4i32:
+; CHECK-VL256: // %bb.0:
+; CHECK-VL256-NEXT: ushll v1.4s, v1.4h, #0
+; CHECK-VL256-NEXT: ptrue p0.s, vl4
+; CHECK-VL256-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT: shl v1.4s, v1.4s, #31
+; CHECK-VL256-NEXT: cmpne p1.s, p0/z, z1.s, #0
+; CHECK-VL256-NEXT: cntp x8, p1, p1.s
+; CHECK-VL256-NEXT: compact z0.s, p1, z0.s
+; CHECK-VL256-NEXT: whilelo p0.s, xzr, x8
+; CHECK-VL256-NEXT: st1w { z0.s }, p0, [x0]
+; CHECK-VL256-NEXT: ret
+;
+; CHECK-SME2p2-LABEL: test_compressstore_v4i32:
+; CHECK-SME2p2: // %bb.0:
+; CHECK-SME2p2-NEXT: ushll v1.4s, v1.4h, #0
+; CHECK-SME2p2-NEXT: ptrue p0.s, vl4
+; CHECK-SME2p2-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-SME2p2-NEXT: shl v1.4s, v1.4s, #31
+; CHECK-SME2p2-NEXT: cmpne p1.s, p0/z, z1.s, #0
+; CHECK-SME2p2-NEXT: cntp x8, p1, p1.s
+; CHECK-SME2p2-NEXT: compact z0.s, p1, z0.s
+; CHECK-SME2p2-NEXT: whilelo p0.s, xzr, x8
+; CHECK-SME2p2-NEXT: st1w { z0.s }, p0, [x0]
+; CHECK-SME2p2-NEXT: ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v4i32:
+; CHECK-SME2p2-STREAMING: // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT: uunpklo z1.s, z1.h
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p0.s, vl4
+; CHECK-SME2p2-STREAMING-NEXT: lsl z1.s, z1.s, #31
+; CHECK-SME2p2-STREAMING-NEXT: asr z1.s, z1.s, #31
+; CHECK-SME2p2-STREAMING-NEXT: cmpne p1.s, p0/z, z1.s, #0
+; CHECK-SME2p2-STREAMING-NEXT: cntp x8, p1, p1.s
+; CHECK-SME2p2-STREAMING-NEXT: compact z0.s, p1, z0.s
+; CHECK-SME2p2-STREAMING-NEXT: whilelo p0.s, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT: st1w { z0.s }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT: ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v4i32:
+; CHECK-STREAMING-COMPAT: // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p0.s, vl4
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT: uunpklo z1.s, z1.h
+; CHECK-STREAMING-COMPAT-NEXT: lsl z1.s, z1.s, #31
+; CHECK-STREAMING-COMPAT-NEXT: asr z1.s, z1.s, #31
+; CHECK-STREAMING-COMPAT-NEXT: cmpne p1.s, p0/z, z1.s, #0
+; CHECK-STREAMING-COMPAT-NEXT: cntp x8, p1, p1.s
+; CHECK-STREAMING-COMPAT-NEXT: compact z0.s, p1, z0.s
+; CHECK-STREAMING-COMPAT-NEXT: whilelo p0.s, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT: st1w { z0.s }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT: ret
tail call void @llvm.masked.compressstore.v4i32(<4 x i32> %vec, ptr align 4 %p, <4 x i1> %mask)
ret void
}
define void @test_compressstore_v2i64(ptr %p, <2 x i64> %vec, <2 x i1> %mask) {
-; CHECK-LABEL: test_compressstore_v2i64:
-; CHECK: // %bb.0:
-; CHECK-NEXT: ushll v1.2d, v1.2s, #0
-; CHECK-NEXT: ptrue p0.d, vl2
-; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
-; CHECK-NEXT: shl v1.2d, v1.2d, #63
-; CHECK-NEXT: cmpne p1.d, p0/z, z1.d, #0
-; CHECK-NEXT: cntp x8, p1, p1.d
-; CHECK-NEXT: compact z0.d, p1, z0.d
-; CHECK-NEXT: whilelo p0.d, xzr, x8
-; CHECK-NEXT: st1d { z0.d }, p0, [x0]
-; CHECK-NEXT: ret
+; CHECK-BASE-LABEL: test_compressstore_v2i64:
+; CHECK-BASE: // %bb.0:
+; CHECK-BASE-NEXT: ushll v1.2d, v1.2s, #0
+; CHECK-BASE-NEXT: ptrue p0.d, vl2
+; CHECK-BASE-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT: shl v1.2d, v1.2d, #63
+; CHECK-BASE-NEXT: cmpne p1.d, p0/z, z1.d, #0
+; CHECK-BASE-NEXT: cntp x8, p1, p1.d
+; CHECK-BASE-NEXT: compact z0.d, p1, z0.d
+; CHECK-BASE-NEXT: whilelo p0.d, xzr, x8
+; CHECK-BASE-NEXT: st1d { z0.d }, p0, [x0]
+; CHECK-BASE-NEXT: ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v2i64:
+; CHECK-VL256: // %bb.0:
+; CHECK-VL256-NEXT: ushll v1.2d, v1.2s, #0
+; CHECK-VL256-NEXT: ptrue p0.d, vl2
+; CHECK-VL256-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT: shl v1.2d, v1.2d, #63
+; CHECK-VL256-NEXT: cmpne p1.d, p0/z, z1.d, #0
+; CHECK-VL256-NEXT: cntp x8, p1, p1.d
+; CHECK-VL256-NEXT: compact z0.d, p1, z0.d
+; CHECK-VL256-NEXT: whilelo p0.d, xzr, x8
+; CHECK-VL256-NEXT: st1d { z0.d }, p0, [x0]
+; CHECK-VL256-NEXT: ret
+;
+; CHECK-SME2p2-LABEL: test_compressstore_v2i64:
+; CHECK-SME2p2: // %bb.0:
+; CHECK-SME2p2-NEXT: ushll v1.2d, v1.2s, #0
+; CHECK-SME2p2-NEXT: ptrue p0.d, vl2
+; CHECK-SME2p2-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-SME2p2-NEXT: shl v1.2d, v1.2d, #63
+; CHECK-SME2p2-NEXT: cmpne p1.d, p0/z, z1.d, #0
+; CHECK-SME2p2-NEXT: cntp x8, p1, p1.d
+; CHECK-SME2p2-NEXT: compact z0.d, p1, z0.d
+; CHECK-SME2p2-NEXT: whilelo p0.d, xzr, x8
+; CHECK-SME2p2-NEXT: st1d { z0.d }, p0, [x0]
+; CHECK-SME2p2-NEXT: ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v2i64:
+; CHECK-SME2p2-STREAMING: // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT: uunpklo z1.d, z1.s
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p0.d, vl2
+; CHECK-SME2p2-STREAMING-NEXT: lsl z1.d, z1.d, #63
+; CHECK-SME2p2-STREAMING-NEXT: asr z1.d, z1.d, #63
+; CHECK-SME2p2-STREAMING-NEXT: cmpne p1.d, p0/z, z1.d, #0
+; CHECK-SME2p2-STREAMING-NEXT: cntp x8, p1, p1.d
+; CHECK-SME2p2-STREAMING-NEXT: compact z0.d, p1, z0.d
+; CHECK-SME2p2-STREAMING-NEXT: whilelo p0.d, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT: st1d { z0.d }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT: ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v2i64:
+; CHECK-STREAMING-COMPAT: // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p0.d, vl2
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT: uunpklo z1.d, z1.s
+; CHECK-STREAMING-COMPAT-NEXT: lsl z1.d, z1.d, #63
+; CHECK-STREAMING-COMPAT-NEXT: asr z1.d, z1.d, #63
+; CHECK-STREAMING-COMPAT-NEXT: cmpne p1.d, p0/z, z1.d, #0
+; CHECK-STREAMING-COMPAT-NEXT: cntp x8, p1, p1.d
+; CHECK-STREAMING-COMPAT-NEXT: compact z0.d, p1, z0.d
+; CHECK-STREAMING-COMPAT-NEXT: whilelo p0.d, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT: st1d { z0.d }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT: ret
tail call void @llvm.masked.compressstore.v2i64(<2 x i64> %vec, ptr align 8 %p, <2 x i1> %mask)
ret void
}
@@ -219,6 +383,144 @@ define void @test_compressstore_v8i32(ptr %p, <8 x i32> %vec, <8 x i1> %mask) {
; CHECK-VL256-NEXT: whilelo p0.s, xzr, x8
; CHECK-VL256-NEXT: st1w { z0.s }, p0, [x0]
; CHECK-VL256-NEXT: ret
+;
+; CHECK-SME2p2-LABEL: test_compressstore_v8i32:
+; CHECK-SME2p2: // %bb.0:
+; CHECK-SME2p2-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-SME2p2-NEXT: zip1 v3.8b, v2.8b, v0.8b
+; CHECK-SME2p2-NEXT: zip2 v2.8b, v2.8b, v0.8b
+; CHECK-SME2p2-NEXT: adrp x8, .LCPI11_0
+; CHECK-SME2p2-NEXT: ldr d5, [x8, :lo12:.LCPI11_0]
+; CHECK-SME2p2-NEXT: ptrue p0.s, vl4
+; CHECK-SME2p2-NEXT: // kill: def $q1 killed $q1 def $z1
+; CHECK-SME2p2-NEXT: shl v4.4h, v3.4h, #15
+; CHECK-SME2p2-NEXT: ushll v2.4s, v2.4h, #0
+; CHECK-SME2p2-NEXT: ushll v3.4s, v3.4h, #0
+; CHECK-SME2p2-NEXT: cmlt v4.4h, v4.4h, #0
+; CHECK-SME2p2-NEXT: shl v2.4s, v2.4s, #31
+; CHECK-SME2p2-NEXT: shl v3.4s, v3.4s, #31
+; CHECK-SME2p2-NEXT: and v4.8b, v4.8b, v5.8b
+; CHECK-SME2p2-NEXT: cmpne p1.s, p0/z, z2.s, #0
+; CHECK-SME2p2-NEXT: cmpne p2.s, p0/z, z3.s, #0
+; CHECK-SME2p2-NEXT: ptrue p0.s
+; CHECK-SME2p2-NEXT: addv h2, v4.4h
+; CHECK-SME2p2-NEXT: cntp x9, p1, p1.s
+; CHECK-SME2p2-NEXT: compact z1.s, p1, z1.s
+; CHECK-SME2p2-NEXT: compact z0.s, p2, z0.s
+; CHECK-SME2p2-NEXT: cntp x10, p2, p2.s
+; CHECK-SME2p2-NEXT: fmov w8, s2
+; CHECK-SME2p2-NEXT: and w8, w8, #0xf
+; CHECK-SME2p2-NEXT: whilelo p1.s, xzr, x10
+; CHECK-SME2p2-NEXT: fmov s2, w8
+; CHECK-SME2p2-NEXT: cnt z2.s, p0/z, z2.s
+; CHECK-SME2p2-NEXT: whilelo p0.s, xzr, x9
+; CHECK-SME2p2-NEXT: fmov w8, s2
+; CHECK-SME2p2-NEXT: st1w { z1.s }, p0, [x0, x8, lsl #2]
+; CHECK-SME2p2-NEXT: st1w { z0.s }, p1, [x0]
+; CHECK-SME2p2-NEXT: ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v8i32:
+; CHECK-SME2p2-STREAMING: // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT: sub sp, sp, #16
+; CHECK-SME2p2-STREAMING-NEXT: .cfi_def_cfa_offset 16
+; CHECK-SME2p2-STREAMING-NEXT: mov z3.b, z2.b[7]
+; CHECK-SME2p2-STREAMING-NEXT: mov z4.b, z2.b[6]
+; CHECK-SME2p2-STREAMING-NEXT: mov z5.b, z2.b[5]
+; CHECK-SME2p2-STREAMING-NEXT: mov z6.b, z2.b[4]
+; CHECK-SME2p2-STREAMING-NEXT: mov z7.b, z2.b[1]
+; CHECK-SME2p2-STREAMING-NEXT: mov z16.b, z2.b[2]
+; CHECK-SME2p2-STREAMING-NEXT: mov z17.b, z2.b[3]
+; CHECK-SME2p2-STREAMING-NEXT: fmov w8, s2
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p0.s, vl4
+; CHECK-SME2p2-STREAMING-NEXT: zip1 z3.h, z4.h, z3.h
+; CHECK-SME2p2-STREAMING-NEXT: zip1 z4.h, z6.h, z5.h
+; CHECK-SME2p2-STREAMING-NEXT: zip1 z6.h, z2.h, z7.h
+; CHECK-SME2p2-STREAMING-NEXT: fmov w9, s7
+; CHECK-SME2p2-STREAMING-NEXT: and w8, w8, #0x1
+; CHECK-SME2p2-STREAMING-NEXT: zip1 z5.h, z16.h, z17.h
+; CHECK-SME2p2-STREAMING-NEXT: zip1 z3.s, z4.s, z3.s
+; CHECK-SME2p2-STREAMING-NEXT: bfi w8, w9, #1, #1
+; CHECK-SME2p2-STREAMING-NEXT: fmov w9, s16
+; CHECK-SME2p2-STREAMING-NEXT: zip1 z4.s, z6.s, z5.s
+; CHECK-SME2p2-STREAMING-NEXT: uunpklo z3.s, z3.h
+; CHECK-SME2p2-STREAMING-NEXT: bfi w8, w9, #2, #1
+; CHECK-SME2p2-STREAMING-NEXT: fmov w9, s17
+; CHECK-SME2p2-STREAMING-NEXT: uunpklo z4.s, z4.h
+; CHECK-SME2p2-STREAMING-NEXT: orr w8, w8, w9, lsl #3
+; CHECK-SME2p2-STREAMING-NEXT: lsl z3.s, z3.s, #31
+; CHECK-SME2p2-STREAMING-NEXT: lsl z4.s, z4.s, #31
+; CHECK-SME2p2-STREAMING-NEXT: and w8, w8, #0xf
+; CHECK-SME2p2-STREAMING-NEXT: asr z2.s, z3.s, #31
+; CHECK-SME2p2-STREAMING-NEXT: asr z3.s, z4.s, #31
+; CHECK-SME2p2-STREAMING-NEXT: cmpne p1.s, p0/z, z2.s, #0
+; CHECK-SME2p2-STREAMING-NEXT: fmov s2, w8
+; CHECK-SME2p2-STREAMING-NEXT: cmpne p2.s, p0/z, z3.s, #0
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p0.s
+; CHECK-SME2p2-STREAMING-NEXT: cnt z2.s, p0/z, z2.s
+; CHECK-SME2p2-STREAMING-NEXT: cntp x9, p1, p1.s
+; CHECK-SME2p2-STREAMING-NEXT: compact z1.s, p1, z1.s
+; CHECK-SME2p2-STREAMING-NEXT: fmov w10, s2
+; CHECK-SME2p2-STREAMING-NEXT: cntp x8, p2, p2.s
+; CHECK-SME2p2-STREAMING-NEXT: compact z0.s, p2, z0.s
+; CHECK-SME2p2-STREAMING-NEXT: whilelo p0.s, xzr, x9
+; CHECK-SME2p2-STREAMING-NEXT: whilelo p1.s, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT: st1w { z1.s }, p0, [x0, x10, lsl #2]
+; CHECK-SME2p2-STREAMING-NEXT: st1w { z0.s }, p1, [x0]
+; CHECK-SME2p2-STREAMING-NEXT: add sp, sp, #16
+; CHECK-SME2p2-STREAMING-NEXT: ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v8i32:
+; CHECK-STREAMING-COMPAT: // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT: sub sp, sp, #16
+; CHECK-STREAMING-COMPAT-NEXT: .cfi_def_cfa_offset 16
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $d2 killed $d2 def $z2
+; CHECK-STREAMING-COMPAT-NEXT: mov z3.b, z2.b[7]
+; CHECK-STREAMING-COMPAT-NEXT: mov z4.b, z2.b[6]
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q1 killed $q1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT: mov z5.b, z2.b[5]
+; CHECK-STREAMING-COMPAT-NEXT: mov z6.b, z2.b[4]
+; CHECK-STREAMING-COMPAT-NEXT: mov z7.b, z2.b[1]
+; CHECK-STREAMING-COMPAT-NEXT: mov z16.b, z2.b[2]
+; CHECK-STREAMING-COMPAT-NEXT: mov z17.b, z2.b[3]
+; CHECK-STREAMING-COMPAT-NEXT: fmov w8, s2
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p0.s, vl4
+; CHECK-STREAMING-COMPAT-NEXT: zip1 z3.h, z4.h, z3.h
+; CHECK-STREAMING-COMPAT-NEXT: zip1 z4.h, z6.h, z5.h
+; CHECK-STREAMING-COMPAT-NEXT: zip1 z6.h, z2.h, z7.h
+; CHECK-STREAMING-COMPAT-NEXT: fmov w9, s7
+; CHECK-STREAMING-COMPAT-NEXT: and w8, w8, #0x1
+; CHECK-STREAMING-COMPAT-NEXT: zip1 z5.h, z16.h, z17.h
+; CHECK-STREAMING-COMPAT-NEXT: zip1 z3.s, z4.s, z3.s
+; CHECK-STREAMING-COMPAT-NEXT: bfi w8, w9, #1, #1
+; CHECK-STREAMING-COMPAT-NEXT: fmov w9, s16
+; CHECK-STREAMING-COMPAT-NEXT: zip1 z4.s, z6.s, z5.s
+; CHECK-STREAMING-COMPAT-NEXT: uunpklo z3.s, z3.h
+; CHECK-STREAMING-COMPAT-NEXT: bfi w8, w9, #2, #1
+; CHECK-STREAMING-COMPAT-NEXT: fmov w9, s17
+; CHECK-STREAMING-COMPAT-NEXT: uunpklo z4.s, z4.h
+; CHECK-STREAMING-COMPAT-NEXT: orr w8, w8, w9, lsl #3
+; CHECK-STREAMING-COMPAT-NEXT: lsl z3.s, z3.s, #31
+; CHECK-STREAMING-COMPAT-NEXT: lsl z4.s, z4.s, #31
+; CHECK-STREAMING-COMPAT-NEXT: and w8, w8, #0xf
+; CHECK-STREAMING-COMPAT-NEXT: asr z2.s, z3.s, #31
+; CHECK-STREAMING-COMPAT-NEXT: asr z3.s, z4.s, #31
+; CHECK-STREAMING-COMPAT-NEXT: cmpne p1.s, p0/z, z2.s, #0
+; CHECK-STREAMING-COMPAT-NEXT: fmov s2, w8
+; CHECK-STREAMING-COMPAT-NEXT: cmpne p2.s, p0/z, z3.s, #0
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p0.s
+; CHECK-STREAMING-COMPAT-NEXT: cnt z2.s, p0/z, z2.s
+; CHECK-STREAMING-COMPAT-NEXT: cntp x9, p1, p1.s
+; CHECK-STREAMING-COMPAT-NEXT: compact z1.s, p1, z1.s
+; CHECK-STREAMING-COMPAT-NEXT: fmov w10, s2
+; CHECK-STREAMING-COMPAT-NEXT: cntp x8, p2, p2.s
+; CHECK-STREAMING-COMPAT-NEXT: compact z0.s, p2, z0.s
+; CHECK-STREAMING-COMPAT-NEXT: whilelo p0.s, xzr, x9
+; CHECK-STREAMING-COMPAT-NEXT: whilelo p1.s, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT: st1w { z1.s }, p0, [x0, x10, lsl #2]
+; CHECK-STREAMING-COMPAT-NEXT: st1w { z0.s }, p1, [x0]
+; CHECK-STREAMING-COMPAT-NEXT: add sp, sp, #16
+; CHECK-STREAMING-COMPAT-NEXT: ret
tail call void @llvm.masked.compressstore.v8i32(<8 x i32> %vec, ptr align 4 %p, <8 x i1> %mask)
ret void
}
@@ -275,6 +577,120 @@ define void @test_compressstore_v4i64(ptr %p, <4 x i64> %vec, <4 x i1> %mask) {
; CHECK-VL256-NEXT: whilelo p0.d, xzr, x8
; CHECK-VL256-NEXT: st1d { z0.d }, p0, [x0]
; CHECK-VL256-NEXT: ret
+;
+; CHECK-SME2p2-LABEL: test_compressstore_v4i64:
+; CHECK-SME2p2: // %bb.0:
+; CHECK-SME2p2-NEXT: ushll v2.4s, v2.4h, #0
+; CHECK-SME2p2-NEXT: index z4.s, #1, #1
+; CHECK-SME2p2-NEXT: ptrue p0.d, vl2
+; CHECK-SME2p2-NEXT: // kill: def $q1 killed $q1 def $z1
+; CHECK-SME2p2-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-SME2p2-NEXT: shl v3.2s, v2.2s, #31
+; CHECK-SME2p2-NEXT: ushll2 v5.2d, v2.4s, #0
+; CHECK-SME2p2-NEXT: ushll v2.2d, v2.2s, #0
+; CHECK-SME2p2-NEXT: cmlt v3.2s, v3.2s, #0
+; CHECK-SME2p2-NEXT: shl v2.2d, v2.2d, #63
+; CHECK-SME2p2-NEXT: and v3.8b, v3.8b, v4.8b
+; CHECK-SME2p2-NEXT: shl v4.2d, v5.2d, #63
+; CHECK-SME2p2-NEXT: cmpne p2.d, p0/z, z2.d, #0
+; CHECK-SME2p2-NEXT: addp v3.2s, v3.2s, v3.2s
+; CHECK-SME2p2-NEXT: cmpne p1.d, p0/z, z4.d, #0
+; CHECK-SME2p2-NEXT: ptrue p0.s
+; CHECK-SME2p2-NEXT: cntp x10, p2, p2.d
+; CHECK-SME2p2-NEXT: compact z0.d, p2, z0.d
+; CHECK-SME2p2-NEXT: fmov w8, s3
+; CHECK-SME2p2-NEXT: cntp x9, p1, p1.d
+; CHECK-SME2p2-NEXT: compact z1.d, p1, z1.d
+; CHECK-SME2p2-NEXT: whilelo p1.d, xzr, x10
+; CHECK-SME2p2-NEXT: and w8, w8, #0x3
+; CHECK-SME2p2-NEXT: fmov s2, w8
+; CHECK-SME2p2-NEXT: cnt z2.s, p0/z, z2.s
+; CHECK-SME2p2-NEXT: whilelo p0.d, xzr, x9
+; CHECK-SME2p2-NEXT: fmov w8, s2
+; CHECK-SME2p2-NEXT: st1d { z1.d }, p0, [x0, x8, lsl #3]
+; CHECK-SME2p2-NEXT: st1d { z0.d }, p1, [x0]
+; CHECK-SME2p2-NEXT: ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v4i64:
+; CHECK-SME2p2-STREAMING: // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT: sub sp, sp, #16
+; CHECK-SME2p2-STREAMING-NEXT: .cfi_def_cfa_offset 16
+; CHECK-SME2p2-STREAMING-NEXT: uunpklo z2.s, z2.h
+; CHECK-SME2p2-STREAMING-NEXT: index z5.s, #1, #1
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p0.s, vl2
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p1.d, vl2
+; CHECK-SME2p2-STREAMING-NEXT: movprfx z3, z2
+; CHECK-SME2p2-STREAMING-NEXT: ext z3.b, z3.b, z2.b, #8
+; CHECK-SME2p2-STREAMING-NEXT: lsl z4.s, z2.s, #31
+; CHECK-SME2p2-STREAMING-NEXT: uunpklo z2.d, z2.s
+; CHECK-SME2p2-STREAMING-NEXT: uunpklo z3.d, z3.s
+; CHECK-SME2p2-STREAMING-NEXT: asr z4.s, z4.s, #31
+; CHECK-SME2p2-STREAMING-NEXT: lsl z2.d, z2.d, #63
+; CHECK-SME2p2-STREAMING-NEXT: and z4.d, z4.d, z5.d
+; CHECK-SME2p2-STREAMING-NEXT: lsl z3.d, z3.d, #63
+; CHECK-SME2p2-STREAMING-NEXT: asr z2.d, z2.d, #63
+; CHECK-SME2p2-STREAMING-NEXT: uaddv d4, p0, z4.s
+; CHECK-SME2p2-STREAMING-NEXT: asr z3.d, z3.d, #63
+; CHECK-SME2p2-STREAMING-NEXT: cmpne p2.d, p1/z, z2.d, #0
+; CHECK-SME2p2-STREAMING-NEXT: cmpne p0.d, p1/z, z3.d, #0
+; CHECK-SME2p2-STREAMING-NEXT: str b4, [sp, #12]
+; CHECK-SME2p2-STREAMING-NEXT: ptrue p1.s
+; CHECK-SME2p2-STREAMING-NEXT: ldrb w8, [sp, #12]
+; CHECK-SME2p2-STREAMING-NEXT: cntp x10, p2, p2.d
+; CHECK-SME2p2-STREAMING-NEXT: compact z0.d, p2, z0.d
+; CHECK-SME2p2-STREAMING-NEXT: fmov s2, w8
+; CHECK-SME2p2-STREAMING-NEXT: cntp x9, p0, p0.d
+; CHECK-SME2p2-STREAMING-NEXT: compact z1.d, p0, z1.d
+; CHECK-SME2p2-STREAMING-NEXT: cnt z2.s, p1/z, z2.s
+; CHECK-SME2p2-STREAMING-NEXT: whilelo p1.d, xzr, x10
+; CHECK-SME2p2-STREAMING-NEXT: fmov w8, s2
+; CHECK-SME2p2-STREAMING-NEXT: whilelo p0.d, xzr, x9
+; CHECK-SME2p2-STREAMING-NEXT: st1d { z1.d }, p0, [x0, x8, lsl #3]
+; CHECK-SME2p2-STREAMING-NEXT: st1d { z0.d }, p1, [x0]
+; CHECK-SME2p2-STREAMING-NEXT: add sp, sp, #16
+; CHECK-SME2p2-STREAMING-NEXT: ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v4i64:
+; CHECK-STREAMING-COMPAT: // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT: sub sp, sp, #16
+; CHECK-STREAMING-COMPAT-NEXT: .cfi_def_cfa_offset 16
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $d2 killed $d2 def $z2
+; CHECK-STREAMING-COMPAT-NEXT: index z5.s, #1, #1
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p0.s, vl2
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q1 killed $q1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT: uunpklo z2.s, z2.h
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p1.d, vl2
+; CHECK-STREAMING-COMPAT-NEXT: movprfx z3, z2
+; CHECK-STREAMING-COMPAT-NEXT: ext z3.b, z3.b, z2.b, #8
+; CHECK-STREAMING-COMPAT-NEXT: lsl z4.s, z2.s, #31
+; CHECK-STREAMING-COMPAT-NEXT: uunpklo z2.d, z2.s
+; CHECK-STREAMING-COMPAT-NEXT: uunpklo z3.d, z3.s
+; CHECK-STREAMING-COMPAT-NEXT: asr z4.s, z4.s, #31
+; CHECK-STREAMING-COMPAT-NEXT: lsl z2.d, z2.d, #63
+; CHECK-STREAMING-COMPAT-NEXT: and z4.d, z4.d, z5.d
+; CHECK-STREAMING-COMPAT-NEXT: lsl z3.d, z3.d, #63
+; CHECK-STREAMING-COMPAT-NEXT: asr z2.d, z2.d, #63
+; CHECK-STREAMING-COMPAT-NEXT: uaddv d4, p0, z4.s
+; CHECK-STREAMING-COMPAT-NEXT: asr z3.d, z3.d, #63
+; CHECK-STREAMING-COMPAT-NEXT: cmpne p2.d, p1/z, z2.d, #0
+; CHECK-STREAMING-COMPAT-NEXT: cmpne p0.d, p1/z, z3.d, #0
+; CHECK-STREAMING-COMPAT-NEXT: str b4, [sp, #12]
+; CHECK-STREAMING-COMPAT-NEXT: ptrue p1.s
+; CHECK-STREAMING-COMPAT-NEXT: ldrb w8, [sp, #12]
+; CHECK-STREAMING-COMPAT-NEXT: cntp x10, p2, p2.d
+; CHECK-STREAMING-COMPAT-NEXT: compact z0.d, p2, z0.d
+; CHECK-STREAMING-COMPAT-NEXT: fmov s2, w8
+; CHECK-STREAMING-COMPAT-NEXT: cntp x9, p0, p0.d
+; CHECK-STREAMING-COMPAT-NEXT: compact z1.d, p0, z1.d
+; CHECK-STREAMING-COMPAT-NEXT: cnt z2.s, p1/z, z2.s
+; CHECK-STREAMING-COMPAT-NEXT: whilelo p1.d, xzr, x10
+; CHECK-STREAMING-COMPAT-NEXT: fmov w8, s2
+; CHECK-STREAMING-COMPAT-NEXT: whilelo p0.d, xzr, x9
+; CHECK-STREAMING-COMPAT-NEXT: st1d { z1.d }, p0, [x0, x8, lsl #3]
+; CHECK-STREAMING-COMPAT-NEXT: st1d { z0.d }, p1, [x0]
+; CHECK-STREAMING-COMPAT-NEXT: add sp, sp, #16
+; CHECK-STREAMING-COMPAT-NEXT: ret
tail call void @llvm.masked.compressstore.v4i64(<4 x i64> %vec, ptr align 8 %p, <4 x i1> %mask)
ret void
}
More information about the llvm-commits
mailing list