[llvm] [AArch64][SDAG] Enable +sme2p2/+sve2p2 lowering for compressstore (PR #218624)

Jack Styles via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 03:50:57 PDT 2026


https://github.com/Stylie777 updated https://github.com/llvm/llvm-project/pull/218624

>From cb617e2f9f6a33d7138787d374a2da7f0bffe9ba Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 24 Aug 2026 09:44:28 +0100
Subject: [PATCH 01/12] [AArch64][SDAG][NFC] Use CNTP intrinsic for MSTORE
 lowering

Currently, when lowering MSTORE the process is to Zero Extend,
VECREDUCE_ADD and then Zero Extend the result for types that are
not i64. This for i8/i16 types does not work as AArch64 cannot
zero extend these types to i64. Instead, use the CNTP intrinsic
to combine the values before storing them.
---
 llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 13 ++++---------
 1 file changed, 4 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 36a86241fe3b2..c1ca0f789e534 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -33922,16 +33922,11 @@ SDValue AArch64TargetLowering::LowerMSTORE(SDValue Op,
     return SDValue();
 
   EVT MaskVT = Store->getMask().getValueType();
-  EVT MaskExtVT = getPromotedVTForPredicate(MaskVT);
-  EVT MaskReduceVT = MaskExtVT.getScalarType();
   SDValue Zero = DAG.getConstant(0, DL, MVT::i64);
-
-  SDValue MaskExt =
-      DAG.getNode(ISD::ZERO_EXTEND, DL, MaskExtVT, Store->getMask());
-  SDValue CntActive =
-      DAG.getNode(ISD::VECREDUCE_ADD, DL, MaskReduceVT, MaskExt);
-  if (MaskReduceVT != MVT::i64)
-    CntActive = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i64, CntActive);
+  SDValue CntActive = DAG.getNode(
+      ISD::INTRINSIC_WO_CHAIN, DL, MVT::i64,
+      DAG.getTargetConstant(Intrinsic::aarch64_sve_cntp, DL, MVT::i64),
+      Store->getMask(), Store->getMask());
 
   SDValue CompressedValue =
       DAG.getNode(ISD::VECTOR_COMPRESS, DL, VT, Store->getValue(),

>From 785b0aa4f2e22c887458160186ad84bcae3befa0 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 20 Aug 2026 13:52:13 +0100
Subject: [PATCH 02/12] [AArch64][SDAG] Enable +sme2p2/+sve2p2 lowering for
 MSTORE

Following on from #215219 it will now be possible to lower
llvm.masked.compressstore for sve2p2/sme2p2 using compact
for i8/i16/f16/bf16 types.
---
 .../Target/AArch64/AArch64ISelLowering.cpp    |  6 +-
 .../AArch64/AArch64TargetTransformInfo.h      |  5 ++
 .../CostModel/AArch64/masked_compress_load.ll | 74 +++++++++----------
 .../sve-masked-compressstore-sve2p2.ll        | 48 ++++++++++--
 4 files changed, 88 insertions(+), 45 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index c1ca0f789e534..6a08bf6bc0e38 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2283,6 +2283,11 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
       // lowering for non-compressing masked stores.
       setOperationAction(ISD::MSTORE, VT, Custom);
     }
+    if (Subtarget->hasSVE2p2() || Subtarget->hasSME2p2()) {
+      // With +sve2p2/+sme2p2 the full range of vector types are supported.
+      for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
+        setOperationAction(ISD::MSTORE, VT, Custom);
+    }
 
     // Histcnt is SVE2 only
     if (Subtarget->hasSVE2()) {
@@ -33927,7 +33932,6 @@ SDValue AArch64TargetLowering::LowerMSTORE(SDValue Op,
       ISD::INTRINSIC_WO_CHAIN, DL, MVT::i64,
       DAG.getTargetConstant(Intrinsic::aarch64_sve_cntp, DL, MVT::i64),
       Store->getMask(), Store->getMask());
-
   SDValue CompressedValue =
       DAG.getNode(ISD::VECTOR_COMPRESS, DL, VT, Store->getValue(),
                   Store->getMask(), DAG.getPOISON(VT));
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index 19de90427cb74..64337ff4d5db7 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -334,6 +334,11 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
   }
 
   bool isElementTypeLegalForCompressStore(Type *Ty) const {
+    if ((ST->hasSVE2p2() || ST->hasSME2p2()) &&
+        ((Ty->isIntegerTy(8) || Ty->isIntegerTy(16)) ||
+         ((Ty->isHalfTy() || Ty->isBFloatTy()) &&
+          Ty->getScalarSizeInBits() == 16)))
+      return true;
     return Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
            Ty->isIntegerTy(64);
   }
diff --git a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
index cb59f1016a4d1..7cc880ce5b66d 100644
--- a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
@@ -95,21 +95,21 @@ define void @fixed() {
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i8.p0(<2 x i8> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i8.p0(<4 x i8> poison, ptr poison, <4 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8i8.p0(<8 x i8> poison, ptr poison, <8 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v16i8.p0(<16 x i8> poison, ptr poison, <16 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v16i8.p0(<16 x i8> poison, ptr poison, <16 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i16.p0(<2 x i16> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i16.p0(<4 x i16> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8i16.p0(<8 x i16> poison, ptr poison, <8 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8i16.p0(<8 x i16> poison, ptr poison, <8 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i32.p0(<2 x i32> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v4i32.p0(<4 x i32> poison, ptr poison, <4 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v2i64.p0(<2 x i64> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f16.p0(<2 x half> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4f16.p0(<4 x half> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8f16.p0(<8 x half> poison, ptr poison, <8 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8f16.p0(<8 x half> poison, ptr poison, <8 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f32.p0(<2 x float> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v4f32.p0(<4 x float> poison, ptr poison, <4 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v2f64.p0(<2 x double> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v4i64.p0(<4 x i64> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v32f16.p0(<32 x half> poison, ptr poison, <32 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.v32f16.p0(<32 x half> poison, ptr poison, <32 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 entry:
@@ -142,25 +142,25 @@ entry:
 
 define void @scalable() {
 ; SVE2p2-SME2p2-NON-STREAMING-LABEL: 'scalable'
-; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
 ; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
 ; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
 ; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
 ; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv1i64.p0(<vscale x 1 x i64> poison, ptr poison, <vscale x 1 x i1> poison)
 ; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
+; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
 ; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 ; SVE2p2-SME2p2-STREAMING-LABEL: 'scalable'
@@ -186,25 +186,25 @@ define void @scalable() {
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 ; SVE2p2-SME2p2-LABEL: 'scalable'
-; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
-; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
+; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
 ; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
 ; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
 ; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
 ; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv1i64.p0(<vscale x 1 x i64> poison, ptr poison, <vscale x 1 x i1> poison)
 ; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
+; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
 ; SVE2p2-SME2p2-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 ; SVE-ONLY-LABEL: 'scalable'
@@ -230,25 +230,25 @@ define void @scalable() {
 ; SVE-ONLY-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 ; SVE2p2-SME2p2-SVE256-LABEL: 'scalable'
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv1i64.p0(<vscale x 1 x i64> poison, ptr poison, <vscale x 1 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
+; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
 ; SVE2p2-SME2p2-SVE256-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 entry:
diff --git a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
index 92ecc3c83e2c5..5655d53b5ddaa 100644
--- a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
+++ b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
@@ -1,17 +1,51 @@
-; RUN: llc -mtriple=aarch64 -mattr=+sve2p2 < %s
-
-;; These masked.compressstore operations could be natively supported with +sve2p2
-;; (or by promoting to 32/64 bit elements + a truncstore), but currently are not
-;; supported.
-
-; XFAIL: *
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=aarch64 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BASE --allow-unused-prefixes
+; RUN: llc -mtriple=aarch64 -aarch64-sve-vector-bits-min=256 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-VL256 --allow-unused-prefixes
 
 define void @test_compressstore_nxv8i16(ptr %p, <vscale x 8 x i16> %vec, <vscale x 8 x i1> %mask) {
+; CHECK-LABEL: test_compressstore_nxv8i16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    cntp x8, p0, p0.h
+; CHECK-NEXT:    compact z0.h, p0, z0.h
+; CHECK-NEXT:    whilelo p1.h, xzr, x8
+; CHECK-NEXT:    st1h { z0.h }, p1, [x0]
+; CHECK-NEXT:    ret
   tail call void @llvm.masked.compressstore.nxv8i16(<vscale x 8 x i16> %vec, ptr align 2 %p, <vscale x 8 x i1> %mask)
   ret void
 }
 
 define void @test_compressstore_nxv16i8(ptr %p, <vscale x 16 x i8> %vec, <vscale x 16 x i1> %mask) {
+; CHECK-LABEL: test_compressstore_nxv16i8:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    cntp x8, p0, p0.b
+; CHECK-NEXT:    compact z0.b, p0, z0.b
+; CHECK-NEXT:    whilelo p1.b, xzr, x8
+; CHECK-NEXT:    st1b { z0.b }, p1, [x0]
+; CHECK-NEXT:    ret
   tail call void @llvm.masked.compressstore.nxv16i8(<vscale x 16 x i8> %vec, ptr align 1 %p, <vscale x 16 x i1> %mask)
   ret void
 }
+
+define void @test_compressstore_nxv8f16(ptr %p, <vscale x 8 x half> %vec, <vscale x 8 x i1> %mask) {
+; CHECK-LABEL: test_compressstore_nxv8f16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    cntp x8, p0, p0.h
+; CHECK-NEXT:    compact z0.h, p0, z0.h
+; CHECK-NEXT:    whilelo p1.h, xzr, x8
+; CHECK-NEXT:    st1h { z0.h }, p1, [x0]
+; CHECK-NEXT:    ret
+  tail call void @llvm.masked.compressstore.nxv8f16(<vscale x 8 x half> %vec, ptr align 1 %p, <vscale x 8 x i1> %mask)
+  ret void
+}
+
+define void @test_compressstore_nxv8bf16(ptr %p, <vscale x 8 x bfloat> %vec, <vscale x 8 x i1> %mask) {
+; CHECK-LABEL: test_compressstore_nxv8bf16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    cntp x8, p0, p0.h
+; CHECK-NEXT:    compact z0.h, p0, z0.h
+; CHECK-NEXT:    whilelo p1.h, xzr, x8
+; CHECK-NEXT:    st1h { z0.h }, p1, [x0]
+; CHECK-NEXT:    ret
+  tail call void @llvm.masked.compressstore.nxv8bf16(<vscale x 8 x bfloat> %vec, ptr align 1 %p, <vscale x 8 x i1> %mask)
+  ret void
+}

>From f8a722e61b99837d18fda72e3460662e8d5f8292 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 24 Aug 2026 10:03:30 +0100
Subject: [PATCH 03/12] Remove whitespace change

---
 llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 1 +
 1 file changed, 1 insertion(+)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 6a08bf6bc0e38..07b28395c0dfe 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -33932,6 +33932,7 @@ SDValue AArch64TargetLowering::LowerMSTORE(SDValue Op,
       ISD::INTRINSIC_WO_CHAIN, DL, MVT::i64,
       DAG.getTargetConstant(Intrinsic::aarch64_sve_cntp, DL, MVT::i64),
       Store->getMask(), Store->getMask());
+
   SDValue CompressedValue =
       DAG.getNode(ISD::VECTOR_COMPRESS, DL, VT, Store->getValue(),
                   Store->getMask(), DAG.getPOISON(VT));

>From 8e09f73faf8315ede14485f2797935482a9a3452 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 24 Aug 2026 10:48:06 +0100
Subject: [PATCH 04/12] Enable Streaming SME2p2

---
 llvm/lib/Target/AArch64/AArch64ISelLowering.cpp       | 11 ++++++-----
 .../AArch64/sve-masked-compressstore-sve2p2.ll        |  3 +++
 2 files changed, 9 insertions(+), 5 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 07b28395c0dfe..639d5b079f734 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2283,11 +2283,6 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
       // lowering for non-compressing masked stores.
       setOperationAction(ISD::MSTORE, VT, Custom);
     }
-    if (Subtarget->hasSVE2p2() || Subtarget->hasSME2p2()) {
-      // With +sve2p2/+sme2p2 the full range of vector types are supported.
-      for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
-        setOperationAction(ISD::MSTORE, VT, Custom);
-    }
 
     // Histcnt is SVE2 only
     if (Subtarget->hasSVE2()) {
@@ -2308,6 +2303,12 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
     }
   }
 
+  if ((Subtarget->isSVEAvailable() && Subtarget->hasSVE2p2()) || (Subtarget->isSVEorStreamingSVEAvailable() && Subtarget->hasSME2p2())) {
+    // With +sve2p2/+sme2p2 the full range of vector types are supported.
+    for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
+      setOperationAction(ISD::MSTORE, VT, Custom);
+  }
+
   if (Subtarget->hasMOPS() && Subtarget->hasMTE()) {
     // Only required for llvm.aarch64.mops.memset.tag
     setOperationAction(ISD::INTRINSIC_W_CHAIN, MVT::i8, Custom);
diff --git a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
index 5655d53b5ddaa..4fca173a712ba 100644
--- a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
+++ b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
@@ -1,6 +1,9 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
 ; RUN: llc -mtriple=aarch64 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BASE --allow-unused-prefixes
 ; RUN: llc -mtriple=aarch64 -aarch64-sve-vector-bits-min=256 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-VL256 --allow-unused-prefixes
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve,+sme2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SVE2p2 --allow-unused-prefixes
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2 -force-streaming < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SME2p2-STREAMING --allow-unused-prefixes
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2,+sve2p2 -force-streaming-compatible < %s | FileCheck %s --check-prefixes=CHECK,CHECK-STREAMING-COMPAT --allow-unused-prefixes
 
 define void @test_compressstore_nxv8i16(ptr %p, <vscale x 8 x i16> %vec, <vscale x 8 x i1> %mask) {
 ; CHECK-LABEL: test_compressstore_nxv8i16:

>From 3865d30d79af0986c28e23b2f713e22d70ebd905 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 24 Aug 2026 10:50:40 +0100
Subject: [PATCH 05/12] format

---
 llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 3 ++-
 1 file changed, 2 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 639d5b079f734..a0fb5365cd843 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2303,7 +2303,8 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
     }
   }
 
-  if ((Subtarget->isSVEAvailable() && Subtarget->hasSVE2p2()) || (Subtarget->isSVEorStreamingSVEAvailable() && Subtarget->hasSME2p2())) {
+  if ((Subtarget->isSVEAvailable() && Subtarget->hasSVE2p2()) ||
+      (Subtarget->isSVEorStreamingSVEAvailable() && Subtarget->hasSME2p2())) {
     // With +sve2p2/+sme2p2 the full range of vector types are supported.
     for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
       setOperationAction(ISD::MSTORE, VT, Custom);

>From 929e6c0139c46fea7227a2185de5a0ef4987ba73 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 26 Aug 2026 14:38:12 +0100
Subject: [PATCH 06/12] Respond to review comments

---
 .../Target/AArch64/AArch64ISelLowering.cpp    |  11 +-
 .../AArch64/AArch64TargetTransformInfo.h      |   8 +-
 .../sve-masked-compressstore-sve2p2.ll        | 289 +++++++++++++++++-
 3 files changed, 290 insertions(+), 18 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 555a3582c37ee..cdbf7fed84e10 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2227,8 +2227,10 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
 
     if (Subtarget->hasSVE2p2() || Subtarget->hasSME2p2()) {
       // With +sve2p2/+sme2p2 the full range of vector types are supported.
-      for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
+      for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16}) {
+        setOperationAction(ISD::MSTORE, VT, Custom);
         setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
+      }
 
       for (auto VT : {MVT::v8i8, MVT::v16i8, MVT::v4i16, MVT::v8i16, MVT::v4f16,
                       MVT::v8f16, MVT::v4bf16, MVT::v8bf16})
@@ -2303,13 +2305,6 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
     }
   }
 
-  if ((Subtarget->isSVEAvailable() && Subtarget->hasSVE2p2()) ||
-      (Subtarget->isSVEorStreamingSVEAvailable() && Subtarget->hasSME2p2())) {
-    // With +sve2p2/+sme2p2 the full range of vector types are supported.
-    for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16})
-      setOperationAction(ISD::MSTORE, VT, Custom);
-  }
-
   if (Subtarget->hasMOPS() && Subtarget->hasMTE()) {
     // Only required for llvm.aarch64.mops.memset.tag
     setOperationAction(ISD::INTRINSIC_W_CHAIN, MVT::i8, Custom);
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index a6cb5bdebcfab..5c7b2703b4257 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -334,10 +334,8 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
   }
 
   bool isElementTypeLegalForCompressStore(Type *Ty) const {
-    if ((ST->hasSVE2p2() || ST->hasSME2p2()) &&
-        ((Ty->isIntegerTy(8) || Ty->isIntegerTy(16)) ||
-         ((Ty->isHalfTy() || Ty->isBFloatTy()) &&
-          Ty->getScalarSizeInBits() == 16)))
+    if ((ST->hasSVE2p2() || (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())) &&
+        (Ty->isIntegerTy(8) || Ty->isIntegerTy(16) || Ty->getScalarSizeInBits() == 16))
       return true;
     return Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
            Ty->isIntegerTy(64);
@@ -345,7 +343,7 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
 
   bool isLegalMaskedCompressStore(Type *DataType,
                                   Align Alignment) const override {
-    if (!ST->isSVEAvailable())
+    if (!ST->isSVEorStreamingSVEAvailable())
       return false;
 
     if (isa<FixedVectorType>(DataType) &&
diff --git a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
index 4fca173a712ba..42cef84285a0f 100644
--- a/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
+++ b/llvm/test/CodeGen/AArch64/sve-masked-compressstore-sve2p2.ll
@@ -1,9 +1,9 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -mtriple=aarch64 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BASE --allow-unused-prefixes
-; RUN: llc -mtriple=aarch64 -aarch64-sve-vector-bits-min=256 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-VL256 --allow-unused-prefixes
-; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve,+sme2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SVE2p2 --allow-unused-prefixes
-; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2 -force-streaming < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SME2p2-STREAMING --allow-unused-prefixes
-; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2,+sve2p2 -force-streaming-compatible < %s | FileCheck %s --check-prefixes=CHECK,CHECK-STREAMING-COMPAT --allow-unused-prefixes
+; RUN: llc -mtriple=aarch64 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BASE
+; RUN: llc -mtriple=aarch64 -aarch64-sve-vector-bits-min=256 -mattr=+sve2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-VL256
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve,+sme2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SVE2p2
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2 -force-streaming < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SME2p2-STREAMING
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2,+sve2p2 -force-streaming-compatible < %s | FileCheck %s --check-prefixes=CHECK,CHECK-STREAMING-COMPAT
 
 define void @test_compressstore_nxv8i16(ptr %p, <vscale x 8 x i16> %vec, <vscale x 8 x i1> %mask) {
 ; CHECK-LABEL: test_compressstore_nxv8i16:
@@ -52,3 +52,282 @@ define void @test_compressstore_nxv8bf16(ptr %p, <vscale x 8 x bfloat> %vec, <vs
   tail call void @llvm.masked.compressstore.nxv8bf16(<vscale x 8 x bfloat> %vec, ptr align 1 %p, <vscale x 8 x i1> %mask)
   ret void
 }
+
+define void @test_compressstore_v8i16(ptr %p, <8 x i16> %vec, <8 x i1> %mask) {
+; CHECK-BASE-LABEL: test_compressstore_v8i16:
+; CHECK-BASE:       // %bb.0:
+; CHECK-BASE-NEXT:    ushll v1.8h, v1.8b, #0
+; CHECK-BASE-NEXT:    ptrue p0.h, vl8
+; CHECK-BASE-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT:    shl v1.8h, v1.8h, #15
+; CHECK-BASE-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-BASE-NEXT:    cntp x8, p1, p1.h
+; CHECK-BASE-NEXT:    compact z0.h, p1, z0.h
+; CHECK-BASE-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-BASE-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-BASE-NEXT:    ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v8i16:
+; CHECK-VL256:       // %bb.0:
+; CHECK-VL256-NEXT:    ushll v1.8h, v1.8b, #0
+; CHECK-VL256-NEXT:    ptrue p0.h, vl8
+; CHECK-VL256-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT:    shl v1.8h, v1.8h, #15
+; CHECK-VL256-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-VL256-NEXT:    cntp x8, p1, p1.h
+; CHECK-VL256-NEXT:    compact z0.h, p1, z0.h
+; CHECK-VL256-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-VL256-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-VL256-NEXT:    ret
+;
+; CHECK-SVE2p2-LABEL: test_compressstore_v8i16:
+; CHECK-SVE2p2:       // %bb.0:
+; CHECK-SVE2p2-NEXT:    ushll v1.8h, v1.8b, #0
+; CHECK-SVE2p2-NEXT:    ptrue p0.h, vl8
+; CHECK-SVE2p2-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-SVE2p2-NEXT:    shl v1.8h, v1.8h, #15
+; CHECK-SVE2p2-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SVE2p2-NEXT:    cntp x8, p1, p1.h
+; CHECK-SVE2p2-NEXT:    compact z0.h, p1, z0.h
+; CHECK-SVE2p2-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-SVE2p2-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-SVE2p2-NEXT:    ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v8i16:
+; CHECK-SME2p2-STREAMING:       // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT:    uunpklo z1.h, z1.b
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p0.h, vl8
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT:    asr z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SME2p2-STREAMING-NEXT:    cntp x8, p1, p1.h
+; CHECK-SME2p2-STREAMING-NEXT:    compact z0.h, p1, z0.h
+; CHECK-SME2p2-STREAMING-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT:    ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v8i16:
+; CHECK-STREAMING-COMPAT:       // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p0.h, vl8
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT:    uunpklo z1.h, z1.b
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT:    asr z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-STREAMING-COMPAT-NEXT:    cntp x8, p1, p1.h
+; CHECK-STREAMING-COMPAT-NEXT:    compact z0.h, p1, z0.h
+; CHECK-STREAMING-COMPAT-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT:    ret
+  tail call void @llvm.masked.compressstore.v8i16(<8 x i16> %vec, ptr align 2 %p, <8 x i1> %mask)
+  ret void
+}
+
+define void @test_compressstore_v16i8(ptr %p, <16 x i8> %vec, <16 x i1> %mask) {
+; CHECK-BASE-LABEL: test_compressstore_v16i8:
+; CHECK-BASE:       // %bb.0:
+; CHECK-BASE-NEXT:    shl v1.16b, v1.16b, #7
+; CHECK-BASE-NEXT:    ptrue p0.b, vl16
+; CHECK-BASE-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT:    cmpne p1.b, p0/z, z1.b, #0
+; CHECK-BASE-NEXT:    cntp x8, p1, p1.b
+; CHECK-BASE-NEXT:    compact z0.b, p1, z0.b
+; CHECK-BASE-NEXT:    whilelo p0.b, xzr, x8
+; CHECK-BASE-NEXT:    st1b { z0.b }, p0, [x0]
+; CHECK-BASE-NEXT:    ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v16i8:
+; CHECK-VL256:       // %bb.0:
+; CHECK-VL256-NEXT:    shl v1.16b, v1.16b, #7
+; CHECK-VL256-NEXT:    ptrue p0.b, vl16
+; CHECK-VL256-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT:    cmpne p1.b, p0/z, z1.b, #0
+; CHECK-VL256-NEXT:    cntp x8, p1, p1.b
+; CHECK-VL256-NEXT:    compact z0.b, p1, z0.b
+; CHECK-VL256-NEXT:    whilelo p0.b, xzr, x8
+; CHECK-VL256-NEXT:    st1b { z0.b }, p0, [x0]
+; CHECK-VL256-NEXT:    ret
+;
+; CHECK-SVE2p2-LABEL: test_compressstore_v16i8:
+; CHECK-SVE2p2:       // %bb.0:
+; CHECK-SVE2p2-NEXT:    shl v1.16b, v1.16b, #7
+; CHECK-SVE2p2-NEXT:    ptrue p0.b, vl16
+; CHECK-SVE2p2-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-SVE2p2-NEXT:    cmpne p1.b, p0/z, z1.b, #0
+; CHECK-SVE2p2-NEXT:    cntp x8, p1, p1.b
+; CHECK-SVE2p2-NEXT:    compact z0.b, p1, z0.b
+; CHECK-SVE2p2-NEXT:    whilelo p0.b, xzr, x8
+; CHECK-SVE2p2-NEXT:    st1b { z0.b }, p0, [x0]
+; CHECK-SVE2p2-NEXT:    ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v16i8:
+; CHECK-SME2p2-STREAMING:       // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z1.b, z1.b, #7
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p0.b, vl16
+; CHECK-SME2p2-STREAMING-NEXT:    asr z1.b, z1.b, #7
+; CHECK-SME2p2-STREAMING-NEXT:    cmpne p1.b, p0/z, z1.b, #0
+; CHECK-SME2p2-STREAMING-NEXT:    cntp x8, p1, p1.b
+; CHECK-SME2p2-STREAMING-NEXT:    compact z0.b, p1, z0.b
+; CHECK-SME2p2-STREAMING-NEXT:    whilelo p0.b, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT:    st1b { z0.b }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT:    ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v16i8:
+; CHECK-STREAMING-COMPAT:       // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q1 killed $q1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p0.b, vl16
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z1.b, z1.b, #7
+; CHECK-STREAMING-COMPAT-NEXT:    asr z1.b, z1.b, #7
+; CHECK-STREAMING-COMPAT-NEXT:    cmpne p1.b, p0/z, z1.b, #0
+; CHECK-STREAMING-COMPAT-NEXT:    cntp x8, p1, p1.b
+; CHECK-STREAMING-COMPAT-NEXT:    compact z0.b, p1, z0.b
+; CHECK-STREAMING-COMPAT-NEXT:    whilelo p0.b, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT:    st1b { z0.b }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT:    ret
+  tail call void @llvm.masked.compressstore.v16i8(<16 x i8> %vec, ptr align 1 %p, <16 x i1> %mask)
+  ret void
+}
+
+define void @test_compressstore_v8f16(ptr %p, <8 x half> %vec, <8 x i1> %mask) {
+; CHECK-BASE-LABEL: test_compressstore_v8f16:
+; CHECK-BASE:       // %bb.0:
+; CHECK-BASE-NEXT:    ushll v1.8h, v1.8b, #0
+; CHECK-BASE-NEXT:    ptrue p0.h, vl8
+; CHECK-BASE-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT:    shl v1.8h, v1.8h, #15
+; CHECK-BASE-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-BASE-NEXT:    cntp x8, p1, p1.h
+; CHECK-BASE-NEXT:    compact z0.h, p1, z0.h
+; CHECK-BASE-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-BASE-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-BASE-NEXT:    ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v8f16:
+; CHECK-VL256:       // %bb.0:
+; CHECK-VL256-NEXT:    ushll v1.8h, v1.8b, #0
+; CHECK-VL256-NEXT:    ptrue p0.h, vl8
+; CHECK-VL256-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT:    shl v1.8h, v1.8h, #15
+; CHECK-VL256-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-VL256-NEXT:    cntp x8, p1, p1.h
+; CHECK-VL256-NEXT:    compact z0.h, p1, z0.h
+; CHECK-VL256-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-VL256-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-VL256-NEXT:    ret
+;
+; CHECK-SVE2p2-LABEL: test_compressstore_v8f16:
+; CHECK-SVE2p2:       // %bb.0:
+; CHECK-SVE2p2-NEXT:    ushll v1.8h, v1.8b, #0
+; CHECK-SVE2p2-NEXT:    ptrue p0.h, vl8
+; CHECK-SVE2p2-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-SVE2p2-NEXT:    shl v1.8h, v1.8h, #15
+; CHECK-SVE2p2-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SVE2p2-NEXT:    cntp x8, p1, p1.h
+; CHECK-SVE2p2-NEXT:    compact z0.h, p1, z0.h
+; CHECK-SVE2p2-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-SVE2p2-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-SVE2p2-NEXT:    ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v8f16:
+; CHECK-SME2p2-STREAMING:       // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT:    uunpklo z1.h, z1.b
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p0.h, vl8
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT:    asr z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SME2p2-STREAMING-NEXT:    cntp x8, p1, p1.h
+; CHECK-SME2p2-STREAMING-NEXT:    compact z0.h, p1, z0.h
+; CHECK-SME2p2-STREAMING-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT:    ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v8f16:
+; CHECK-STREAMING-COMPAT:       // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p0.h, vl8
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT:    uunpklo z1.h, z1.b
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT:    asr z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-STREAMING-COMPAT-NEXT:    cntp x8, p1, p1.h
+; CHECK-STREAMING-COMPAT-NEXT:    compact z0.h, p1, z0.h
+; CHECK-STREAMING-COMPAT-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT:    ret
+  tail call void @llvm.masked.compressstore.v8f16(<8 x half> %vec, ptr align 1 %p, <8 x i1> %mask)
+  ret void
+}
+
+define void @test_compressstore_v8bf16(ptr %p, <8 x bfloat> %vec, <8 x i1> %mask) {
+; CHECK-BASE-LABEL: test_compressstore_v8bf16:
+; CHECK-BASE:       // %bb.0:
+; CHECK-BASE-NEXT:    ushll v1.8h, v1.8b, #0
+; CHECK-BASE-NEXT:    ptrue p0.h, vl8
+; CHECK-BASE-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT:    shl v1.8h, v1.8h, #15
+; CHECK-BASE-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-BASE-NEXT:    cntp x8, p1, p1.h
+; CHECK-BASE-NEXT:    compact z0.h, p1, z0.h
+; CHECK-BASE-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-BASE-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-BASE-NEXT:    ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v8bf16:
+; CHECK-VL256:       // %bb.0:
+; CHECK-VL256-NEXT:    ushll v1.8h, v1.8b, #0
+; CHECK-VL256-NEXT:    ptrue p0.h, vl8
+; CHECK-VL256-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT:    shl v1.8h, v1.8h, #15
+; CHECK-VL256-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-VL256-NEXT:    cntp x8, p1, p1.h
+; CHECK-VL256-NEXT:    compact z0.h, p1, z0.h
+; CHECK-VL256-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-VL256-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-VL256-NEXT:    ret
+;
+; CHECK-SVE2p2-LABEL: test_compressstore_v8bf16:
+; CHECK-SVE2p2:       // %bb.0:
+; CHECK-SVE2p2-NEXT:    ushll v1.8h, v1.8b, #0
+; CHECK-SVE2p2-NEXT:    ptrue p0.h, vl8
+; CHECK-SVE2p2-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-SVE2p2-NEXT:    shl v1.8h, v1.8h, #15
+; CHECK-SVE2p2-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SVE2p2-NEXT:    cntp x8, p1, p1.h
+; CHECK-SVE2p2-NEXT:    compact z0.h, p1, z0.h
+; CHECK-SVE2p2-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-SVE2p2-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-SVE2p2-NEXT:    ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v8bf16:
+; CHECK-SME2p2-STREAMING:       // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT:    uunpklo z1.h, z1.b
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p0.h, vl8
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT:    asr z1.h, z1.h, #15
+; CHECK-SME2p2-STREAMING-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-SME2p2-STREAMING-NEXT:    cntp x8, p1, p1.h
+; CHECK-SME2p2-STREAMING-NEXT:    compact z0.h, p1, z0.h
+; CHECK-SME2p2-STREAMING-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT:    ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v8bf16:
+; CHECK-STREAMING-COMPAT:       // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p0.h, vl8
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT:    uunpklo z1.h, z1.b
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT:    asr z1.h, z1.h, #15
+; CHECK-STREAMING-COMPAT-NEXT:    cmpne p1.h, p0/z, z1.h, #0
+; CHECK-STREAMING-COMPAT-NEXT:    cntp x8, p1, p1.h
+; CHECK-STREAMING-COMPAT-NEXT:    compact z0.h, p1, z0.h
+; CHECK-STREAMING-COMPAT-NEXT:    whilelo p0.h, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT:    st1h { z0.h }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT:    ret
+  tail call void @llvm.masked.compressstore.v8bf16(<8 x bfloat> %vec, ptr align 1 %p, <8 x i1> %mask)
+  ret void
+}

>From e692e9928804b12a995bc31772aabd201a0dbb95 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 26 Aug 2026 14:45:32 +0100
Subject: [PATCH 07/12] Simplify Target Hook Check & Format

---
 llvm/lib/Target/AArch64/AArch64ISelLowering.cpp      | 3 ++-
 llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h | 5 +++--
 2 files changed, 5 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index cdbf7fed84e10..4439d6b826ebb 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2227,7 +2227,8 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
 
     if (Subtarget->hasSVE2p2() || Subtarget->hasSME2p2()) {
       // With +sve2p2/+sme2p2 the full range of vector types are supported.
-      for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16}) {
+      for (auto VT :
+           {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16}) {
         setOperationAction(ISD::MSTORE, VT, Custom);
         setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
       }
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index 5c7b2703b4257..71878619b6570 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -334,8 +334,9 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
   }
 
   bool isElementTypeLegalForCompressStore(Type *Ty) const {
-    if ((ST->hasSVE2p2() || (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())) &&
-        (Ty->isIntegerTy(8) || Ty->isIntegerTy(16) || Ty->getScalarSizeInBits() == 16))
+    if ((ST->hasSVE2p2() ||
+         (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())) &&
+        (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16))
       return true;
     return Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
            Ty->isIntegerTy(64);

>From 8c824d4675bfc1aa37ecaf127e1978d405b66b71 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 27 Aug 2026 10:05:30 +0100
Subject: [PATCH 08/12] Allow i8/i16 types in SME2p2 Streaming Mode

---
 .../AArch64/AArch64TargetTransformInfo.h      | 12 ++++----
 .../CostModel/AArch64/masked_compress_load.ll | 30 +++++++++----------
 2 files changed, 21 insertions(+), 21 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index 71878619b6570..e11b4488d3e60 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -334,17 +334,17 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
   }
 
   bool isElementTypeLegalForCompressStore(Type *Ty) const {
-    if ((ST->hasSVE2p2() ||
-         (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())) &&
-        (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16))
-      return true;
-    return Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
+    // For streaming SME2p2, only consider 8bit or 16bit Scalar types.
+    if ((ST->isStreamingSVEAvailable() && ST->hasSME2p2()))
+      return Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16;
+
+    return ((ST->hasSVE2p2() || ST->hasSME2p2()) && (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16)) || Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
            Ty->isIntegerTy(64);
   }
 
   bool isLegalMaskedCompressStore(Type *DataType,
                                   Align Alignment) const override {
-    if (!ST->isSVEorStreamingSVEAvailable())
+    if (!(ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())))
       return false;
 
     if (isa<FixedVectorType>(DataType) &&
diff --git a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
index 7cc880ce5b66d..69a3edb499b31 100644
--- a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
@@ -32,21 +32,21 @@ define void @fixed() {
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i8.p0(<2 x i8> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i8.p0(<4 x i8> poison, ptr poison, <4 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8i8.p0(<8 x i8> poison, ptr poison, <8 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v16i8.p0(<16 x i8> poison, ptr poison, <16 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v16i8.p0(<16 x i8> poison, ptr poison, <16 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i16.p0(<2 x i16> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i16.p0(<4 x i16> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8i16.p0(<8 x i16> poison, ptr poison, <8 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8i16.p0(<8 x i16> poison, ptr poison, <8 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i32.p0(<2 x i32> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i32.p0(<4 x i32> poison, ptr poison, <4 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i64.p0(<2 x i64> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f16.p0(<2 x half> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4f16.p0(<4 x half> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v8f16.p0(<8 x half> poison, ptr poison, <8 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8f16.p0(<8 x half> poison, ptr poison, <8 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f32.p0(<2 x float> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4f32.p0(<4 x float> poison, ptr poison, <4 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f64.p0(<2 x double> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i64.p0(<4 x i64> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v32f16.p0(<32 x half> poison, ptr poison, <32 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.v32f16.p0(<32 x half> poison, ptr poison, <32 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 ; SVE2p2-SME2p2-LABEL: 'fixed'
@@ -164,25 +164,25 @@ define void @scalable() {
 ; SVE2p2-SME2p2-NON-STREAMING-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 ; SVE2p2-SME2p2-STREAMING-LABEL: 'scalable'
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i8.p0(<vscale x 2 x i8> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i8.p0(<vscale x 4 x i8> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i8.p0(<vscale x 8 x i8> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv16i8.p0(<vscale x 16 x i8> poison, ptr poison, <vscale x 16 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv1i64.p0(<vscale x 1 x i64> poison, ptr poison, <vscale x 1 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 ; SVE2p2-SME2p2-LABEL: 'scalable'

>From 3248b1571b3c524a277e2c329220a7da8c7fde18 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 27 Aug 2026 10:06:36 +0100
Subject: [PATCH 09/12] format

---
 llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h | 7 +++++--
 1 file changed, 5 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index e11b4488d3e60..2420e01eb3b3d 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -338,13 +338,16 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
     if ((ST->isStreamingSVEAvailable() && ST->hasSME2p2()))
       return Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16;
 
-    return ((ST->hasSVE2p2() || ST->hasSME2p2()) && (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16)) || Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
+    return ((ST->hasSVE2p2() || ST->hasSME2p2()) &&
+            (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16)) ||
+           Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
            Ty->isIntegerTy(64);
   }
 
   bool isLegalMaskedCompressStore(Type *DataType,
                                   Align Alignment) const override {
-    if (!(ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())))
+    if (!(ST->isSVEAvailable() ||
+          (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())))
       return false;
 
     if (isa<FixedVectorType>(DataType) &&

>From 20e2f7cb64997577a71d8d4919e476ca411e9b24 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 27 Aug 2026 10:09:27 +0100
Subject: [PATCH 10/12] Add lowering comment

---
 llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 1 +
 1 file changed, 1 insertion(+)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 4439d6b826ebb..8f68e4af04207 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2229,6 +2229,7 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
       // With +sve2p2/+sme2p2 the full range of vector types are supported.
       for (auto VT :
            {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16}) {
+        // Use custom lowering for MSTORE so we can handle compressstore (using VECTOR_COMPRESS).
         setOperationAction(ISD::MSTORE, VT, Custom);
         setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
       }

>From 4550b93b425614f5903fb8c624fab1cdbae3401f Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 27 Aug 2026 10:10:33 +0100
Subject: [PATCH 11/12] format

---
 llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 3 ++-
 1 file changed, 2 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 8f68e4af04207..83c07a19a55e8 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2229,7 +2229,8 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
       // With +sve2p2/+sme2p2 the full range of vector types are supported.
       for (auto VT :
            {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv8f16, MVT::nxv8bf16}) {
-        // Use custom lowering for MSTORE so we can handle compressstore (using VECTOR_COMPRESS).
+        // Use custom lowering for MSTORE so we can handle compressstore (using
+        // VECTOR_COMPRESS).
         setOperationAction(ISD::MSTORE, VT, Custom);
         setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
       }

>From b881222cc321c29b7af7c122fc72d3af3939e3d4 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 27 Aug 2026 11:49:57 +0100
Subject: [PATCH 12/12] Fix SME2p2 lowering to allow 32/64bit types

---
 .../Target/AArch64/AArch64ISelLowering.cpp    |  21 +-
 .../AArch64/AArch64TargetTransformInfo.h      |  17 +-
 .../CostModel/AArch64/masked_compress_load.ll |  24 +-
 .../AArch64/sve-masked-compressstore.ll       | 490 ++++++++++++++++--
 4 files changed, 485 insertions(+), 67 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 83c07a19a55e8..2bf587fd9136e 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2218,12 +2218,22 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
     for (auto VT :
          {MVT::nxv4i32, MVT::nxv2i64, MVT::nxv2f32, MVT::nxv4f32, MVT::nxv2f64})
       setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
+    for (auto VT : {MVT::nxv2i8, MVT::nxv2i16, MVT::nxv2i32, MVT::nxv2i64,
+                    MVT::nxv2f32, MVT::nxv2f64, MVT::nxv4i8, MVT::nxv4i16,
+                    MVT::nxv4i32, MVT::nxv4f32}) {
+      // Use a custom lowering for masked stores that could be a supported
+      // compressing store. Note: These types still use the normal (Legal)
+      // lowering for non-compressing masked stores.
+      setOperationAction(ISD::MSTORE, VT, Custom);
+    }
 
     // If we have SVE, we can use SVE logic for legal NEON vectors in the lowest
     // bits of the SVE register.
     for (auto VT : {MVT::v2i32, MVT::v4i32, MVT::v2i64, MVT::v2f32, MVT::v4f32,
-                    MVT::v2f64})
+                    MVT::v2f64}) {
+      setOperationAction(ISD::MSTORE, VT, Custom);
       setOperationAction(ISD::VECTOR_COMPRESS, VT, Custom);
+    }
 
     if (Subtarget->hasSVE2p2() || Subtarget->hasSME2p2()) {
       // With +sve2p2/+sme2p2 the full range of vector types are supported.
@@ -2280,15 +2290,6 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
                     MVT::v2f32, MVT::v4f32, MVT::v2f64})
       setOperationAction(ISD::VECREDUCE_SEQ_FADD, VT, Custom);
 
-    for (auto VT : {MVT::nxv2i8, MVT::nxv2i16, MVT::nxv2i32, MVT::nxv2i64,
-                    MVT::nxv2f32, MVT::nxv2f64, MVT::nxv4i8, MVT::nxv4i16,
-                    MVT::nxv4i32, MVT::nxv4f32}) {
-      // Use a custom lowering for masked stores that could be a supported
-      // compressing store. Note: These types still use the normal (Legal)
-      // lowering for non-compressing masked stores.
-      setOperationAction(ISD::MSTORE, VT, Custom);
-    }
-
     // Histcnt is SVE2 only
     if (Subtarget->hasSVE2()) {
       setOperationAction(ISD::EXPERIMENTAL_VECTOR_HISTOGRAM, MVT::nxv4i32,
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
index 2420e01eb3b3d..3038978be259e 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.h
@@ -334,14 +334,15 @@ class AArch64TTIImpl final : public BasicTTIImplBase<AArch64TTIImpl> {
   }
 
   bool isElementTypeLegalForCompressStore(Type *Ty) const {
-    // For streaming SME2p2, only consider 8bit or 16bit Scalar types.
-    if ((ST->isStreamingSVEAvailable() && ST->hasSME2p2()))
-      return Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16;
-
-    return ((ST->hasSVE2p2() || ST->hasSME2p2()) &&
-            (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16)) ||
-           Ty->isFloatTy() || Ty->isDoubleTy() || Ty->isIntegerTy(32) ||
-           Ty->isIntegerTy(64);
+    // 32-bit and 64-bit element types are legal if we have SVE.
+    if (Ty->getScalarSizeInBits() == 32 || Ty->getScalarSizeInBits() == 64)
+      return true;
+
+    // 8-bit and 16-bit types require +sve2p2 or +sme2p2.
+    if (Ty->isIntegerTy(8) || Ty->getScalarSizeInBits() == 16)
+      return ST->hasSVE2p2() || ST->hasSME2p2();
+
+    return false;
   }
 
   bool isLegalMaskedCompressStore(Type *DataType,
diff --git a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
index 69a3edb499b31..54396e133d800 100644
--- a/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/masked_compress_load.ll
@@ -37,15 +37,15 @@ define void @fixed() {
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i16.p0(<4 x i16> poison, ptr poison, <4 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8i16.p0(<8 x i16> poison, ptr poison, <8 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i32.p0(<2 x i32> poison, ptr poison, <2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i32.p0(<4 x i32> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2i64.p0(<2 x i64> poison, ptr poison, <2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v4i32.p0(<4 x i32> poison, ptr poison, <4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v2i64.p0(<2 x i64> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f16.p0(<2 x half> poison, ptr poison, <2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4f16.p0(<4 x half> poison, ptr poison, <4 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v8f16.p0(<8 x half> poison, ptr poison, <8 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f32.p0(<2 x float> poison, ptr poison, <2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4f32.p0(<4 x float> poison, ptr poison, <4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v2f64.p0(<2 x double> poison, ptr poison, <2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.v4i64.p0(<4 x i64> poison, ptr poison, <4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v4f32.p0(<4 x float> poison, ptr poison, <4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.v2f64.p0(<2 x double> poison, ptr poison, <2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.v4i64.p0(<4 x i64> poison, ptr poison, <4 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.v32f16.p0(<32 x half> poison, ptr poison, <32 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
@@ -171,17 +171,17 @@ define void @scalable() {
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i16.p0(<vscale x 2 x i16> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i16.p0(<vscale x 4 x i16> poison, ptr poison, <vscale x 4 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8i16.p0(<vscale x 8 x i16> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i32.p0(<vscale x 2 x i32> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4i32.p0(<vscale x 4 x i32> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2i64.p0(<vscale x 2 x i64> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f16.p0(<vscale x 2 x half> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f16.p0(<vscale x 4 x half> poison, ptr poison, <vscale x 4 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv8f16.p0(<vscale x 8 x half> poison, ptr poison, <vscale x 8 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f32.p0(<vscale x 2 x float> poison, ptr poison, <vscale x 2 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv4f32.p0(<vscale x 4 x float> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 2 for: call void @llvm.masked.compressstore.nxv2f64.p0(<vscale x 2 x double> poison, ptr poison, <vscale x 2 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv1i64.p0(<vscale x 1 x i64> poison, ptr poison, <vscale x 1 x i1> poison)
-; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of Invalid for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
+; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 8 for: call void @llvm.masked.compressstore.nxv4i64.p0(<vscale x 4 x i64> poison, ptr poison, <vscale x 4 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of 16 for: call void @llvm.masked.compressstore.nxv32f16.p0(<vscale x 32 x half> poison, ptr poison, <vscale x 32 x i1> poison)
 ; SVE2p2-SME2p2-STREAMING-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
diff --git a/llvm/test/CodeGen/AArch64/sve-masked-compressstore.ll b/llvm/test/CodeGen/AArch64/sve-masked-compressstore.ll
index df449f79cc9b5..b6fa9ea943c75 100644
--- a/llvm/test/CodeGen/AArch64/sve-masked-compressstore.ll
+++ b/llvm/test/CodeGen/AArch64/sve-masked-compressstore.ll
@@ -1,7 +1,9 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
 ; RUN: llc -mtriple=aarch64 -mattr=+sve < %s | FileCheck %s --check-prefixes=CHECK,CHECK-BASE
 ; RUN: llc -mtriple=aarch64 -aarch64-sve-vector-bits-min=256 -mattr=+sve < %s | FileCheck %s --check-prefixes=CHECK,CHECK-VL256
-
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve,+sme2p2 < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SME2p2
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2 -force-streaming < %s | FileCheck %s --check-prefixes=CHECK,CHECK-SME2p2-STREAMING
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sme2p2,+sve2p2 -force-streaming-compatible < %s | FileCheck %s --check-prefixes=CHECK,CHECK-STREAMING-COMPAT
 ;; Full SVE vectors (supported with +sve)
 
 define void @test_compressstore_nxv4i32(ptr %p, <vscale x 4 x i32> %vec, <vscale x 4 x i1> %mask) {
@@ -115,52 +117,214 @@ define void @test_compressstore_nxv4i16(ptr %p, <vscale x 4 x i16> %vec, <vscale
 ;; NEON vector types (promoted to SVE)
 
 define void @test_compressstore_v2f64(ptr %p, <2 x double> %vec, <2 x i1> %mask) {
-; CHECK-LABEL: test_compressstore_v2f64:
-; CHECK:       // %bb.0:
-; CHECK-NEXT:    ushll v1.2d, v1.2s, #0
-; CHECK-NEXT:    ptrue p0.d, vl2
-; CHECK-NEXT:    // kill: def $q0 killed $q0 def $z0
-; CHECK-NEXT:    shl v1.2d, v1.2d, #63
-; CHECK-NEXT:    cmpne p1.d, p0/z, z1.d, #0
-; CHECK-NEXT:    cntp x8, p1, p1.d
-; CHECK-NEXT:    compact z0.d, p1, z0.d
-; CHECK-NEXT:    whilelo p0.d, xzr, x8
-; CHECK-NEXT:    st1d { z0.d }, p0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-BASE-LABEL: test_compressstore_v2f64:
+; CHECK-BASE:       // %bb.0:
+; CHECK-BASE-NEXT:    ushll v1.2d, v1.2s, #0
+; CHECK-BASE-NEXT:    ptrue p0.d, vl2
+; CHECK-BASE-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT:    shl v1.2d, v1.2d, #63
+; CHECK-BASE-NEXT:    cmpne p1.d, p0/z, z1.d, #0
+; CHECK-BASE-NEXT:    cntp x8, p1, p1.d
+; CHECK-BASE-NEXT:    compact z0.d, p1, z0.d
+; CHECK-BASE-NEXT:    whilelo p0.d, xzr, x8
+; CHECK-BASE-NEXT:    st1d { z0.d }, p0, [x0]
+; CHECK-BASE-NEXT:    ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v2f64:
+; CHECK-VL256:       // %bb.0:
+; CHECK-VL256-NEXT:    ushll v1.2d, v1.2s, #0
+; CHECK-VL256-NEXT:    ptrue p0.d, vl2
+; CHECK-VL256-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT:    shl v1.2d, v1.2d, #63
+; CHECK-VL256-NEXT:    cmpne p1.d, p0/z, z1.d, #0
+; CHECK-VL256-NEXT:    cntp x8, p1, p1.d
+; CHECK-VL256-NEXT:    compact z0.d, p1, z0.d
+; CHECK-VL256-NEXT:    whilelo p0.d, xzr, x8
+; CHECK-VL256-NEXT:    st1d { z0.d }, p0, [x0]
+; CHECK-VL256-NEXT:    ret
+;
+; CHECK-SME2p2-LABEL: test_compressstore_v2f64:
+; CHECK-SME2p2:       // %bb.0:
+; CHECK-SME2p2-NEXT:    ushll v1.2d, v1.2s, #0
+; CHECK-SME2p2-NEXT:    ptrue p0.d, vl2
+; CHECK-SME2p2-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-SME2p2-NEXT:    shl v1.2d, v1.2d, #63
+; CHECK-SME2p2-NEXT:    cmpne p1.d, p0/z, z1.d, #0
+; CHECK-SME2p2-NEXT:    cntp x8, p1, p1.d
+; CHECK-SME2p2-NEXT:    compact z0.d, p1, z0.d
+; CHECK-SME2p2-NEXT:    whilelo p0.d, xzr, x8
+; CHECK-SME2p2-NEXT:    st1d { z0.d }, p0, [x0]
+; CHECK-SME2p2-NEXT:    ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v2f64:
+; CHECK-SME2p2-STREAMING:       // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT:    uunpklo z1.d, z1.s
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p0.d, vl2
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z1.d, z1.d, #63
+; CHECK-SME2p2-STREAMING-NEXT:    asr z1.d, z1.d, #63
+; CHECK-SME2p2-STREAMING-NEXT:    cmpne p1.d, p0/z, z1.d, #0
+; CHECK-SME2p2-STREAMING-NEXT:    cntp x8, p1, p1.d
+; CHECK-SME2p2-STREAMING-NEXT:    compact z0.d, p1, z0.d
+; CHECK-SME2p2-STREAMING-NEXT:    whilelo p0.d, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT:    st1d { z0.d }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT:    ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v2f64:
+; CHECK-STREAMING-COMPAT:       // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p0.d, vl2
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT:    uunpklo z1.d, z1.s
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z1.d, z1.d, #63
+; CHECK-STREAMING-COMPAT-NEXT:    asr z1.d, z1.d, #63
+; CHECK-STREAMING-COMPAT-NEXT:    cmpne p1.d, p0/z, z1.d, #0
+; CHECK-STREAMING-COMPAT-NEXT:    cntp x8, p1, p1.d
+; CHECK-STREAMING-COMPAT-NEXT:    compact z0.d, p1, z0.d
+; CHECK-STREAMING-COMPAT-NEXT:    whilelo p0.d, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT:    st1d { z0.d }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT:    ret
   tail call void @llvm.masked.compressstore.v2f64(<2 x double> %vec, ptr align 8 %p, <2 x i1> %mask)
   ret void
 }
 
 define void @test_compressstore_v4i32(ptr %p, <4 x i32> %vec, <4 x i1> %mask) {
-; CHECK-LABEL: test_compressstore_v4i32:
-; CHECK:       // %bb.0:
-; CHECK-NEXT:    ushll v1.4s, v1.4h, #0
-; CHECK-NEXT:    ptrue p0.s, vl4
-; CHECK-NEXT:    // kill: def $q0 killed $q0 def $z0
-; CHECK-NEXT:    shl v1.4s, v1.4s, #31
-; CHECK-NEXT:    cmpne p1.s, p0/z, z1.s, #0
-; CHECK-NEXT:    cntp x8, p1, p1.s
-; CHECK-NEXT:    compact z0.s, p1, z0.s
-; CHECK-NEXT:    whilelo p0.s, xzr, x8
-; CHECK-NEXT:    st1w { z0.s }, p0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-BASE-LABEL: test_compressstore_v4i32:
+; CHECK-BASE:       // %bb.0:
+; CHECK-BASE-NEXT:    ushll v1.4s, v1.4h, #0
+; CHECK-BASE-NEXT:    ptrue p0.s, vl4
+; CHECK-BASE-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT:    shl v1.4s, v1.4s, #31
+; CHECK-BASE-NEXT:    cmpne p1.s, p0/z, z1.s, #0
+; CHECK-BASE-NEXT:    cntp x8, p1, p1.s
+; CHECK-BASE-NEXT:    compact z0.s, p1, z0.s
+; CHECK-BASE-NEXT:    whilelo p0.s, xzr, x8
+; CHECK-BASE-NEXT:    st1w { z0.s }, p0, [x0]
+; CHECK-BASE-NEXT:    ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v4i32:
+; CHECK-VL256:       // %bb.0:
+; CHECK-VL256-NEXT:    ushll v1.4s, v1.4h, #0
+; CHECK-VL256-NEXT:    ptrue p0.s, vl4
+; CHECK-VL256-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT:    shl v1.4s, v1.4s, #31
+; CHECK-VL256-NEXT:    cmpne p1.s, p0/z, z1.s, #0
+; CHECK-VL256-NEXT:    cntp x8, p1, p1.s
+; CHECK-VL256-NEXT:    compact z0.s, p1, z0.s
+; CHECK-VL256-NEXT:    whilelo p0.s, xzr, x8
+; CHECK-VL256-NEXT:    st1w { z0.s }, p0, [x0]
+; CHECK-VL256-NEXT:    ret
+;
+; CHECK-SME2p2-LABEL: test_compressstore_v4i32:
+; CHECK-SME2p2:       // %bb.0:
+; CHECK-SME2p2-NEXT:    ushll v1.4s, v1.4h, #0
+; CHECK-SME2p2-NEXT:    ptrue p0.s, vl4
+; CHECK-SME2p2-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-SME2p2-NEXT:    shl v1.4s, v1.4s, #31
+; CHECK-SME2p2-NEXT:    cmpne p1.s, p0/z, z1.s, #0
+; CHECK-SME2p2-NEXT:    cntp x8, p1, p1.s
+; CHECK-SME2p2-NEXT:    compact z0.s, p1, z0.s
+; CHECK-SME2p2-NEXT:    whilelo p0.s, xzr, x8
+; CHECK-SME2p2-NEXT:    st1w { z0.s }, p0, [x0]
+; CHECK-SME2p2-NEXT:    ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v4i32:
+; CHECK-SME2p2-STREAMING:       // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT:    uunpklo z1.s, z1.h
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p0.s, vl4
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z1.s, z1.s, #31
+; CHECK-SME2p2-STREAMING-NEXT:    asr z1.s, z1.s, #31
+; CHECK-SME2p2-STREAMING-NEXT:    cmpne p1.s, p0/z, z1.s, #0
+; CHECK-SME2p2-STREAMING-NEXT:    cntp x8, p1, p1.s
+; CHECK-SME2p2-STREAMING-NEXT:    compact z0.s, p1, z0.s
+; CHECK-SME2p2-STREAMING-NEXT:    whilelo p0.s, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT:    st1w { z0.s }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT:    ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v4i32:
+; CHECK-STREAMING-COMPAT:       // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p0.s, vl4
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT:    uunpklo z1.s, z1.h
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z1.s, z1.s, #31
+; CHECK-STREAMING-COMPAT-NEXT:    asr z1.s, z1.s, #31
+; CHECK-STREAMING-COMPAT-NEXT:    cmpne p1.s, p0/z, z1.s, #0
+; CHECK-STREAMING-COMPAT-NEXT:    cntp x8, p1, p1.s
+; CHECK-STREAMING-COMPAT-NEXT:    compact z0.s, p1, z0.s
+; CHECK-STREAMING-COMPAT-NEXT:    whilelo p0.s, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT:    st1w { z0.s }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT:    ret
   tail call void @llvm.masked.compressstore.v4i32(<4 x i32> %vec, ptr align 4 %p, <4 x i1> %mask)
   ret void
 }
 
 define void @test_compressstore_v2i64(ptr %p, <2 x i64> %vec, <2 x i1> %mask) {
-; CHECK-LABEL: test_compressstore_v2i64:
-; CHECK:       // %bb.0:
-; CHECK-NEXT:    ushll v1.2d, v1.2s, #0
-; CHECK-NEXT:    ptrue p0.d, vl2
-; CHECK-NEXT:    // kill: def $q0 killed $q0 def $z0
-; CHECK-NEXT:    shl v1.2d, v1.2d, #63
-; CHECK-NEXT:    cmpne p1.d, p0/z, z1.d, #0
-; CHECK-NEXT:    cntp x8, p1, p1.d
-; CHECK-NEXT:    compact z0.d, p1, z0.d
-; CHECK-NEXT:    whilelo p0.d, xzr, x8
-; CHECK-NEXT:    st1d { z0.d }, p0, [x0]
-; CHECK-NEXT:    ret
+; CHECK-BASE-LABEL: test_compressstore_v2i64:
+; CHECK-BASE:       // %bb.0:
+; CHECK-BASE-NEXT:    ushll v1.2d, v1.2s, #0
+; CHECK-BASE-NEXT:    ptrue p0.d, vl2
+; CHECK-BASE-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-BASE-NEXT:    shl v1.2d, v1.2d, #63
+; CHECK-BASE-NEXT:    cmpne p1.d, p0/z, z1.d, #0
+; CHECK-BASE-NEXT:    cntp x8, p1, p1.d
+; CHECK-BASE-NEXT:    compact z0.d, p1, z0.d
+; CHECK-BASE-NEXT:    whilelo p0.d, xzr, x8
+; CHECK-BASE-NEXT:    st1d { z0.d }, p0, [x0]
+; CHECK-BASE-NEXT:    ret
+;
+; CHECK-VL256-LABEL: test_compressstore_v2i64:
+; CHECK-VL256:       // %bb.0:
+; CHECK-VL256-NEXT:    ushll v1.2d, v1.2s, #0
+; CHECK-VL256-NEXT:    ptrue p0.d, vl2
+; CHECK-VL256-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-VL256-NEXT:    shl v1.2d, v1.2d, #63
+; CHECK-VL256-NEXT:    cmpne p1.d, p0/z, z1.d, #0
+; CHECK-VL256-NEXT:    cntp x8, p1, p1.d
+; CHECK-VL256-NEXT:    compact z0.d, p1, z0.d
+; CHECK-VL256-NEXT:    whilelo p0.d, xzr, x8
+; CHECK-VL256-NEXT:    st1d { z0.d }, p0, [x0]
+; CHECK-VL256-NEXT:    ret
+;
+; CHECK-SME2p2-LABEL: test_compressstore_v2i64:
+; CHECK-SME2p2:       // %bb.0:
+; CHECK-SME2p2-NEXT:    ushll v1.2d, v1.2s, #0
+; CHECK-SME2p2-NEXT:    ptrue p0.d, vl2
+; CHECK-SME2p2-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-SME2p2-NEXT:    shl v1.2d, v1.2d, #63
+; CHECK-SME2p2-NEXT:    cmpne p1.d, p0/z, z1.d, #0
+; CHECK-SME2p2-NEXT:    cntp x8, p1, p1.d
+; CHECK-SME2p2-NEXT:    compact z0.d, p1, z0.d
+; CHECK-SME2p2-NEXT:    whilelo p0.d, xzr, x8
+; CHECK-SME2p2-NEXT:    st1d { z0.d }, p0, [x0]
+; CHECK-SME2p2-NEXT:    ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v2i64:
+; CHECK-SME2p2-STREAMING:       // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT:    uunpklo z1.d, z1.s
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p0.d, vl2
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z1.d, z1.d, #63
+; CHECK-SME2p2-STREAMING-NEXT:    asr z1.d, z1.d, #63
+; CHECK-SME2p2-STREAMING-NEXT:    cmpne p1.d, p0/z, z1.d, #0
+; CHECK-SME2p2-STREAMING-NEXT:    cntp x8, p1, p1.d
+; CHECK-SME2p2-STREAMING-NEXT:    compact z0.d, p1, z0.d
+; CHECK-SME2p2-STREAMING-NEXT:    whilelo p0.d, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT:    st1d { z0.d }, p0, [x0]
+; CHECK-SME2p2-STREAMING-NEXT:    ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v2i64:
+; CHECK-STREAMING-COMPAT:       // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $d1 killed $d1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p0.d, vl2
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT:    uunpklo z1.d, z1.s
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z1.d, z1.d, #63
+; CHECK-STREAMING-COMPAT-NEXT:    asr z1.d, z1.d, #63
+; CHECK-STREAMING-COMPAT-NEXT:    cmpne p1.d, p0/z, z1.d, #0
+; CHECK-STREAMING-COMPAT-NEXT:    cntp x8, p1, p1.d
+; CHECK-STREAMING-COMPAT-NEXT:    compact z0.d, p1, z0.d
+; CHECK-STREAMING-COMPAT-NEXT:    whilelo p0.d, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT:    st1d { z0.d }, p0, [x0]
+; CHECK-STREAMING-COMPAT-NEXT:    ret
   tail call void @llvm.masked.compressstore.v2i64(<2 x i64> %vec, ptr align 8 %p, <2 x i1> %mask)
   ret void
 }
@@ -219,6 +383,144 @@ define void @test_compressstore_v8i32(ptr %p, <8 x i32> %vec, <8 x i1> %mask) {
 ; CHECK-VL256-NEXT:    whilelo p0.s, xzr, x8
 ; CHECK-VL256-NEXT:    st1w { z0.s }, p0, [x0]
 ; CHECK-VL256-NEXT:    ret
+;
+; CHECK-SME2p2-LABEL: test_compressstore_v8i32:
+; CHECK-SME2p2:       // %bb.0:
+; CHECK-SME2p2-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-SME2p2-NEXT:    zip1 v3.8b, v2.8b, v0.8b
+; CHECK-SME2p2-NEXT:    zip2 v2.8b, v2.8b, v0.8b
+; CHECK-SME2p2-NEXT:    adrp x8, .LCPI11_0
+; CHECK-SME2p2-NEXT:    ldr d5, [x8, :lo12:.LCPI11_0]
+; CHECK-SME2p2-NEXT:    ptrue p0.s, vl4
+; CHECK-SME2p2-NEXT:    // kill: def $q1 killed $q1 def $z1
+; CHECK-SME2p2-NEXT:    shl v4.4h, v3.4h, #15
+; CHECK-SME2p2-NEXT:    ushll v2.4s, v2.4h, #0
+; CHECK-SME2p2-NEXT:    ushll v3.4s, v3.4h, #0
+; CHECK-SME2p2-NEXT:    cmlt v4.4h, v4.4h, #0
+; CHECK-SME2p2-NEXT:    shl v2.4s, v2.4s, #31
+; CHECK-SME2p2-NEXT:    shl v3.4s, v3.4s, #31
+; CHECK-SME2p2-NEXT:    and v4.8b, v4.8b, v5.8b
+; CHECK-SME2p2-NEXT:    cmpne p1.s, p0/z, z2.s, #0
+; CHECK-SME2p2-NEXT:    cmpne p2.s, p0/z, z3.s, #0
+; CHECK-SME2p2-NEXT:    ptrue p0.s
+; CHECK-SME2p2-NEXT:    addv h2, v4.4h
+; CHECK-SME2p2-NEXT:    cntp x9, p1, p1.s
+; CHECK-SME2p2-NEXT:    compact z1.s, p1, z1.s
+; CHECK-SME2p2-NEXT:    compact z0.s, p2, z0.s
+; CHECK-SME2p2-NEXT:    cntp x10, p2, p2.s
+; CHECK-SME2p2-NEXT:    fmov w8, s2
+; CHECK-SME2p2-NEXT:    and w8, w8, #0xf
+; CHECK-SME2p2-NEXT:    whilelo p1.s, xzr, x10
+; CHECK-SME2p2-NEXT:    fmov s2, w8
+; CHECK-SME2p2-NEXT:    cnt z2.s, p0/z, z2.s
+; CHECK-SME2p2-NEXT:    whilelo p0.s, xzr, x9
+; CHECK-SME2p2-NEXT:    fmov w8, s2
+; CHECK-SME2p2-NEXT:    st1w { z1.s }, p0, [x0, x8, lsl #2]
+; CHECK-SME2p2-NEXT:    st1w { z0.s }, p1, [x0]
+; CHECK-SME2p2-NEXT:    ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v8i32:
+; CHECK-SME2p2-STREAMING:       // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT:    sub sp, sp, #16
+; CHECK-SME2p2-STREAMING-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-SME2p2-STREAMING-NEXT:    mov z3.b, z2.b[7]
+; CHECK-SME2p2-STREAMING-NEXT:    mov z4.b, z2.b[6]
+; CHECK-SME2p2-STREAMING-NEXT:    mov z5.b, z2.b[5]
+; CHECK-SME2p2-STREAMING-NEXT:    mov z6.b, z2.b[4]
+; CHECK-SME2p2-STREAMING-NEXT:    mov z7.b, z2.b[1]
+; CHECK-SME2p2-STREAMING-NEXT:    mov z16.b, z2.b[2]
+; CHECK-SME2p2-STREAMING-NEXT:    mov z17.b, z2.b[3]
+; CHECK-SME2p2-STREAMING-NEXT:    fmov w8, s2
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p0.s, vl4
+; CHECK-SME2p2-STREAMING-NEXT:    zip1 z3.h, z4.h, z3.h
+; CHECK-SME2p2-STREAMING-NEXT:    zip1 z4.h, z6.h, z5.h
+; CHECK-SME2p2-STREAMING-NEXT:    zip1 z6.h, z2.h, z7.h
+; CHECK-SME2p2-STREAMING-NEXT:    fmov w9, s7
+; CHECK-SME2p2-STREAMING-NEXT:    and w8, w8, #0x1
+; CHECK-SME2p2-STREAMING-NEXT:    zip1 z5.h, z16.h, z17.h
+; CHECK-SME2p2-STREAMING-NEXT:    zip1 z3.s, z4.s, z3.s
+; CHECK-SME2p2-STREAMING-NEXT:    bfi w8, w9, #1, #1
+; CHECK-SME2p2-STREAMING-NEXT:    fmov w9, s16
+; CHECK-SME2p2-STREAMING-NEXT:    zip1 z4.s, z6.s, z5.s
+; CHECK-SME2p2-STREAMING-NEXT:    uunpklo z3.s, z3.h
+; CHECK-SME2p2-STREAMING-NEXT:    bfi w8, w9, #2, #1
+; CHECK-SME2p2-STREAMING-NEXT:    fmov w9, s17
+; CHECK-SME2p2-STREAMING-NEXT:    uunpklo z4.s, z4.h
+; CHECK-SME2p2-STREAMING-NEXT:    orr w8, w8, w9, lsl #3
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z3.s, z3.s, #31
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z4.s, z4.s, #31
+; CHECK-SME2p2-STREAMING-NEXT:    and w8, w8, #0xf
+; CHECK-SME2p2-STREAMING-NEXT:    asr z2.s, z3.s, #31
+; CHECK-SME2p2-STREAMING-NEXT:    asr z3.s, z4.s, #31
+; CHECK-SME2p2-STREAMING-NEXT:    cmpne p1.s, p0/z, z2.s, #0
+; CHECK-SME2p2-STREAMING-NEXT:    fmov s2, w8
+; CHECK-SME2p2-STREAMING-NEXT:    cmpne p2.s, p0/z, z3.s, #0
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p0.s
+; CHECK-SME2p2-STREAMING-NEXT:    cnt z2.s, p0/z, z2.s
+; CHECK-SME2p2-STREAMING-NEXT:    cntp x9, p1, p1.s
+; CHECK-SME2p2-STREAMING-NEXT:    compact z1.s, p1, z1.s
+; CHECK-SME2p2-STREAMING-NEXT:    fmov w10, s2
+; CHECK-SME2p2-STREAMING-NEXT:    cntp x8, p2, p2.s
+; CHECK-SME2p2-STREAMING-NEXT:    compact z0.s, p2, z0.s
+; CHECK-SME2p2-STREAMING-NEXT:    whilelo p0.s, xzr, x9
+; CHECK-SME2p2-STREAMING-NEXT:    whilelo p1.s, xzr, x8
+; CHECK-SME2p2-STREAMING-NEXT:    st1w { z1.s }, p0, [x0, x10, lsl #2]
+; CHECK-SME2p2-STREAMING-NEXT:    st1w { z0.s }, p1, [x0]
+; CHECK-SME2p2-STREAMING-NEXT:    add sp, sp, #16
+; CHECK-SME2p2-STREAMING-NEXT:    ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v8i32:
+; CHECK-STREAMING-COMPAT:       // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT:    sub sp, sp, #16
+; CHECK-STREAMING-COMPAT-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $d2 killed $d2 def $z2
+; CHECK-STREAMING-COMPAT-NEXT:    mov z3.b, z2.b[7]
+; CHECK-STREAMING-COMPAT-NEXT:    mov z4.b, z2.b[6]
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q1 killed $q1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT:    mov z5.b, z2.b[5]
+; CHECK-STREAMING-COMPAT-NEXT:    mov z6.b, z2.b[4]
+; CHECK-STREAMING-COMPAT-NEXT:    mov z7.b, z2.b[1]
+; CHECK-STREAMING-COMPAT-NEXT:    mov z16.b, z2.b[2]
+; CHECK-STREAMING-COMPAT-NEXT:    mov z17.b, z2.b[3]
+; CHECK-STREAMING-COMPAT-NEXT:    fmov w8, s2
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p0.s, vl4
+; CHECK-STREAMING-COMPAT-NEXT:    zip1 z3.h, z4.h, z3.h
+; CHECK-STREAMING-COMPAT-NEXT:    zip1 z4.h, z6.h, z5.h
+; CHECK-STREAMING-COMPAT-NEXT:    zip1 z6.h, z2.h, z7.h
+; CHECK-STREAMING-COMPAT-NEXT:    fmov w9, s7
+; CHECK-STREAMING-COMPAT-NEXT:    and w8, w8, #0x1
+; CHECK-STREAMING-COMPAT-NEXT:    zip1 z5.h, z16.h, z17.h
+; CHECK-STREAMING-COMPAT-NEXT:    zip1 z3.s, z4.s, z3.s
+; CHECK-STREAMING-COMPAT-NEXT:    bfi w8, w9, #1, #1
+; CHECK-STREAMING-COMPAT-NEXT:    fmov w9, s16
+; CHECK-STREAMING-COMPAT-NEXT:    zip1 z4.s, z6.s, z5.s
+; CHECK-STREAMING-COMPAT-NEXT:    uunpklo z3.s, z3.h
+; CHECK-STREAMING-COMPAT-NEXT:    bfi w8, w9, #2, #1
+; CHECK-STREAMING-COMPAT-NEXT:    fmov w9, s17
+; CHECK-STREAMING-COMPAT-NEXT:    uunpklo z4.s, z4.h
+; CHECK-STREAMING-COMPAT-NEXT:    orr w8, w8, w9, lsl #3
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z3.s, z3.s, #31
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z4.s, z4.s, #31
+; CHECK-STREAMING-COMPAT-NEXT:    and w8, w8, #0xf
+; CHECK-STREAMING-COMPAT-NEXT:    asr z2.s, z3.s, #31
+; CHECK-STREAMING-COMPAT-NEXT:    asr z3.s, z4.s, #31
+; CHECK-STREAMING-COMPAT-NEXT:    cmpne p1.s, p0/z, z2.s, #0
+; CHECK-STREAMING-COMPAT-NEXT:    fmov s2, w8
+; CHECK-STREAMING-COMPAT-NEXT:    cmpne p2.s, p0/z, z3.s, #0
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p0.s
+; CHECK-STREAMING-COMPAT-NEXT:    cnt z2.s, p0/z, z2.s
+; CHECK-STREAMING-COMPAT-NEXT:    cntp x9, p1, p1.s
+; CHECK-STREAMING-COMPAT-NEXT:    compact z1.s, p1, z1.s
+; CHECK-STREAMING-COMPAT-NEXT:    fmov w10, s2
+; CHECK-STREAMING-COMPAT-NEXT:    cntp x8, p2, p2.s
+; CHECK-STREAMING-COMPAT-NEXT:    compact z0.s, p2, z0.s
+; CHECK-STREAMING-COMPAT-NEXT:    whilelo p0.s, xzr, x9
+; CHECK-STREAMING-COMPAT-NEXT:    whilelo p1.s, xzr, x8
+; CHECK-STREAMING-COMPAT-NEXT:    st1w { z1.s }, p0, [x0, x10, lsl #2]
+; CHECK-STREAMING-COMPAT-NEXT:    st1w { z0.s }, p1, [x0]
+; CHECK-STREAMING-COMPAT-NEXT:    add sp, sp, #16
+; CHECK-STREAMING-COMPAT-NEXT:    ret
   tail call void @llvm.masked.compressstore.v8i32(<8 x i32> %vec, ptr align 4 %p, <8 x i1> %mask)
   ret void
 }
@@ -275,6 +577,120 @@ define void @test_compressstore_v4i64(ptr %p, <4 x i64> %vec, <4 x i1> %mask) {
 ; CHECK-VL256-NEXT:    whilelo p0.d, xzr, x8
 ; CHECK-VL256-NEXT:    st1d { z0.d }, p0, [x0]
 ; CHECK-VL256-NEXT:    ret
+;
+; CHECK-SME2p2-LABEL: test_compressstore_v4i64:
+; CHECK-SME2p2:       // %bb.0:
+; CHECK-SME2p2-NEXT:    ushll v2.4s, v2.4h, #0
+; CHECK-SME2p2-NEXT:    index z4.s, #1, #1
+; CHECK-SME2p2-NEXT:    ptrue p0.d, vl2
+; CHECK-SME2p2-NEXT:    // kill: def $q1 killed $q1 def $z1
+; CHECK-SME2p2-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-SME2p2-NEXT:    shl v3.2s, v2.2s, #31
+; CHECK-SME2p2-NEXT:    ushll2 v5.2d, v2.4s, #0
+; CHECK-SME2p2-NEXT:    ushll v2.2d, v2.2s, #0
+; CHECK-SME2p2-NEXT:    cmlt v3.2s, v3.2s, #0
+; CHECK-SME2p2-NEXT:    shl v2.2d, v2.2d, #63
+; CHECK-SME2p2-NEXT:    and v3.8b, v3.8b, v4.8b
+; CHECK-SME2p2-NEXT:    shl v4.2d, v5.2d, #63
+; CHECK-SME2p2-NEXT:    cmpne p2.d, p0/z, z2.d, #0
+; CHECK-SME2p2-NEXT:    addp v3.2s, v3.2s, v3.2s
+; CHECK-SME2p2-NEXT:    cmpne p1.d, p0/z, z4.d, #0
+; CHECK-SME2p2-NEXT:    ptrue p0.s
+; CHECK-SME2p2-NEXT:    cntp x10, p2, p2.d
+; CHECK-SME2p2-NEXT:    compact z0.d, p2, z0.d
+; CHECK-SME2p2-NEXT:    fmov w8, s3
+; CHECK-SME2p2-NEXT:    cntp x9, p1, p1.d
+; CHECK-SME2p2-NEXT:    compact z1.d, p1, z1.d
+; CHECK-SME2p2-NEXT:    whilelo p1.d, xzr, x10
+; CHECK-SME2p2-NEXT:    and w8, w8, #0x3
+; CHECK-SME2p2-NEXT:    fmov s2, w8
+; CHECK-SME2p2-NEXT:    cnt z2.s, p0/z, z2.s
+; CHECK-SME2p2-NEXT:    whilelo p0.d, xzr, x9
+; CHECK-SME2p2-NEXT:    fmov w8, s2
+; CHECK-SME2p2-NEXT:    st1d { z1.d }, p0, [x0, x8, lsl #3]
+; CHECK-SME2p2-NEXT:    st1d { z0.d }, p1, [x0]
+; CHECK-SME2p2-NEXT:    ret
+;
+; CHECK-SME2p2-STREAMING-LABEL: test_compressstore_v4i64:
+; CHECK-SME2p2-STREAMING:       // %bb.0:
+; CHECK-SME2p2-STREAMING-NEXT:    sub sp, sp, #16
+; CHECK-SME2p2-STREAMING-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-SME2p2-STREAMING-NEXT:    uunpklo z2.s, z2.h
+; CHECK-SME2p2-STREAMING-NEXT:    index z5.s, #1, #1
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p0.s, vl2
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p1.d, vl2
+; CHECK-SME2p2-STREAMING-NEXT:    movprfx z3, z2
+; CHECK-SME2p2-STREAMING-NEXT:    ext z3.b, z3.b, z2.b, #8
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z4.s, z2.s, #31
+; CHECK-SME2p2-STREAMING-NEXT:    uunpklo z2.d, z2.s
+; CHECK-SME2p2-STREAMING-NEXT:    uunpklo z3.d, z3.s
+; CHECK-SME2p2-STREAMING-NEXT:    asr z4.s, z4.s, #31
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z2.d, z2.d, #63
+; CHECK-SME2p2-STREAMING-NEXT:    and z4.d, z4.d, z5.d
+; CHECK-SME2p2-STREAMING-NEXT:    lsl z3.d, z3.d, #63
+; CHECK-SME2p2-STREAMING-NEXT:    asr z2.d, z2.d, #63
+; CHECK-SME2p2-STREAMING-NEXT:    uaddv d4, p0, z4.s
+; CHECK-SME2p2-STREAMING-NEXT:    asr z3.d, z3.d, #63
+; CHECK-SME2p2-STREAMING-NEXT:    cmpne p2.d, p1/z, z2.d, #0
+; CHECK-SME2p2-STREAMING-NEXT:    cmpne p0.d, p1/z, z3.d, #0
+; CHECK-SME2p2-STREAMING-NEXT:    str b4, [sp, #12]
+; CHECK-SME2p2-STREAMING-NEXT:    ptrue p1.s
+; CHECK-SME2p2-STREAMING-NEXT:    ldrb w8, [sp, #12]
+; CHECK-SME2p2-STREAMING-NEXT:    cntp x10, p2, p2.d
+; CHECK-SME2p2-STREAMING-NEXT:    compact z0.d, p2, z0.d
+; CHECK-SME2p2-STREAMING-NEXT:    fmov s2, w8
+; CHECK-SME2p2-STREAMING-NEXT:    cntp x9, p0, p0.d
+; CHECK-SME2p2-STREAMING-NEXT:    compact z1.d, p0, z1.d
+; CHECK-SME2p2-STREAMING-NEXT:    cnt z2.s, p1/z, z2.s
+; CHECK-SME2p2-STREAMING-NEXT:    whilelo p1.d, xzr, x10
+; CHECK-SME2p2-STREAMING-NEXT:    fmov w8, s2
+; CHECK-SME2p2-STREAMING-NEXT:    whilelo p0.d, xzr, x9
+; CHECK-SME2p2-STREAMING-NEXT:    st1d { z1.d }, p0, [x0, x8, lsl #3]
+; CHECK-SME2p2-STREAMING-NEXT:    st1d { z0.d }, p1, [x0]
+; CHECK-SME2p2-STREAMING-NEXT:    add sp, sp, #16
+; CHECK-SME2p2-STREAMING-NEXT:    ret
+;
+; CHECK-STREAMING-COMPAT-LABEL: test_compressstore_v4i64:
+; CHECK-STREAMING-COMPAT:       // %bb.0:
+; CHECK-STREAMING-COMPAT-NEXT:    sub sp, sp, #16
+; CHECK-STREAMING-COMPAT-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $d2 killed $d2 def $z2
+; CHECK-STREAMING-COMPAT-NEXT:    index z5.s, #1, #1
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p0.s, vl2
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q1 killed $q1 def $z1
+; CHECK-STREAMING-COMPAT-NEXT:    // kill: def $q0 killed $q0 def $z0
+; CHECK-STREAMING-COMPAT-NEXT:    uunpklo z2.s, z2.h
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p1.d, vl2
+; CHECK-STREAMING-COMPAT-NEXT:    movprfx z3, z2
+; CHECK-STREAMING-COMPAT-NEXT:    ext z3.b, z3.b, z2.b, #8
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z4.s, z2.s, #31
+; CHECK-STREAMING-COMPAT-NEXT:    uunpklo z2.d, z2.s
+; CHECK-STREAMING-COMPAT-NEXT:    uunpklo z3.d, z3.s
+; CHECK-STREAMING-COMPAT-NEXT:    asr z4.s, z4.s, #31
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z2.d, z2.d, #63
+; CHECK-STREAMING-COMPAT-NEXT:    and z4.d, z4.d, z5.d
+; CHECK-STREAMING-COMPAT-NEXT:    lsl z3.d, z3.d, #63
+; CHECK-STREAMING-COMPAT-NEXT:    asr z2.d, z2.d, #63
+; CHECK-STREAMING-COMPAT-NEXT:    uaddv d4, p0, z4.s
+; CHECK-STREAMING-COMPAT-NEXT:    asr z3.d, z3.d, #63
+; CHECK-STREAMING-COMPAT-NEXT:    cmpne p2.d, p1/z, z2.d, #0
+; CHECK-STREAMING-COMPAT-NEXT:    cmpne p0.d, p1/z, z3.d, #0
+; CHECK-STREAMING-COMPAT-NEXT:    str b4, [sp, #12]
+; CHECK-STREAMING-COMPAT-NEXT:    ptrue p1.s
+; CHECK-STREAMING-COMPAT-NEXT:    ldrb w8, [sp, #12]
+; CHECK-STREAMING-COMPAT-NEXT:    cntp x10, p2, p2.d
+; CHECK-STREAMING-COMPAT-NEXT:    compact z0.d, p2, z0.d
+; CHECK-STREAMING-COMPAT-NEXT:    fmov s2, w8
+; CHECK-STREAMING-COMPAT-NEXT:    cntp x9, p0, p0.d
+; CHECK-STREAMING-COMPAT-NEXT:    compact z1.d, p0, z1.d
+; CHECK-STREAMING-COMPAT-NEXT:    cnt z2.s, p1/z, z2.s
+; CHECK-STREAMING-COMPAT-NEXT:    whilelo p1.d, xzr, x10
+; CHECK-STREAMING-COMPAT-NEXT:    fmov w8, s2
+; CHECK-STREAMING-COMPAT-NEXT:    whilelo p0.d, xzr, x9
+; CHECK-STREAMING-COMPAT-NEXT:    st1d { z1.d }, p0, [x0, x8, lsl #3]
+; CHECK-STREAMING-COMPAT-NEXT:    st1d { z0.d }, p1, [x0]
+; CHECK-STREAMING-COMPAT-NEXT:    add sp, sp, #16
+; CHECK-STREAMING-COMPAT-NEXT:    ret
   tail call void @llvm.masked.compressstore.v4i64(<4 x i64> %vec, ptr align 8 %p, <4 x i1> %mask)
   ret void
 }



More information about the llvm-commits mailing list