[llvm] [AArch64][InstCombine] Fold zext of all-active SVE cmpne-zero (PR #207720)

Matthew Devereau via llvm-commits llvm-commits at lists.llvm.org
Thu Jul 9 06:58:22 PDT 2026


https://github.com/MDevereau updated https://github.com/llvm/llvm-project/pull/207720

>From 2b2a195a8deefb5232cd489b629cfc1466621ae8 Mon Sep 17 00:00:00 2001
From: Matt Devereau <matthew.devereau at arm.com>
Date: Thu, 2 Jul 2026 11:50:15 +0000
Subject: [PATCH 1/3] [AArch64][InstCombine] Fold zext of all-active SVE
 cmpne-zero

Fold zext(cmpne(ptrue, x, 0)) to umin(ptrue, x, 1).

Do not fold non-ptrue predicates, since cmpne zeros inactive lanes and umin
merges inactive lanes
---
 .../AArch64/AArch64TargetTransformInfo.cpp    | 25 ++++++++
 .../AArch64/sve-intrinsic-opts-cmpne.ll       | 60 +++++++++++++++++++
 2 files changed, 85 insertions(+)

diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index b7d8a8ff4657b..5e9f56a83562c 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -2173,6 +2173,28 @@ static std::optional<Instruction *> instCombineXorSVECmpNE(InstCombiner &IC,
   return &II;
 }
 
+// zext(cmpne(ptrue, %v, 0))
+// -> umin(ptrue, %v, 1)
+static std::optional<Instruction *> instCombineZExtSVECmpNE(InstCombiner &IC,
+                                                            IntrinsicInst &II) {
+  if (!isAllActivePredicate(II.getOperand(0)) ||
+      !match(II.getOperand(2), m_Zero()) || !II.hasOneUse())
+    return std::nullopt;
+
+  auto *User = cast<Instruction>(*II.user_begin());
+  if (!match(User, m_ZExt(m_Specific(&II))))
+    return std::nullopt;
+
+  IC.Builder.SetInsertPoint(User);
+  Value *UMIN = IC.Builder.CreateIntrinsic(
+      Intrinsic::aarch64_sve_umin, II.getOperand(1)->getType(),
+      {II.getOperand(0), II.getOperand(1),
+       ConstantInt::get(II.getOperand(1)->getType(), 1)});
+  IC.replaceInstUsesWith(*User, UMIN);
+  IC.eraseInstFromFunction(*User);
+  return &II;
+}
+
 static std::optional<Instruction *> instCombineSVECmpNE(InstCombiner &IC,
                                                         IntrinsicInst &II) {
   LLVMContext &Ctx = II.getContext();
@@ -2180,6 +2202,9 @@ static std::optional<Instruction *> instCombineSVECmpNE(InstCombiner &IC,
   if (auto Res = instCombineXorSVECmpNE(IC, II))
     return Res;
 
+  if (auto Res = instCombineZExtSVECmpNE(IC, II))
+    return Res;
+
   if (!isAllActivePredicate(II.getArgOperand(0)))
     return std::nullopt;
 
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
index e803572a06523..f3e1c67511347 100644
--- a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
@@ -261,6 +261,66 @@ define <vscale x 16 x i1> @not_cmpne_wrong_xor_operand(<vscale x 16 x i8> %vec_a
   ret <vscale x 16 x i1> %not
 }
 
+define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_i8(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[ZEXT:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.umin.u.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
+; CHECK-NEXT:    ret <vscale x 16 x i8> [[ZEXT]]
+;
+  %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
+  %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+  ret <vscale x 16 x i8> %zext
+}
+
+define <vscale x 4 x i32> @zext_cmpne_wide_i32(<vscale x 4 x i32> %vec) #0 {
+; CHECK-LABEL: define <vscale x 4 x i32> @zext_cmpne_wide_i32(
+; CHECK-SAME: <vscale x 4 x i32> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[ZEXT:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.umin.u.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> [[VEC]], <vscale x 4 x i32> splat (i32 1))
+; CHECK-NEXT:    ret <vscale x 4 x i32> [[ZEXT]]
+;
+  %cmp = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> %vec, <vscale x 2 x i64> zeroinitializer)
+  %zext = zext <vscale x 4 x i1> %cmp to <vscale x 4 x i32>
+  ret <vscale x 4 x i32> %zext
+}
+
+define <vscale x 16 x i8> @zext_cmpne_non_ptrue(<vscale x 16 x i8> %vec, <vscale x 16 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_non_ptrue(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]], <vscale x 16 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[CMP:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> [[PG]], <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> zeroinitializer)
+; CHECK-NEXT:    [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i8>
+; CHECK-NEXT:    ret <vscale x 16 x i8> [[ZEXT]]
+;
+  %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
+  %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+  ret <vscale x 16 x i8> %zext
+}
+
+define <vscale x 16 x i8> @zext_cmpne_nonzero_rhs(<vscale x 16 x i8> %vec) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_nonzero_rhs(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[CMP:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
+; CHECK-NEXT:    [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i8>
+; CHECK-NEXT:    ret <vscale x 16 x i8> [[ZEXT]]
+;
+  %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> splat (i8 1))
+  %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+  ret <vscale x 16 x i8> %zext
+}
+
+define <vscale x 16 x i8> @zext_cmpne_multiple_uses(<vscale x 16 x i8> %vec) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_multiple_uses(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[CMP:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> zeroinitializer)
+; CHECK-NEXT:    call void (...) @llvm.fake.use(<vscale x 16 x i1> [[CMP]])
+; CHECK-NEXT:    [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i8>
+; CHECK-NEXT:    ret <vscale x 16 x i8> [[ZEXT]]
+;
+  %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
+  call void (...) @llvm.fake.use(<vscale x 16 x i1> %cmp)
+  %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+  ret <vscale x 16 x i8> %zext
+}
+
 ; Cases that cannot be converted
 
 define <vscale x 2 x i1> @dupq_neg1() #0 {

>From 9ff1cacd2b7f3253bb18928181103eea5c4b07c7 Mon Sep 17 00:00:00 2001
From: Matt Devereau <matthew.devereau at arm.com>
Date: Wed, 8 Jul 2026 17:25:17 +0000
Subject: [PATCH 2/3] Add commutability, use Intrinsic::min

---
 .../AArch64/AArch64TargetTransformInfo.cpp    | 20 ++++++++++++-------
 .../AArch64/sve-intrinsic-opts-cmpne.ll       | 18 +++++++++++++++--
 2 files changed, 29 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 5e9f56a83562c..7bb595062fab4 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -2177,20 +2177,26 @@ static std::optional<Instruction *> instCombineXorSVECmpNE(InstCombiner &IC,
 // -> umin(ptrue, %v, 1)
 static std::optional<Instruction *> instCombineZExtSVECmpNE(InstCombiner &IC,
                                                             IntrinsicInst &II) {
-  if (!isAllActivePredicate(II.getOperand(0)) ||
-      !match(II.getOperand(2), m_Zero()) || !II.hasOneUse())
+  if (!isAllActivePredicate(II.getOperand(0)) || !II.hasOneUse())
     return std::nullopt;
 
+  Value *Zero = II.getOperand(1);
+  Value *Op = II.getOperand(2);
+  if (!match(Zero, m_Zero())) {
+    std::swap(Zero, Op);
+    if (!match(Zero, m_Zero()))
+      return std::nullopt;
+  }
+
   auto *User = cast<Instruction>(*II.user_begin());
   if (!match(User, m_ZExt(m_Specific(&II))))
     return std::nullopt;
 
   IC.Builder.SetInsertPoint(User);
-  Value *UMIN = IC.Builder.CreateIntrinsic(
-      Intrinsic::aarch64_sve_umin, II.getOperand(1)->getType(),
-      {II.getOperand(0), II.getOperand(1),
-       ConstantInt::get(II.getOperand(1)->getType(), 1)});
-  IC.replaceInstUsesWith(*User, UMIN);
+  Value *UMin =
+      IC.Builder.CreateIntrinsic(Intrinsic::umin, {Op->getType()},
+                                 {Op, ConstantInt::get(Op->getType(), 1)});
+  IC.replaceInstUsesWith(*User, UMin);
   IC.eraseInstFromFunction(*User);
   return &II;
 }
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
index f3e1c67511347..c20ed9354d0ae 100644
--- a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
@@ -264,7 +264,8 @@ define <vscale x 16 x i1> @not_cmpne_wrong_xor_operand(<vscale x 16 x i8> %vec_a
 define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
 ; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_i8(
 ; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[ZEXT:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.umin.u.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
+; CHECK-NEXT:    [[TMP1:%.*]] = icmp ne <vscale x 16 x i8> [[VEC]], zeroinitializer
+; CHECK-NEXT:    [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[TMP1]] to <vscale x 16 x i8>
 ; CHECK-NEXT:    ret <vscale x 16 x i8> [[ZEXT]]
 ;
   %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
@@ -272,10 +273,23 @@ define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
   ret <vscale x 16 x i8> %zext
 }
 
+define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(<vscale x 16 x i8> %vec) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[TMP1:%.*]] = icmp ne <vscale x 16 x i8> [[VEC]], zeroinitializer
+; CHECK-NEXT:    [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[TMP1]] to <vscale x 16 x i8>
+; CHECK-NEXT:    ret <vscale x 16 x i8> [[ZEXT]]
+;
+  %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> zeroinitializer, <vscale x 16 x i8> %vec)
+  %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+  ret <vscale x 16 x i8> %zext
+}
+
 define <vscale x 4 x i32> @zext_cmpne_wide_i32(<vscale x 4 x i32> %vec) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i32> @zext_cmpne_wide_i32(
 ; CHECK-SAME: <vscale x 4 x i32> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[ZEXT:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.umin.u.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> [[VEC]], <vscale x 4 x i32> splat (i32 1))
+; CHECK-NEXT:    [[TMP1:%.*]] = icmp ne <vscale x 4 x i32> [[VEC]], zeroinitializer
+; CHECK-NEXT:    [[ZEXT:%.*]] = zext <vscale x 4 x i1> [[TMP1]] to <vscale x 4 x i32>
 ; CHECK-NEXT:    ret <vscale x 4 x i32> [[ZEXT]]
 ;
   %cmp = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> %vec, <vscale x 2 x i64> zeroinitializer)

>From e1b7dc8a836285b92c10db1ef4c175205f4a45a0 Mon Sep 17 00:00:00 2001
From: Matt Devereau <matthew.devereau at arm.com>
Date: Thu, 9 Jul 2026 12:53:36 +0000
Subject: [PATCH 3/3] Remove commutativity, revert back to aarch64.sve.umin,
 drop one use

---
 .../AArch64/AArch64TargetTransformInfo.cpp    | 39 ++++++++++---------
 .../AArch64/sve-intrinsic-opts-cmpne.ll       | 32 +++++++++------
 2 files changed, 40 insertions(+), 31 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 7bb595062fab4..419491cc0b42e 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -2174,31 +2174,32 @@ static std::optional<Instruction *> instCombineXorSVECmpNE(InstCombiner &IC,
 }
 
 // zext(cmpne(ptrue, %v, 0))
-// -> umin(ptrue, %v, 1)
+// -> umin(%pg, %v, 1)
 static std::optional<Instruction *> instCombineZExtSVECmpNE(InstCombiner &IC,
                                                             IntrinsicInst &II) {
-  if (!isAllActivePredicate(II.getOperand(0)) || !II.hasOneUse())
+  if (!isAllActivePredicate(II.getOperand(0)) ||
+      !match(II.getOperand(2), m_Zero()))
     return std::nullopt;
 
-  Value *Zero = II.getOperand(1);
-  Value *Op = II.getOperand(2);
-  if (!match(Zero, m_Zero())) {
-    std::swap(Zero, Op);
-    if (!match(Zero, m_Zero()))
-      return std::nullopt;
+  bool Changed = false;
+  for (auto *U : II.users()) {
+    if (match(U, m_ZExt(m_Specific(&II)))) {
+      auto *User = cast<Instruction>(U);
+      Type *Ty = II.getOperand(1)->getType();
+      if (User->getType() != Ty)
+        continue;
+      IC.Builder.SetInsertPoint(User);
+      Value *UMin = IC.Builder.CreateIntrinsic(
+          Intrinsic::aarch64_sve_umin, Ty,
+          {II.getOperand(0), II.getOperand(1), ConstantInt::get(Ty, 1)});
+      IC.replaceInstUsesWith(*User, UMin);
+      IC.eraseInstFromFunction(*User);
+      Changed = true;
+    } else
+      continue;
   }
 
-  auto *User = cast<Instruction>(*II.user_begin());
-  if (!match(User, m_ZExt(m_Specific(&II))))
-    return std::nullopt;
-
-  IC.Builder.SetInsertPoint(User);
-  Value *UMin =
-      IC.Builder.CreateIntrinsic(Intrinsic::umin, {Op->getType()},
-                                 {Op, ConstantInt::get(Op->getType(), 1)});
-  IC.replaceInstUsesWith(*User, UMin);
-  IC.eraseInstFromFunction(*User);
-  return &II;
+  return Changed ? std::optional<Instruction *>(&II) : std::nullopt;
 }
 
 static std::optional<Instruction *> instCombineSVECmpNE(InstCombiner &IC,
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
index c20ed9354d0ae..11e6e0089293a 100644
--- a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
@@ -264,8 +264,7 @@ define <vscale x 16 x i1> @not_cmpne_wrong_xor_operand(<vscale x 16 x i8> %vec_a
 define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
 ; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_i8(
 ; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp ne <vscale x 16 x i8> [[VEC]], zeroinitializer
-; CHECK-NEXT:    [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[TMP1]] to <vscale x 16 x i8>
+; CHECK-NEXT:    [[ZEXT:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.umin.u.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
 ; CHECK-NEXT:    ret <vscale x 16 x i8> [[ZEXT]]
 ;
   %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
@@ -276,7 +275,7 @@ define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
 define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(<vscale x 16 x i8> %vec) #0 {
 ; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(
 ; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp ne <vscale x 16 x i8> [[VEC]], zeroinitializer
+; CHECK-NEXT:    [[TMP1:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> zeroinitializer, <vscale x 16 x i8> [[VEC]])
 ; CHECK-NEXT:    [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[TMP1]] to <vscale x 16 x i8>
 ; CHECK-NEXT:    ret <vscale x 16 x i8> [[ZEXT]]
 ;
@@ -288,8 +287,7 @@ define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(<vscale x 16 x i8> %vec) #0 {
 define <vscale x 4 x i32> @zext_cmpne_wide_i32(<vscale x 4 x i32> %vec) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i32> @zext_cmpne_wide_i32(
 ; CHECK-SAME: <vscale x 4 x i32> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp ne <vscale x 4 x i32> [[VEC]], zeroinitializer
-; CHECK-NEXT:    [[ZEXT:%.*]] = zext <vscale x 4 x i1> [[TMP1]] to <vscale x 4 x i32>
+; CHECK-NEXT:    [[ZEXT:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.umin.u.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> [[VEC]], <vscale x 4 x i32> splat (i32 1))
 ; CHECK-NEXT:    ret <vscale x 4 x i32> [[ZEXT]]
 ;
   %cmp = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> %vec, <vscale x 2 x i64> zeroinitializer)
@@ -321,18 +319,28 @@ define <vscale x 16 x i8> @zext_cmpne_nonzero_rhs(<vscale x 16 x i8> %vec) #0 {
   ret <vscale x 16 x i8> %zext
 }
 
-define <vscale x 16 x i8> @zext_cmpne_multiple_uses(<vscale x 16 x i8> %vec) #0 {
+define <vscale x 16 x i8> @zext_cmpne_multiple_uses(<vscale x 16 x i8> %vec, <vscale x 16 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_multiple_uses(
-; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]], <vscale x 16 x i1> [[PG:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:    [[CMP:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> zeroinitializer)
-; CHECK-NEXT:    call void (...) @llvm.fake.use(<vscale x 16 x i1> [[CMP]])
-; CHECK-NEXT:    [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i8>
+; CHECK-NEXT:    [[AND_1:%.*]] = and <vscale x 16 x i1> [[CMP]], [[PG]]
+; CHECK-NEXT:    call void (...) @llvm.fake.use(<vscale x 16 x i1> [[AND_1]])
+; CHECK-NEXT:    [[ZEXT:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.umin.u.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
+; CHECK-NEXT:    [[ZEXT_1:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i32>
+; CHECK-NEXT:    call void (...) @llvm.fake.use(<vscale x 16 x i32> [[ZEXT_1]])
 ; CHECK-NEXT:    ret <vscale x 16 x i8> [[ZEXT]]
 ;
   %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
-  call void (...) @llvm.fake.use(<vscale x 16 x i1> %cmp)
-  %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
-  ret <vscale x 16 x i8> %zext
+
+  %and.1 = and <vscale x 16 x i1> %cmp, %pg
+  call void (...) @llvm.fake.use(<vscale x 16 x i1> %and.1)
+
+  %zext.0 = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+
+  %zext.1 = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i32>
+  call void (...) @llvm.fake.use(<vscale x 16 x i32> %zext.1)
+
+  ret <vscale x 16 x i8> %zext.0
 }
 
 ; Cases that cannot be converted



More information about the llvm-commits mailing list