[llvm] [AArch64][InstCombine] Fold zext of all-active SVE cmpne-zero (PR #207720)
Matthew Devereau via llvm-commits
llvm-commits at lists.llvm.org
Thu Jul 9 06:58:22 PDT 2026
https://github.com/MDevereau updated https://github.com/llvm/llvm-project/pull/207720
>From 2b2a195a8deefb5232cd489b629cfc1466621ae8 Mon Sep 17 00:00:00 2001
From: Matt Devereau <matthew.devereau at arm.com>
Date: Thu, 2 Jul 2026 11:50:15 +0000
Subject: [PATCH 1/3] [AArch64][InstCombine] Fold zext of all-active SVE
cmpne-zero
Fold zext(cmpne(ptrue, x, 0)) to umin(ptrue, x, 1).
Do not fold non-ptrue predicates, since cmpne zeros inactive lanes and umin
merges inactive lanes
---
.../AArch64/AArch64TargetTransformInfo.cpp | 25 ++++++++
.../AArch64/sve-intrinsic-opts-cmpne.ll | 60 +++++++++++++++++++
2 files changed, 85 insertions(+)
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index b7d8a8ff4657b..5e9f56a83562c 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -2173,6 +2173,28 @@ static std::optional<Instruction *> instCombineXorSVECmpNE(InstCombiner &IC,
return &II;
}
+// zext(cmpne(ptrue, %v, 0))
+// -> umin(ptrue, %v, 1)
+static std::optional<Instruction *> instCombineZExtSVECmpNE(InstCombiner &IC,
+ IntrinsicInst &II) {
+ if (!isAllActivePredicate(II.getOperand(0)) ||
+ !match(II.getOperand(2), m_Zero()) || !II.hasOneUse())
+ return std::nullopt;
+
+ auto *User = cast<Instruction>(*II.user_begin());
+ if (!match(User, m_ZExt(m_Specific(&II))))
+ return std::nullopt;
+
+ IC.Builder.SetInsertPoint(User);
+ Value *UMIN = IC.Builder.CreateIntrinsic(
+ Intrinsic::aarch64_sve_umin, II.getOperand(1)->getType(),
+ {II.getOperand(0), II.getOperand(1),
+ ConstantInt::get(II.getOperand(1)->getType(), 1)});
+ IC.replaceInstUsesWith(*User, UMIN);
+ IC.eraseInstFromFunction(*User);
+ return &II;
+}
+
static std::optional<Instruction *> instCombineSVECmpNE(InstCombiner &IC,
IntrinsicInst &II) {
LLVMContext &Ctx = II.getContext();
@@ -2180,6 +2202,9 @@ static std::optional<Instruction *> instCombineSVECmpNE(InstCombiner &IC,
if (auto Res = instCombineXorSVECmpNE(IC, II))
return Res;
+ if (auto Res = instCombineZExtSVECmpNE(IC, II))
+ return Res;
+
if (!isAllActivePredicate(II.getArgOperand(0)))
return std::nullopt;
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
index e803572a06523..f3e1c67511347 100644
--- a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
@@ -261,6 +261,66 @@ define <vscale x 16 x i1> @not_cmpne_wrong_xor_operand(<vscale x 16 x i8> %vec_a
ret <vscale x 16 x i1> %not
}
+define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_i8(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ZEXT:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.umin.u.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
+; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
+;
+ %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
+ %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %zext
+}
+
+define <vscale x 4 x i32> @zext_cmpne_wide_i32(<vscale x 4 x i32> %vec) #0 {
+; CHECK-LABEL: define <vscale x 4 x i32> @zext_cmpne_wide_i32(
+; CHECK-SAME: <vscale x 4 x i32> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ZEXT:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.umin.u.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> [[VEC]], <vscale x 4 x i32> splat (i32 1))
+; CHECK-NEXT: ret <vscale x 4 x i32> [[ZEXT]]
+;
+ %cmp = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> %vec, <vscale x 2 x i64> zeroinitializer)
+ %zext = zext <vscale x 4 x i1> %cmp to <vscale x 4 x i32>
+ ret <vscale x 4 x i32> %zext
+}
+
+define <vscale x 16 x i8> @zext_cmpne_non_ptrue(<vscale x 16 x i8> %vec, <vscale x 16 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_non_ptrue(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]], <vscale x 16 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[CMP:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> [[PG]], <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> zeroinitializer)
+; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i8>
+; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
+;
+ %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
+ %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %zext
+}
+
+define <vscale x 16 x i8> @zext_cmpne_nonzero_rhs(<vscale x 16 x i8> %vec) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_nonzero_rhs(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[CMP:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
+; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i8>
+; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
+;
+ %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> splat (i8 1))
+ %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %zext
+}
+
+define <vscale x 16 x i8> @zext_cmpne_multiple_uses(<vscale x 16 x i8> %vec) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_multiple_uses(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[CMP:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> zeroinitializer)
+; CHECK-NEXT: call void (...) @llvm.fake.use(<vscale x 16 x i1> [[CMP]])
+; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i8>
+; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
+;
+ %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
+ call void (...) @llvm.fake.use(<vscale x 16 x i1> %cmp)
+ %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %zext
+}
+
; Cases that cannot be converted
define <vscale x 2 x i1> @dupq_neg1() #0 {
>From 9ff1cacd2b7f3253bb18928181103eea5c4b07c7 Mon Sep 17 00:00:00 2001
From: Matt Devereau <matthew.devereau at arm.com>
Date: Wed, 8 Jul 2026 17:25:17 +0000
Subject: [PATCH 2/3] Add commutability, use Intrinsic::min
---
.../AArch64/AArch64TargetTransformInfo.cpp | 20 ++++++++++++-------
.../AArch64/sve-intrinsic-opts-cmpne.ll | 18 +++++++++++++++--
2 files changed, 29 insertions(+), 9 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 5e9f56a83562c..7bb595062fab4 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -2177,20 +2177,26 @@ static std::optional<Instruction *> instCombineXorSVECmpNE(InstCombiner &IC,
// -> umin(ptrue, %v, 1)
static std::optional<Instruction *> instCombineZExtSVECmpNE(InstCombiner &IC,
IntrinsicInst &II) {
- if (!isAllActivePredicate(II.getOperand(0)) ||
- !match(II.getOperand(2), m_Zero()) || !II.hasOneUse())
+ if (!isAllActivePredicate(II.getOperand(0)) || !II.hasOneUse())
return std::nullopt;
+ Value *Zero = II.getOperand(1);
+ Value *Op = II.getOperand(2);
+ if (!match(Zero, m_Zero())) {
+ std::swap(Zero, Op);
+ if (!match(Zero, m_Zero()))
+ return std::nullopt;
+ }
+
auto *User = cast<Instruction>(*II.user_begin());
if (!match(User, m_ZExt(m_Specific(&II))))
return std::nullopt;
IC.Builder.SetInsertPoint(User);
- Value *UMIN = IC.Builder.CreateIntrinsic(
- Intrinsic::aarch64_sve_umin, II.getOperand(1)->getType(),
- {II.getOperand(0), II.getOperand(1),
- ConstantInt::get(II.getOperand(1)->getType(), 1)});
- IC.replaceInstUsesWith(*User, UMIN);
+ Value *UMin =
+ IC.Builder.CreateIntrinsic(Intrinsic::umin, {Op->getType()},
+ {Op, ConstantInt::get(Op->getType(), 1)});
+ IC.replaceInstUsesWith(*User, UMin);
IC.eraseInstFromFunction(*User);
return &II;
}
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
index f3e1c67511347..c20ed9354d0ae 100644
--- a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
@@ -264,7 +264,8 @@ define <vscale x 16 x i1> @not_cmpne_wrong_xor_operand(<vscale x 16 x i8> %vec_a
define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_i8(
; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ZEXT:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.umin.u.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <vscale x 16 x i8> [[VEC]], zeroinitializer
+; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[TMP1]] to <vscale x 16 x i8>
; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
;
%cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
@@ -272,10 +273,23 @@ define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
ret <vscale x 16 x i8> %zext
}
+define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(<vscale x 16 x i8> %vec) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <vscale x 16 x i8> [[VEC]], zeroinitializer
+; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[TMP1]] to <vscale x 16 x i8>
+; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
+;
+ %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> zeroinitializer, <vscale x 16 x i8> %vec)
+ %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %zext
+}
+
define <vscale x 4 x i32> @zext_cmpne_wide_i32(<vscale x 4 x i32> %vec) #0 {
; CHECK-LABEL: define <vscale x 4 x i32> @zext_cmpne_wide_i32(
; CHECK-SAME: <vscale x 4 x i32> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ZEXT:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.umin.u.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> [[VEC]], <vscale x 4 x i32> splat (i32 1))
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <vscale x 4 x i32> [[VEC]], zeroinitializer
+; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 4 x i1> [[TMP1]] to <vscale x 4 x i32>
; CHECK-NEXT: ret <vscale x 4 x i32> [[ZEXT]]
;
%cmp = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> %vec, <vscale x 2 x i64> zeroinitializer)
>From e1b7dc8a836285b92c10db1ef4c175205f4a45a0 Mon Sep 17 00:00:00 2001
From: Matt Devereau <matthew.devereau at arm.com>
Date: Thu, 9 Jul 2026 12:53:36 +0000
Subject: [PATCH 3/3] Remove commutativity, revert back to aarch64.sve.umin,
drop one use
---
.../AArch64/AArch64TargetTransformInfo.cpp | 39 ++++++++++---------
.../AArch64/sve-intrinsic-opts-cmpne.ll | 32 +++++++++------
2 files changed, 40 insertions(+), 31 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 7bb595062fab4..419491cc0b42e 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -2174,31 +2174,32 @@ static std::optional<Instruction *> instCombineXorSVECmpNE(InstCombiner &IC,
}
// zext(cmpne(ptrue, %v, 0))
-// -> umin(ptrue, %v, 1)
+// -> umin(%pg, %v, 1)
static std::optional<Instruction *> instCombineZExtSVECmpNE(InstCombiner &IC,
IntrinsicInst &II) {
- if (!isAllActivePredicate(II.getOperand(0)) || !II.hasOneUse())
+ if (!isAllActivePredicate(II.getOperand(0)) ||
+ !match(II.getOperand(2), m_Zero()))
return std::nullopt;
- Value *Zero = II.getOperand(1);
- Value *Op = II.getOperand(2);
- if (!match(Zero, m_Zero())) {
- std::swap(Zero, Op);
- if (!match(Zero, m_Zero()))
- return std::nullopt;
+ bool Changed = false;
+ for (auto *U : II.users()) {
+ if (match(U, m_ZExt(m_Specific(&II)))) {
+ auto *User = cast<Instruction>(U);
+ Type *Ty = II.getOperand(1)->getType();
+ if (User->getType() != Ty)
+ continue;
+ IC.Builder.SetInsertPoint(User);
+ Value *UMin = IC.Builder.CreateIntrinsic(
+ Intrinsic::aarch64_sve_umin, Ty,
+ {II.getOperand(0), II.getOperand(1), ConstantInt::get(Ty, 1)});
+ IC.replaceInstUsesWith(*User, UMin);
+ IC.eraseInstFromFunction(*User);
+ Changed = true;
+ } else
+ continue;
}
- auto *User = cast<Instruction>(*II.user_begin());
- if (!match(User, m_ZExt(m_Specific(&II))))
- return std::nullopt;
-
- IC.Builder.SetInsertPoint(User);
- Value *UMin =
- IC.Builder.CreateIntrinsic(Intrinsic::umin, {Op->getType()},
- {Op, ConstantInt::get(Op->getType(), 1)});
- IC.replaceInstUsesWith(*User, UMin);
- IC.eraseInstFromFunction(*User);
- return &II;
+ return Changed ? std::optional<Instruction *>(&II) : std::nullopt;
}
static std::optional<Instruction *> instCombineSVECmpNE(InstCombiner &IC,
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
index c20ed9354d0ae..11e6e0089293a 100644
--- a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
@@ -264,8 +264,7 @@ define <vscale x 16 x i1> @not_cmpne_wrong_xor_operand(<vscale x 16 x i8> %vec_a
define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_i8(
; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <vscale x 16 x i8> [[VEC]], zeroinitializer
-; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[TMP1]] to <vscale x 16 x i8>
+; CHECK-NEXT: [[ZEXT:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.umin.u.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
;
%cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
@@ -276,7 +275,7 @@ define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(<vscale x 16 x i8> %vec) #0 {
; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(
; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <vscale x 16 x i8> [[VEC]], zeroinitializer
+; CHECK-NEXT: [[TMP1:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> zeroinitializer, <vscale x 16 x i8> [[VEC]])
; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[TMP1]] to <vscale x 16 x i8>
; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
;
@@ -288,8 +287,7 @@ define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(<vscale x 16 x i8> %vec) #0 {
define <vscale x 4 x i32> @zext_cmpne_wide_i32(<vscale x 4 x i32> %vec) #0 {
; CHECK-LABEL: define <vscale x 4 x i32> @zext_cmpne_wide_i32(
; CHECK-SAME: <vscale x 4 x i32> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <vscale x 4 x i32> [[VEC]], zeroinitializer
-; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 4 x i1> [[TMP1]] to <vscale x 4 x i32>
+; CHECK-NEXT: [[ZEXT:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.umin.u.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> [[VEC]], <vscale x 4 x i32> splat (i32 1))
; CHECK-NEXT: ret <vscale x 4 x i32> [[ZEXT]]
;
%cmp = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> %vec, <vscale x 2 x i64> zeroinitializer)
@@ -321,18 +319,28 @@ define <vscale x 16 x i8> @zext_cmpne_nonzero_rhs(<vscale x 16 x i8> %vec) #0 {
ret <vscale x 16 x i8> %zext
}
-define <vscale x 16 x i8> @zext_cmpne_multiple_uses(<vscale x 16 x i8> %vec) #0 {
+define <vscale x 16 x i8> @zext_cmpne_multiple_uses(<vscale x 16 x i8> %vec, <vscale x 16 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_multiple_uses(
-; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]], <vscale x 16 x i1> [[PG:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[CMP:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> zeroinitializer)
-; CHECK-NEXT: call void (...) @llvm.fake.use(<vscale x 16 x i1> [[CMP]])
-; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i8>
+; CHECK-NEXT: [[AND_1:%.*]] = and <vscale x 16 x i1> [[CMP]], [[PG]]
+; CHECK-NEXT: call void (...) @llvm.fake.use(<vscale x 16 x i1> [[AND_1]])
+; CHECK-NEXT: [[ZEXT:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.umin.u.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
+; CHECK-NEXT: [[ZEXT_1:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i32>
+; CHECK-NEXT: call void (...) @llvm.fake.use(<vscale x 16 x i32> [[ZEXT_1]])
; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
;
%cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
- call void (...) @llvm.fake.use(<vscale x 16 x i1> %cmp)
- %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
- ret <vscale x 16 x i8> %zext
+
+ %and.1 = and <vscale x 16 x i1> %cmp, %pg
+ call void (...) @llvm.fake.use(<vscale x 16 x i1> %and.1)
+
+ %zext.0 = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+
+ %zext.1 = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i32>
+ call void (...) @llvm.fake.use(<vscale x 16 x i32> %zext.1)
+
+ ret <vscale x 16 x i8> %zext.0
}
; Cases that cannot be converted
More information about the llvm-commits
mailing list