[llvm] [AArch64][InstCombine] Fold zext of all-active SVE cmpne-zero (PR #207720)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Jul 6 05:37:38 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
@llvm/pr-subscribers-backend-aarch64
Author: Matthew Devereau (MDevereau)
<details>
<summary>Changes</summary>
Fold zext(cmpne(ptrue, x, 0)) to umin(ptrue, x, 1).
Do not fold non-ptrue predicates, since cmpne zeros inactive lanes and umin merges inactive lanes
---
Full diff: https://github.com/llvm/llvm-project/pull/207720.diff
2 Files Affected:
- (modified) llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp (+25)
- (modified) llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll (+60)
``````````diff
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index b7d8a8ff4657b..5e9f56a83562c 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -2173,6 +2173,28 @@ static std::optional<Instruction *> instCombineXorSVECmpNE(InstCombiner &IC,
return &II;
}
+// zext(cmpne(ptrue, %v, 0))
+// -> umin(ptrue, %v, 1)
+static std::optional<Instruction *> instCombineZExtSVECmpNE(InstCombiner &IC,
+ IntrinsicInst &II) {
+ if (!isAllActivePredicate(II.getOperand(0)) ||
+ !match(II.getOperand(2), m_Zero()) || !II.hasOneUse())
+ return std::nullopt;
+
+ auto *User = cast<Instruction>(*II.user_begin());
+ if (!match(User, m_ZExt(m_Specific(&II))))
+ return std::nullopt;
+
+ IC.Builder.SetInsertPoint(User);
+ Value *UMIN = IC.Builder.CreateIntrinsic(
+ Intrinsic::aarch64_sve_umin, II.getOperand(1)->getType(),
+ {II.getOperand(0), II.getOperand(1),
+ ConstantInt::get(II.getOperand(1)->getType(), 1)});
+ IC.replaceInstUsesWith(*User, UMIN);
+ IC.eraseInstFromFunction(*User);
+ return &II;
+}
+
static std::optional<Instruction *> instCombineSVECmpNE(InstCombiner &IC,
IntrinsicInst &II) {
LLVMContext &Ctx = II.getContext();
@@ -2180,6 +2202,9 @@ static std::optional<Instruction *> instCombineSVECmpNE(InstCombiner &IC,
if (auto Res = instCombineXorSVECmpNE(IC, II))
return Res;
+ if (auto Res = instCombineZExtSVECmpNE(IC, II))
+ return Res;
+
if (!isAllActivePredicate(II.getArgOperand(0)))
return std::nullopt;
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
index e803572a06523..f3e1c67511347 100644
--- a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
@@ -261,6 +261,66 @@ define <vscale x 16 x i1> @not_cmpne_wrong_xor_operand(<vscale x 16 x i8> %vec_a
ret <vscale x 16 x i1> %not
}
+define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_i8(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ZEXT:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.umin.u.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
+; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
+;
+ %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
+ %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %zext
+}
+
+define <vscale x 4 x i32> @zext_cmpne_wide_i32(<vscale x 4 x i32> %vec) #0 {
+; CHECK-LABEL: define <vscale x 4 x i32> @zext_cmpne_wide_i32(
+; CHECK-SAME: <vscale x 4 x i32> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ZEXT:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.umin.u.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> [[VEC]], <vscale x 4 x i32> splat (i32 1))
+; CHECK-NEXT: ret <vscale x 4 x i32> [[ZEXT]]
+;
+ %cmp = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> %vec, <vscale x 2 x i64> zeroinitializer)
+ %zext = zext <vscale x 4 x i1> %cmp to <vscale x 4 x i32>
+ ret <vscale x 4 x i32> %zext
+}
+
+define <vscale x 16 x i8> @zext_cmpne_non_ptrue(<vscale x 16 x i8> %vec, <vscale x 16 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_non_ptrue(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]], <vscale x 16 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[CMP:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> [[PG]], <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> zeroinitializer)
+; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i8>
+; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
+;
+ %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
+ %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %zext
+}
+
+define <vscale x 16 x i8> @zext_cmpne_nonzero_rhs(<vscale x 16 x i8> %vec) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_nonzero_rhs(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[CMP:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
+; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i8>
+; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
+;
+ %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> splat (i8 1))
+ %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %zext
+}
+
+define <vscale x 16 x i8> @zext_cmpne_multiple_uses(<vscale x 16 x i8> %vec) #0 {
+; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_multiple_uses(
+; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[CMP:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> zeroinitializer)
+; CHECK-NEXT: call void (...) @llvm.fake.use(<vscale x 16 x i1> [[CMP]])
+; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[CMP]] to <vscale x 16 x i8>
+; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
+;
+ %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> %vec, <vscale x 16 x i8> zeroinitializer)
+ call void (...) @llvm.fake.use(<vscale x 16 x i1> %cmp)
+ %zext = zext <vscale x 16 x i1> %cmp to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %zext
+}
+
; Cases that cannot be converted
define <vscale x 2 x i1> @dupq_neg1() #0 {
``````````
</details>
https://github.com/llvm/llvm-project/pull/207720
More information about the llvm-commits
mailing list