[llvm] [LLVM][InstCombine] Add simplification of SVE compare intrinsics. (PR #211249)

Paul Walker via llvm-commits llvm-commits at lists.llvm.org
Tue Jul 28 08:51:31 PDT 2026


https://github.com/paulwalker-arm updated https://github.com/llvm/llvm-project/pull/211249

>From b5c203dcc88631c860250c0cc0c1075a7405837f Mon Sep 17 00:00:00 2001
From: Paul Walker <paul.walker at arm.com>
Date: Wed, 22 Jul 2026 12:22:43 +0000
Subject: [PATCH 1/3] Add tests for compare simplification.

---
 .../AArch64/sve-intrinsic-simplify-cmp.ll     | 335 ++++++++++++++++++
 1 file changed, 335 insertions(+)
 create mode 100644 llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll

diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll
new file mode 100644
index 0000000000000..9efa9dee1b6cb
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll
@@ -0,0 +1,335 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=instcombine < %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; The following tests verify the mechanics of simplification. The operation is
+; not important beyond being commutative.
+
+define <vscale x 4 x i1> @cannonicalise_constant_to_rhs(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @cannonicalise_constant_to_rhs(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i32> [[A:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> [[A]])
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> %a)
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @cannot_cannonicalise_constant_to_rhs_not_commutative_int(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @cannot_cannonicalise_constant_to_rhs_not_commutative_int(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i32> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> [[A]])
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> %a)
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @cannot_cannonicalise_constant_to_rhs_operand_type_mismatch(<vscale x 4 x i1> %pg, <vscale x 2 x i64> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @cannot_cannonicalise_constant_to_rhs_operand_type_mismatch(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 2 x i64> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> [[A]])
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> %a)
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @cannonicalise_fp_constant_to_rhs(<vscale x 4 x i1> %pg, <vscale x 4 x float> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @cannonicalise_fp_constant_to_rhs(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x float> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> [[A]])
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> %a)
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @cannot_cannonicalise_fp_constant_to_rhs_not_commutative_fp(<vscale x 4 x i1> %pg, <vscale x 4 x float> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @cannot_cannonicalise_fp_constant_to_rhs_not_commutative_fp(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x float> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> [[A]])
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> %a)
+  ret <vscale x 4 x i1> %r
+}
+
+; Show that we only need to know the active lanes are constant.
+define <vscale x 4 x i1> @constant_icmp_after_striping_inactive_lanes(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a, <vscale x 4 x i32> %b) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmp_after_striping_inactive_lanes(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i32> [[A:%.*]], <vscale x 4 x i32> [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[A_DUP:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> [[A]], <vscale x 4 x i1> [[PG]], i32 3)
+; CHECK-NEXT:    [[B_DUP:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> [[B]], <vscale x 4 x i1> [[PG]], i32 2)
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> [[A_DUP]], <vscale x 4 x i32> [[B_DUP]])
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %a.dup = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> %a, <vscale x 4 x i1> %pg, i32 3)
+  %b.dup = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> %b, <vscale x 4 x i1> %pg, i32 2)
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a.dup, <vscale x 4 x i32> %b.dup)
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 16 x i1> @constant_icmp_due_to_range_of_type(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a) #0 {
+; CHECK-LABEL: define <vscale x 16 x i1> @constant_icmp_due_to_range_of_type(
+; CHECK-SAME: <vscale x 16 x i1> [[PG:%.*]], <vscale x 16 x i8> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.nxv16i8(<vscale x 16 x i1> [[PG]], <vscale x 16 x i8> [[A]], <vscale x 16 x i8> splat (i8 127))
+; CHECK-NEXT:    ret <vscale x 16 x i1> [[R]]
+;
+  %r = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.nxv16i8(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a, <vscale x 16 x i8> splat (i8 127))
+  ret <vscale x 16 x i1> %r
+}
+
+; Wide compares implicitly promote the smaller typed operand to the larger type.
+define <vscale x 16 x i1> @non_constant_icmp_wide_due_to_range_of_type(<vscale x 16 x i1> %pg, <vscale x 2 x i64> %a) #0 {
+; CHECK-LABEL: define <vscale x 16 x i1> @non_constant_icmp_wide_due_to_range_of_type(
+; CHECK-SAME: <vscale x 16 x i1> [[PG:%.*]], <vscale x 2 x i64> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv16i8(<vscale x 16 x i1> [[PG]], <vscale x 16 x i8> splat (i8 127), <vscale x 2 x i64> [[A]])
+; CHECK-NEXT:    ret <vscale x 16 x i1> [[R]]
+;
+  %r = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv16i8(<vscale x 16 x i1> %pg, <vscale x 16 x i8> splat (i8 127), <vscale x 2 x i64> %a)
+  ret <vscale x 16 x i1> %r
+}
+
+; TODO: We can do better here.
+; Wide compares implicitly promote the smaller typed operand to the larger type,
+; but with the larger operand being constant we know it is too big.
+define <vscale x 16 x i1> @constant_icmp_wide_due_to_range_of_type(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a) #0 {
+; CHECK-LABEL: define <vscale x 16 x i1> @constant_icmp_wide_due_to_range_of_type(
+; CHECK-SAME: <vscale x 16 x i1> [[PG:%.*]], <vscale x 16 x i8> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv16i8(<vscale x 16 x i1> [[PG]], <vscale x 16 x i8> [[A]], <vscale x 2 x i64> splat (i64 127))
+; CHECK-NEXT:    ret <vscale x 16 x i1> [[R]]
+;
+  %r = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv16i8(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a, <vscale x 2 x i64> splat (i64 127))
+  ret <vscale x 16 x i1> %r
+}
+
+; Ensure inactive lanes are zero'd after simplification.
+define <vscale x 4 x i1> @non_constant_simplification(<vscale x 4 x i1> %pg, <vscale x 4 x i1> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @non_constant_simplification(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i1> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[A_EXT:%.*]] = zext <vscale x 4 x i1> [[A]] to <vscale x 4 x i32>
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> [[A_EXT]], <vscale x 4 x i32> zeroinitializer)
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %a.ext = zext <vscale x 4 x i1> %a to <vscale x 4 x i32>
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a.ext, <vscale x 4 x i32> splat (i32 0))
+  ret <vscale x 4 x i1> %r
+}
+
+; The following tests demonstrate the operations for which hooks are in place to
+; enable simplification. Given the simplications themselves are common code, it
+; is assumed they are already well tested elsewhere.
+
+define <vscale x 4 x i1> @constant_fcmpeq(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpeq(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpeq.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpeq.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_fcmpge(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpge(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_fcmpgt(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpgt(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpgt.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float -5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpgt.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float -5.0), <vscale x 4 x float> splat (float 3.0))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_fcmpne(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpne(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_fcmpuo(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpuo(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpuo.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpuo.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpeq(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpeq(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpeq_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpeq_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpge(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpge(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 -303), <vscale x 4 x i32> splat (i32 777))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 -303), <vscale x 4 x i32> splat (i32 777))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpge_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpge_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpgt(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpgt(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 -777))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 -777))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpgt_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpgt_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmphi(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphi(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmphi_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphi_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmphs(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphs(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmphs_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphs_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmple_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmple_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmple.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 -303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmple.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 -303), <vscale x 2 x i64> splat (i64 303))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmplo_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmplo_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplo.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 777), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplo.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 777), <vscale x 2 x i64> splat (i64 303))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpls_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpls_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpls.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 777))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpls.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 777))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmplt_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmplt_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplt.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplt.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpne(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpne(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+  ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpne_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpne_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+;
+  %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+  ret <vscale x 4 x i1> %r
+}
+
+attributes #0 = { "target-features"="+sve" }

>From f1271b40cad20be5b80c26e39cd3ab12e5c7421b Mon Sep 17 00:00:00 2001
From: Paul Walker <paul.walker at arm.com>
Date: Wed, 22 Jul 2026 12:37:36 +0100
Subject: [PATCH 2/3] [LLVM][InstCombine] Add simplification of SVE compare
 intrinsics.

Extends SVEIntrinsicInfo to accept LLVM IR compare information, which
is then used to call simplifyCmpInst on the data operands of SVE
compare intrinsic calls.
---
 .../AArch64/AArch64TargetTransformInfo.cpp    | 183 +++++++++++++++---
 .../AArch64/sve-intrinsic-opts-cmpne.ll       |   3 +-
 .../AArch64/sve-intrinsic-simplify-cmp.ll     |  78 +++-----
 3 files changed, 187 insertions(+), 77 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 03aa42b31e6aa..d1f8365839777 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -1431,6 +1431,28 @@ struct SVEIntrinsicInfo {
     return *this;
   }
 
+  bool hasCmpPredicate() const {
+    return CmpPredicate != CmpInst::BAD_ICMP_PREDICATE;
+  }
+
+  CmpInst::Predicate getCmpPredicate() const {
+    assert(hasCmpPredicate() && "Property not set!");
+    return CmpPredicate;
+  }
+
+  SVEIntrinsicInfo &setCmpPredicate(CmpInst::Predicate Pred) {
+    assert(!hasCmpPredicate() && "Cannot set property twice!");
+    CmpPredicate = Pred;
+
+    if (CmpInst::isFPPredicate(Pred))
+      return setMatchingIROpcode(Instruction::FCmp);
+
+    if (CmpInst::isIntPredicate(Pred))
+      return setMatchingIROpcode(Instruction::ICmp);
+
+    llvm_unreachable("Unsupported compare predicate!");
+  }
+
   //
   // Properties relating to the result of inactive lanes.
   //
@@ -1507,6 +1529,7 @@ struct SVEIntrinsicInfo {
 
   Intrinsic::ID UndefIntrinsic = Intrinsic::not_intrinsic;
   unsigned IROpcode = 0;
+  CmpInst::Predicate CmpPredicate = CmpInst::BAD_ICMP_PREDICATE;
 
   enum PredicationStyle {
     Uninitialized,
@@ -1755,29 +1778,8 @@ static SVEIntrinsicInfo constructSVEIntrinsicInfo(IntrinsicInst &II) {
   case Intrinsic::aarch64_sve_uaddv:
   case Intrinsic::aarch64_sve_umaxv:
   case Intrinsic::aarch64_sve_umaxqv:
-  case Intrinsic::aarch64_sve_cmpeq:
-  case Intrinsic::aarch64_sve_cmpeq_wide:
-  case Intrinsic::aarch64_sve_cmpge:
-  case Intrinsic::aarch64_sve_cmpge_wide:
-  case Intrinsic::aarch64_sve_cmpgt:
-  case Intrinsic::aarch64_sve_cmpgt_wide:
-  case Intrinsic::aarch64_sve_cmphi:
-  case Intrinsic::aarch64_sve_cmphi_wide:
-  case Intrinsic::aarch64_sve_cmphs:
-  case Intrinsic::aarch64_sve_cmphs_wide:
-  case Intrinsic::aarch64_sve_cmple_wide:
-  case Intrinsic::aarch64_sve_cmplo_wide:
-  case Intrinsic::aarch64_sve_cmpls_wide:
-  case Intrinsic::aarch64_sve_cmplt_wide:
-  case Intrinsic::aarch64_sve_cmpne:
-  case Intrinsic::aarch64_sve_cmpne_wide:
   case Intrinsic::aarch64_sve_facge:
   case Intrinsic::aarch64_sve_facgt:
-  case Intrinsic::aarch64_sve_fcmpeq:
-  case Intrinsic::aarch64_sve_fcmpge:
-  case Intrinsic::aarch64_sve_fcmpgt:
-  case Intrinsic::aarch64_sve_fcmpne:
-  case Intrinsic::aarch64_sve_fcmpuo:
   case Intrinsic::aarch64_sve_ld1:
   case Intrinsic::aarch64_sve_ld1_gather:
   case Intrinsic::aarch64_sve_ld1_gather_index:
@@ -1825,6 +1827,58 @@ static SVEIntrinsicInfo constructSVEIntrinsicInfo(IntrinsicInst &II) {
     return SVEIntrinsicInfo::defaultZeroingOp().setMatchingIROpcode(
         Instruction::Xor);
 
+  case Intrinsic::aarch64_sve_cmpeq:
+  case Intrinsic::aarch64_sve_cmpeq_wide:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::ICMP_EQ);
+  case Intrinsic::aarch64_sve_cmpge:
+  case Intrinsic::aarch64_sve_cmpge_wide:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::ICMP_SGE);
+  case Intrinsic::aarch64_sve_cmpgt:
+  case Intrinsic::aarch64_sve_cmpgt_wide:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::ICMP_SGT);
+  case Intrinsic::aarch64_sve_cmphi:
+  case Intrinsic::aarch64_sve_cmphi_wide:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::ICMP_UGT);
+  case Intrinsic::aarch64_sve_cmphs:
+  case Intrinsic::aarch64_sve_cmphs_wide:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::ICMP_UGE);
+  case Intrinsic::aarch64_sve_cmple_wide:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::ICMP_SLE);
+  case Intrinsic::aarch64_sve_cmplo_wide:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::ICMP_ULT);
+  case Intrinsic::aarch64_sve_cmpls_wide:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::ICMP_ULE);
+  case Intrinsic::aarch64_sve_cmplt_wide:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::ICMP_SLT);
+  case Intrinsic::aarch64_sve_cmpne:
+  case Intrinsic::aarch64_sve_cmpne_wide:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::ICMP_NE);
+  case Intrinsic::aarch64_sve_fcmpeq:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::FCMP_OEQ);
+  case Intrinsic::aarch64_sve_fcmpge:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::FCMP_OGE);
+  case Intrinsic::aarch64_sve_fcmpgt:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::FCMP_OGT);
+  case Intrinsic::aarch64_sve_fcmpne:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::FCMP_UNE);
+  case Intrinsic::aarch64_sve_fcmpuo:
+    return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+        CmpInst::FCMP_UNO);
+
   case Intrinsic::aarch64_sve_prf:
   case Intrinsic::aarch64_sve_prfb_gather_index:
   case Intrinsic::aarch64_sve_prfb_gather_scalar_offset:
@@ -1964,6 +2018,79 @@ simplifySVEIntrinsicBinOp(InstCombiner &IC, IntrinsicInst &II,
   return IC.replaceInstUsesWith(II, SimpleII);
 }
 
+static std::optional<Instruction *>
+simplifySVEIntrinsicCompare(InstCombiner &IC, IntrinsicInst &II,
+                            const SVEIntrinsicInfo &IInfo) {
+  const unsigned Opc = IInfo.getMatchingIROpode();
+  assert((Opc == Instruction::ICmp || Opc == Instruction::FCmp) &&
+         "Expected a compare operation!");
+
+  Value *Pg = II.getOperand(0);
+  Value *LHS = II.getOperand(1);
+  Value *RHS = II.getOperand(2);
+  CmpInst::Predicate CmpPred = IInfo.getCmpPredicate();
+  const DataLayout &DL = II.getDataLayout();
+
+  // Canonicalise integer constants to the RHS.
+  if (Opc == Instruction::ICmp && ICmpInst::isCommutative(CmpPred) &&
+      isa<Constant>(LHS) && !isa<Constant>(RHS) &&
+      LHS->getType() == RHS->getType()) {
+    IC.replaceOperand(II, 1, RHS);
+    IC.replaceOperand(II, 2, LHS);
+    return &II;
+  }
+
+  // Canonicalise floating-point constants to the RHS.
+  if (Opc == Instruction::FCmp && FCmpInst::isCommutative(CmpPred) &&
+      isa<Constant>(LHS) && !isa<Constant>(RHS)) {
+    assert(LHS->getType() == RHS->getType() && "Unexpected wide compare!");
+    IC.replaceOperand(II, 1, RHS);
+    IC.replaceOperand(II, 2, LHS);
+    return &II;
+  }
+
+  // Only active lanes matter when simplifying the operation.
+  LHS = stripInactiveLanes(LHS, Pg);
+  RHS = stripInactiveLanes(RHS, Pg);
+
+  if (LHS->getType() != RHS->getType()) {
+    // We can do more for wide compares, but not using simplifyCmpInst.
+    const APInt *LHSVal, *RHSVal;
+    if (!match(LHS, m_APInt(LHSVal)) || !match(RHS, m_APInt(RHSVal)))
+      return std::nullopt;
+
+    // Consider cmpge.wide(..., <vscale x 4 x i32> LHS, <vscale x 2 x i64> RHS),
+    // we must reconstruct the constants because LHS has the wrong element type,
+    // and RHS the wrong element count.
+    Type *WideVT = VectorType::get(RHS->getType()->getScalarType(),
+                                   cast<VectorType>(LHS->getType()));
+    assert(Opc == Instruction::ICmp && "Only wide integer compares exist!");
+    if (ICmpInst::isSigned(CmpPred)) {
+      LHS = ConstantInt::get(WideVT, LHSVal->getSExtValue());
+      RHS = ConstantInt::get(WideVT, RHSVal->getSExtValue());
+    } else {
+      LHS = ConstantInt::get(WideVT, LHSVal->getZExtValue());
+      RHS = ConstantInt::get(WideVT, RHSVal->getZExtValue());
+    }
+  }
+
+  // TODO: Allow fast-math flags for calls to compare intrinsics.
+  Value *SimpleII = simplifyCmpInst(CmpPred, LHS, RHS, DL);
+
+  // No simplification happened.
+  if (!SimpleII)
+    return std::nullopt;
+
+  assert(IInfo.resultIsZeroInitialized() && "Expected a zeroing operation!");
+
+  if (match(SimpleII, m_ZeroInt()))
+    return IC.replaceInstUsesWith(II, SimpleII);
+
+  // Inactive lanes must be zero'd.
+  SimpleII = IC.Builder.CreateLogicalAnd(Pg, SimpleII);
+  return IC.replaceInstUsesWith(II, SimpleII);
+}
+
 // Use SVE intrinsic info to eliminate redundant operands and/or canonicalise
 // to operations with less strict inactive lane requirements.
 static std::optional<Instruction *>
@@ -2004,11 +2131,21 @@ simplifySVEIntrinsic(InstCombiner &IC, IntrinsicInst &II,
     }
   }
 
+  if (!IInfo.hasMatchingIROpode())
+    return std::nullopt;
+
+  //
   // Operation specific simplifications.
-  if (IInfo.hasMatchingIROpode() &&
-      Instruction::isBinaryOp(IInfo.getMatchingIROpode()))
+  //
+
+  unsigned Opc = IInfo.getMatchingIROpode();
+
+  if (Instruction::isBinaryOp(Opc))
     return simplifySVEIntrinsicBinOp(IC, II, IInfo);
 
+  if (Opc == Instruction::FCmp || Opc == Instruction::ICmp)
+    return simplifySVEIntrinsicCompare(IC, II, IInfo);
+
   return std::nullopt;
 }
 
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
index 11e6e0089293a..d9a5434c35269 100644
--- a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
@@ -275,8 +275,7 @@ define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
 define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(<vscale x 16 x i8> %vec) #0 {
 ; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(
 ; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[TMP1:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> zeroinitializer, <vscale x 16 x i8> [[VEC]])
-; CHECK-NEXT:    [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[TMP1]] to <vscale x 16 x i8>
+; CHECK-NEXT:    [[ZEXT:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.umin.u.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
 ; CHECK-NEXT:    ret <vscale x 16 x i8> [[ZEXT]]
 ;
   %cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> zeroinitializer, <vscale x 16 x i8> %vec)
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll
index 9efa9dee1b6cb..b355c169ae9f6 100644
--- a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll
@@ -9,7 +9,7 @@ target triple = "aarch64-unknown-linux-gnu"
 define <vscale x 4 x i1> @cannonicalise_constant_to_rhs(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @cannonicalise_constant_to_rhs(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i32> [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> [[A]])
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> [[A]], <vscale x 4 x i32> splat (i32 303))
 ; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> %a)
@@ -39,7 +39,7 @@ define <vscale x 4 x i1> @cannot_cannonicalise_constant_to_rhs_operand_type_mism
 define <vscale x 4 x i1> @cannonicalise_fp_constant_to_rhs(<vscale x 4 x i1> %pg, <vscale x 4 x float> %a) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @cannonicalise_fp_constant_to_rhs(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x float> [[A:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> [[A]])
+; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> [[A]], <vscale x 4 x float> splat (float 5.000000e+00))
 ; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> %a)
@@ -60,10 +60,7 @@ define <vscale x 4 x i1> @cannot_cannonicalise_fp_constant_to_rhs_not_commutativ
 define <vscale x 4 x i1> @constant_icmp_after_striping_inactive_lanes(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a, <vscale x 4 x i32> %b) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmp_after_striping_inactive_lanes(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i32> [[A:%.*]], <vscale x 4 x i32> [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[A_DUP:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> [[A]], <vscale x 4 x i1> [[PG]], i32 3)
-; CHECK-NEXT:    [[B_DUP:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> [[B]], <vscale x 4 x i1> [[PG]], i32 2)
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> [[A_DUP]], <vscale x 4 x i32> [[B_DUP]])
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[PG]]
 ;
   %a.dup = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> %a, <vscale x 4 x i1> %pg, i32 3)
   %b.dup = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> %b, <vscale x 4 x i1> %pg, i32 2)
@@ -74,8 +71,7 @@ define <vscale x 4 x i1> @constant_icmp_after_striping_inactive_lanes(<vscale x
 define <vscale x 16 x i1> @constant_icmp_due_to_range_of_type(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a) #0 {
 ; CHECK-LABEL: define <vscale x 16 x i1> @constant_icmp_due_to_range_of_type(
 ; CHECK-SAME: <vscale x 16 x i1> [[PG:%.*]], <vscale x 16 x i8> [[A:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.nxv16i8(<vscale x 16 x i1> [[PG]], <vscale x 16 x i8> [[A]], <vscale x 16 x i8> splat (i8 127))
-; CHECK-NEXT:    ret <vscale x 16 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 16 x i1> zeroinitializer
 ;
   %r = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.nxv16i8(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a, <vscale x 16 x i8> splat (i8 127))
   ret <vscale x 16 x i1> %r
@@ -109,8 +105,7 @@ define <vscale x 16 x i1> @constant_icmp_wide_due_to_range_of_type(<vscale x 16
 define <vscale x 4 x i1> @non_constant_simplification(<vscale x 4 x i1> %pg, <vscale x 4 x i1> %a) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @non_constant_simplification(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i1> [[A:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[A_EXT:%.*]] = zext <vscale x 4 x i1> [[A]] to <vscale x 4 x i32>
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> [[A_EXT]], <vscale x 4 x i32> zeroinitializer)
+; CHECK-NEXT:    [[R:%.*]] = select <vscale x 4 x i1> [[PG]], <vscale x 4 x i1> [[A]], <vscale x 4 x i1> zeroinitializer
 ; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
 ;
   %a.ext = zext <vscale x 4 x i1> %a to <vscale x 4 x i32>
@@ -125,8 +120,7 @@ define <vscale x 4 x i1> @non_constant_simplification(<vscale x 4 x i1> %pg, <vs
 define <vscale x 4 x i1> @constant_fcmpeq(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpeq(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpeq.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> zeroinitializer
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpeq.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
   ret <vscale x 4 x i1> %r
@@ -135,8 +129,7 @@ define <vscale x 4 x i1> @constant_fcmpeq(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_fcmpge(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpge(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[PG]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
   ret <vscale x 4 x i1> %r
@@ -145,8 +138,7 @@ define <vscale x 4 x i1> @constant_fcmpge(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_fcmpgt(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpgt(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpgt.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float -5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> zeroinitializer
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpgt.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float -5.0), <vscale x 4 x float> splat (float 3.0))
   ret <vscale x 4 x i1> %r
@@ -155,8 +147,7 @@ define <vscale x 4 x i1> @constant_fcmpgt(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_fcmpne(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpne(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[PG]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
   ret <vscale x 4 x i1> %r
@@ -165,8 +156,7 @@ define <vscale x 4 x i1> @constant_fcmpne(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_fcmpuo(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpuo(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpuo.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> zeroinitializer
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpuo.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
   ret <vscale x 4 x i1> %r
@@ -175,8 +165,7 @@ define <vscale x 4 x i1> @constant_fcmpuo(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmpeq(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpeq(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> zeroinitializer
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
   ret <vscale x 4 x i1> %r
@@ -185,8 +174,7 @@ define <vscale x 4 x i1> @constant_icmpeq(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmpeq_wide(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpeq_wide(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[PG]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
   ret <vscale x 4 x i1> %r
@@ -195,8 +183,7 @@ define <vscale x 4 x i1> @constant_icmpeq_wide(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmpge(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpge(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 -303), <vscale x 4 x i32> splat (i32 777))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> zeroinitializer
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 -303), <vscale x 4 x i32> splat (i32 777))
   ret <vscale x 4 x i1> %r
@@ -205,8 +192,7 @@ define <vscale x 4 x i1> @constant_icmpge(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmpge_wide(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpge_wide(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[PG]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
   ret <vscale x 4 x i1> %r
@@ -215,8 +201,7 @@ define <vscale x 4 x i1> @constant_icmpge_wide(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmpgt(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpgt(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 -777))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[PG]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 -777))
   ret <vscale x 4 x i1> %r
@@ -225,8 +210,7 @@ define <vscale x 4 x i1> @constant_icmpgt(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmpgt_wide(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpgt_wide(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[PG]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
   ret <vscale x 4 x i1> %r
@@ -235,8 +219,7 @@ define <vscale x 4 x i1> @constant_icmpgt_wide(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmphi(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphi(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> zeroinitializer
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
   ret <vscale x 4 x i1> %r
@@ -245,8 +228,7 @@ define <vscale x 4 x i1> @constant_icmphi(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmphi_wide(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphi_wide(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> zeroinitializer
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
   ret <vscale x 4 x i1> %r
@@ -255,8 +237,7 @@ define <vscale x 4 x i1> @constant_icmphi_wide(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmphs(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphs(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> zeroinitializer
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
   ret <vscale x 4 x i1> %r
@@ -265,8 +246,7 @@ define <vscale x 4 x i1> @constant_icmphs(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmphs_wide(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphs_wide(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[PG]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
   ret <vscale x 4 x i1> %r
@@ -275,8 +255,7 @@ define <vscale x 4 x i1> @constant_icmphs_wide(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmple_wide(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmple_wide(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmple.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 -303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[PG]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmple.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 -303), <vscale x 2 x i64> splat (i64 303))
   ret <vscale x 4 x i1> %r
@@ -285,8 +264,7 @@ define <vscale x 4 x i1> @constant_icmple_wide(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmplo_wide(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmplo_wide(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplo.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 777), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> zeroinitializer
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplo.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 777), <vscale x 2 x i64> splat (i64 303))
   ret <vscale x 4 x i1> %r
@@ -295,8 +273,7 @@ define <vscale x 4 x i1> @constant_icmplo_wide(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmpls_wide(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpls_wide(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpls.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 777))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[PG]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpls.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 777))
   ret <vscale x 4 x i1> %r
@@ -305,8 +282,7 @@ define <vscale x 4 x i1> @constant_icmpls_wide(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmplt_wide(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmplt_wide(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplt.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> zeroinitializer
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplt.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
   ret <vscale x 4 x i1> %r
@@ -315,8 +291,7 @@ define <vscale x 4 x i1> @constant_icmplt_wide(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmpne(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpne(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> [[PG]]
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
   ret <vscale x 4 x i1> %r
@@ -325,8 +300,7 @@ define <vscale x 4 x i1> @constant_icmpne(<vscale x 4 x i1> %pg) #0 {
 define <vscale x 4 x i1> @constant_icmpne_wide(<vscale x 4 x i1> %pg) #0 {
 ; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpne_wide(
 ; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:    [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT:    ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT:    ret <vscale x 4 x i1> zeroinitializer
 ;
   %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
   ret <vscale x 4 x i1> %r

>From 966502960bb37eab7aebb277684fabb837d5b566 Mon Sep 17 00:00:00 2001
From: Paul Walker <paul.walker at arm.com>
Date: Tue, 28 Jul 2026 14:27:42 +0000
Subject: [PATCH 3/3] Make wide icmp handling more explicit.

---
 .../AArch64/AArch64TargetTransformInfo.cpp    | 25 +++++++------------
 1 file changed, 9 insertions(+), 16 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index d1f8365839777..07d2ba315356f 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -2029,21 +2029,14 @@ simplifySVEIntrinsicCompare(InstCombiner &IC, IntrinsicInst &II,
   Value *LHS = II.getOperand(1);
   Value *RHS = II.getOperand(2);
   CmpInst::Predicate CmpPred = IInfo.getCmpPredicate();
-  const DataLayout &DL = II.getDataLayout();
-
-  // Canonicalise integer constants to the RHS.
-  if (Opc == Instruction::ICmp && ICmpInst::isCommutative(CmpPred) &&
-      isa<Constant>(LHS) && !isa<Constant>(RHS) &&
-      LHS->getType() == RHS->getType()) {
-    IC.replaceOperand(II, 1, RHS);
-    IC.replaceOperand(II, 2, LHS);
-    return &II;
-  }
+  bool IsWideICmp =
+      Opc == Instruction::ICmp && LHS->getType() != RHS->getType();
+  assert((IsWideICmp || LHS->getType() == RHS->getType()) &&
+         "Unexpected wide compare!");
 
-  // Canonicalise floating-point constants to the RHS.
-  if (Opc == Instruction::FCmp && FCmpInst::isCommutative(CmpPred) &&
-      isa<Constant>(LHS) && !isa<Constant>(RHS)) {
-    assert(LHS->getType() == RHS->getType() && "Unexpected wide compare!");
+  // Canonicalise constants to the RHS.
+  if ((ICmpInst::isCommutative(CmpPred) || FCmpInst::isCommutative(CmpPred)) &&
+      isa<Constant>(LHS) && !isa<Constant>(RHS) && !IsWideICmp) {
     IC.replaceOperand(II, 1, RHS);
     IC.replaceOperand(II, 2, LHS);
     return &II;
@@ -2053,7 +2046,7 @@ simplifySVEIntrinsicCompare(InstCombiner &IC, IntrinsicInst &II,
   LHS = stripInactiveLanes(LHS, Pg);
   RHS = stripInactiveLanes(RHS, Pg);
 
-  if (LHS->getType() != RHS->getType()) {
+  if (IsWideICmp) {
     // We can do more for wide compares, but not using simplifyCmpInst.
     const APInt *LHSVal, *RHSVal;
     if (!match(LHS, m_APInt(LHSVal)) || !match(RHS, m_APInt(RHSVal)))
@@ -2064,7 +2057,6 @@ simplifySVEIntrinsicCompare(InstCombiner &IC, IntrinsicInst &II,
     // and RHS the wrong element count.
     Type *WideVT = VectorType::get(RHS->getType()->getScalarType(),
                                    cast<VectorType>(LHS->getType()));
-    assert(Opc == Instruction::ICmp && "Only wide integer compares exist!");
     if (ICmpInst::isSigned(CmpPred)) {
       LHS = ConstantInt::get(WideVT, LHSVal->getSExtValue());
       RHS = ConstantInt::get(WideVT, RHSVal->getSExtValue());
@@ -2075,6 +2067,7 @@ simplifySVEIntrinsicCompare(InstCombiner &IC, IntrinsicInst &II,
   }
 
   // TODO: Allow fast-math flags for calls to compare intrinsics.
+  const DataLayout &DL = II.getDataLayout();
   Value *SimpleII = simplifyCmpInst(CmpPred, LHS, RHS, DL);
 
   // No simplification happened.



More information about the llvm-commits mailing list