[llvm] [LLVM][InstCombine] Add simplification of SVE compare intrinsics. (PR #211249)
Paul Walker via llvm-commits
llvm-commits at lists.llvm.org
Tue Jul 28 08:51:31 PDT 2026
https://github.com/paulwalker-arm updated https://github.com/llvm/llvm-project/pull/211249
>From b5c203dcc88631c860250c0cc0c1075a7405837f Mon Sep 17 00:00:00 2001
From: Paul Walker <paul.walker at arm.com>
Date: Wed, 22 Jul 2026 12:22:43 +0000
Subject: [PATCH 1/3] Add tests for compare simplification.
---
.../AArch64/sve-intrinsic-simplify-cmp.ll | 335 ++++++++++++++++++
1 file changed, 335 insertions(+)
create mode 100644 llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll
new file mode 100644
index 0000000000000..9efa9dee1b6cb
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll
@@ -0,0 +1,335 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=instcombine < %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; The following tests verify the mechanics of simplification. The operation is
+; not important beyond being commutative.
+
+define <vscale x 4 x i1> @cannonicalise_constant_to_rhs(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @cannonicalise_constant_to_rhs(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i32> [[A:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> [[A]])
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> %a)
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @cannot_cannonicalise_constant_to_rhs_not_commutative_int(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @cannot_cannonicalise_constant_to_rhs_not_commutative_int(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i32> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> [[A]])
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> %a)
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @cannot_cannonicalise_constant_to_rhs_operand_type_mismatch(<vscale x 4 x i1> %pg, <vscale x 2 x i64> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @cannot_cannonicalise_constant_to_rhs_operand_type_mismatch(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 2 x i64> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> [[A]])
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> %a)
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @cannonicalise_fp_constant_to_rhs(<vscale x 4 x i1> %pg, <vscale x 4 x float> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @cannonicalise_fp_constant_to_rhs(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x float> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> [[A]])
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> %a)
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @cannot_cannonicalise_fp_constant_to_rhs_not_commutative_fp(<vscale x 4 x i1> %pg, <vscale x 4 x float> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @cannot_cannonicalise_fp_constant_to_rhs_not_commutative_fp(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x float> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> [[A]])
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> %a)
+ ret <vscale x 4 x i1> %r
+}
+
+; Show that we only need to know the active lanes are constant.
+define <vscale x 4 x i1> @constant_icmp_after_striping_inactive_lanes(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a, <vscale x 4 x i32> %b) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmp_after_striping_inactive_lanes(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i32> [[A:%.*]], <vscale x 4 x i32> [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[A_DUP:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> [[A]], <vscale x 4 x i1> [[PG]], i32 3)
+; CHECK-NEXT: [[B_DUP:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> [[B]], <vscale x 4 x i1> [[PG]], i32 2)
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> [[A_DUP]], <vscale x 4 x i32> [[B_DUP]])
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %a.dup = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> %a, <vscale x 4 x i1> %pg, i32 3)
+ %b.dup = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> %b, <vscale x 4 x i1> %pg, i32 2)
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a.dup, <vscale x 4 x i32> %b.dup)
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 16 x i1> @constant_icmp_due_to_range_of_type(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a) #0 {
+; CHECK-LABEL: define <vscale x 16 x i1> @constant_icmp_due_to_range_of_type(
+; CHECK-SAME: <vscale x 16 x i1> [[PG:%.*]], <vscale x 16 x i8> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.nxv16i8(<vscale x 16 x i1> [[PG]], <vscale x 16 x i8> [[A]], <vscale x 16 x i8> splat (i8 127))
+; CHECK-NEXT: ret <vscale x 16 x i1> [[R]]
+;
+ %r = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.nxv16i8(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a, <vscale x 16 x i8> splat (i8 127))
+ ret <vscale x 16 x i1> %r
+}
+
+; Wide compares implicitly promote the smaller typed operand to the larger type.
+define <vscale x 16 x i1> @non_constant_icmp_wide_due_to_range_of_type(<vscale x 16 x i1> %pg, <vscale x 2 x i64> %a) #0 {
+; CHECK-LABEL: define <vscale x 16 x i1> @non_constant_icmp_wide_due_to_range_of_type(
+; CHECK-SAME: <vscale x 16 x i1> [[PG:%.*]], <vscale x 2 x i64> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv16i8(<vscale x 16 x i1> [[PG]], <vscale x 16 x i8> splat (i8 127), <vscale x 2 x i64> [[A]])
+; CHECK-NEXT: ret <vscale x 16 x i1> [[R]]
+;
+ %r = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv16i8(<vscale x 16 x i1> %pg, <vscale x 16 x i8> splat (i8 127), <vscale x 2 x i64> %a)
+ ret <vscale x 16 x i1> %r
+}
+
+; TODO: We can do better here.
+; Wide compares implicitly promote the smaller typed operand to the larger type,
+; but with the larger operand being constant we know it is too big.
+define <vscale x 16 x i1> @constant_icmp_wide_due_to_range_of_type(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a) #0 {
+; CHECK-LABEL: define <vscale x 16 x i1> @constant_icmp_wide_due_to_range_of_type(
+; CHECK-SAME: <vscale x 16 x i1> [[PG:%.*]], <vscale x 16 x i8> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv16i8(<vscale x 16 x i1> [[PG]], <vscale x 16 x i8> [[A]], <vscale x 2 x i64> splat (i64 127))
+; CHECK-NEXT: ret <vscale x 16 x i1> [[R]]
+;
+ %r = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv16i8(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a, <vscale x 2 x i64> splat (i64 127))
+ ret <vscale x 16 x i1> %r
+}
+
+; Ensure inactive lanes are zero'd after simplification.
+define <vscale x 4 x i1> @non_constant_simplification(<vscale x 4 x i1> %pg, <vscale x 4 x i1> %a) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @non_constant_simplification(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i1> [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[A_EXT:%.*]] = zext <vscale x 4 x i1> [[A]] to <vscale x 4 x i32>
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> [[A_EXT]], <vscale x 4 x i32> zeroinitializer)
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %a.ext = zext <vscale x 4 x i1> %a to <vscale x 4 x i32>
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a.ext, <vscale x 4 x i32> splat (i32 0))
+ ret <vscale x 4 x i1> %r
+}
+
+; The following tests demonstrate the operations for which hooks are in place to
+; enable simplification. Given the simplications themselves are common code, it
+; is assumed they are already well tested elsewhere.
+
+define <vscale x 4 x i1> @constant_fcmpeq(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpeq(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpeq.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpeq.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_fcmpge(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpge(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_fcmpgt(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpgt(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpgt.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float -5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpgt.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float -5.0), <vscale x 4 x float> splat (float 3.0))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_fcmpne(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpne(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_fcmpuo(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpuo(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpuo.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpuo.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpeq(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpeq(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpeq_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpeq_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpge(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpge(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 -303), <vscale x 4 x i32> splat (i32 777))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 -303), <vscale x 4 x i32> splat (i32 777))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpge_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpge_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpgt(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpgt(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 -777))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 -777))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpgt_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpgt_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmphi(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphi(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmphi_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphi_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmphs(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphs(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmphs_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphs_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmple_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmple_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmple.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 -303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmple.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 -303), <vscale x 2 x i64> splat (i64 303))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmplo_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmplo_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplo.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 777), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplo.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 777), <vscale x 2 x i64> splat (i64 303))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpls_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpls_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpls.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 777))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpls.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 777))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmplt_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmplt_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplt.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplt.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpne(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpne(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
+ ret <vscale x 4 x i1> %r
+}
+
+define <vscale x 4 x i1> @constant_icmpne_wide(<vscale x 4 x i1> %pg) #0 {
+; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpne_wide(
+; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+;
+ %r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
+ ret <vscale x 4 x i1> %r
+}
+
+attributes #0 = { "target-features"="+sve" }
>From f1271b40cad20be5b80c26e39cd3ab12e5c7421b Mon Sep 17 00:00:00 2001
From: Paul Walker <paul.walker at arm.com>
Date: Wed, 22 Jul 2026 12:37:36 +0100
Subject: [PATCH 2/3] [LLVM][InstCombine] Add simplification of SVE compare
intrinsics.
Extends SVEIntrinsicInfo to accept LLVM IR compare information, which
is then used to call simplifyCmpInst on the data operands of SVE
compare intrinsic calls.
---
.../AArch64/AArch64TargetTransformInfo.cpp | 183 +++++++++++++++---
.../AArch64/sve-intrinsic-opts-cmpne.ll | 3 +-
.../AArch64/sve-intrinsic-simplify-cmp.ll | 78 +++-----
3 files changed, 187 insertions(+), 77 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 03aa42b31e6aa..d1f8365839777 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -1431,6 +1431,28 @@ struct SVEIntrinsicInfo {
return *this;
}
+ bool hasCmpPredicate() const {
+ return CmpPredicate != CmpInst::BAD_ICMP_PREDICATE;
+ }
+
+ CmpInst::Predicate getCmpPredicate() const {
+ assert(hasCmpPredicate() && "Property not set!");
+ return CmpPredicate;
+ }
+
+ SVEIntrinsicInfo &setCmpPredicate(CmpInst::Predicate Pred) {
+ assert(!hasCmpPredicate() && "Cannot set property twice!");
+ CmpPredicate = Pred;
+
+ if (CmpInst::isFPPredicate(Pred))
+ return setMatchingIROpcode(Instruction::FCmp);
+
+ if (CmpInst::isIntPredicate(Pred))
+ return setMatchingIROpcode(Instruction::ICmp);
+
+ llvm_unreachable("Unsupported compare predicate!");
+ }
+
//
// Properties relating to the result of inactive lanes.
//
@@ -1507,6 +1529,7 @@ struct SVEIntrinsicInfo {
Intrinsic::ID UndefIntrinsic = Intrinsic::not_intrinsic;
unsigned IROpcode = 0;
+ CmpInst::Predicate CmpPredicate = CmpInst::BAD_ICMP_PREDICATE;
enum PredicationStyle {
Uninitialized,
@@ -1755,29 +1778,8 @@ static SVEIntrinsicInfo constructSVEIntrinsicInfo(IntrinsicInst &II) {
case Intrinsic::aarch64_sve_uaddv:
case Intrinsic::aarch64_sve_umaxv:
case Intrinsic::aarch64_sve_umaxqv:
- case Intrinsic::aarch64_sve_cmpeq:
- case Intrinsic::aarch64_sve_cmpeq_wide:
- case Intrinsic::aarch64_sve_cmpge:
- case Intrinsic::aarch64_sve_cmpge_wide:
- case Intrinsic::aarch64_sve_cmpgt:
- case Intrinsic::aarch64_sve_cmpgt_wide:
- case Intrinsic::aarch64_sve_cmphi:
- case Intrinsic::aarch64_sve_cmphi_wide:
- case Intrinsic::aarch64_sve_cmphs:
- case Intrinsic::aarch64_sve_cmphs_wide:
- case Intrinsic::aarch64_sve_cmple_wide:
- case Intrinsic::aarch64_sve_cmplo_wide:
- case Intrinsic::aarch64_sve_cmpls_wide:
- case Intrinsic::aarch64_sve_cmplt_wide:
- case Intrinsic::aarch64_sve_cmpne:
- case Intrinsic::aarch64_sve_cmpne_wide:
case Intrinsic::aarch64_sve_facge:
case Intrinsic::aarch64_sve_facgt:
- case Intrinsic::aarch64_sve_fcmpeq:
- case Intrinsic::aarch64_sve_fcmpge:
- case Intrinsic::aarch64_sve_fcmpgt:
- case Intrinsic::aarch64_sve_fcmpne:
- case Intrinsic::aarch64_sve_fcmpuo:
case Intrinsic::aarch64_sve_ld1:
case Intrinsic::aarch64_sve_ld1_gather:
case Intrinsic::aarch64_sve_ld1_gather_index:
@@ -1825,6 +1827,58 @@ static SVEIntrinsicInfo constructSVEIntrinsicInfo(IntrinsicInst &II) {
return SVEIntrinsicInfo::defaultZeroingOp().setMatchingIROpcode(
Instruction::Xor);
+ case Intrinsic::aarch64_sve_cmpeq:
+ case Intrinsic::aarch64_sve_cmpeq_wide:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::ICMP_EQ);
+ case Intrinsic::aarch64_sve_cmpge:
+ case Intrinsic::aarch64_sve_cmpge_wide:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::ICMP_SGE);
+ case Intrinsic::aarch64_sve_cmpgt:
+ case Intrinsic::aarch64_sve_cmpgt_wide:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::ICMP_SGT);
+ case Intrinsic::aarch64_sve_cmphi:
+ case Intrinsic::aarch64_sve_cmphi_wide:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::ICMP_UGT);
+ case Intrinsic::aarch64_sve_cmphs:
+ case Intrinsic::aarch64_sve_cmphs_wide:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::ICMP_UGE);
+ case Intrinsic::aarch64_sve_cmple_wide:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::ICMP_SLE);
+ case Intrinsic::aarch64_sve_cmplo_wide:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::ICMP_ULT);
+ case Intrinsic::aarch64_sve_cmpls_wide:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::ICMP_ULE);
+ case Intrinsic::aarch64_sve_cmplt_wide:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::ICMP_SLT);
+ case Intrinsic::aarch64_sve_cmpne:
+ case Intrinsic::aarch64_sve_cmpne_wide:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::ICMP_NE);
+ case Intrinsic::aarch64_sve_fcmpeq:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::FCMP_OEQ);
+ case Intrinsic::aarch64_sve_fcmpge:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::FCMP_OGE);
+ case Intrinsic::aarch64_sve_fcmpgt:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::FCMP_OGT);
+ case Intrinsic::aarch64_sve_fcmpne:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::FCMP_UNE);
+ case Intrinsic::aarch64_sve_fcmpuo:
+ return SVEIntrinsicInfo::defaultZeroingOp().setCmpPredicate(
+ CmpInst::FCMP_UNO);
+
case Intrinsic::aarch64_sve_prf:
case Intrinsic::aarch64_sve_prfb_gather_index:
case Intrinsic::aarch64_sve_prfb_gather_scalar_offset:
@@ -1964,6 +2018,79 @@ simplifySVEIntrinsicBinOp(InstCombiner &IC, IntrinsicInst &II,
return IC.replaceInstUsesWith(II, SimpleII);
}
+static std::optional<Instruction *>
+simplifySVEIntrinsicCompare(InstCombiner &IC, IntrinsicInst &II,
+ const SVEIntrinsicInfo &IInfo) {
+ const unsigned Opc = IInfo.getMatchingIROpode();
+ assert((Opc == Instruction::ICmp || Opc == Instruction::FCmp) &&
+ "Expected a compare operation!");
+
+ Value *Pg = II.getOperand(0);
+ Value *LHS = II.getOperand(1);
+ Value *RHS = II.getOperand(2);
+ CmpInst::Predicate CmpPred = IInfo.getCmpPredicate();
+ const DataLayout &DL = II.getDataLayout();
+
+ // Canonicalise integer constants to the RHS.
+ if (Opc == Instruction::ICmp && ICmpInst::isCommutative(CmpPred) &&
+ isa<Constant>(LHS) && !isa<Constant>(RHS) &&
+ LHS->getType() == RHS->getType()) {
+ IC.replaceOperand(II, 1, RHS);
+ IC.replaceOperand(II, 2, LHS);
+ return &II;
+ }
+
+ // Canonicalise floating-point constants to the RHS.
+ if (Opc == Instruction::FCmp && FCmpInst::isCommutative(CmpPred) &&
+ isa<Constant>(LHS) && !isa<Constant>(RHS)) {
+ assert(LHS->getType() == RHS->getType() && "Unexpected wide compare!");
+ IC.replaceOperand(II, 1, RHS);
+ IC.replaceOperand(II, 2, LHS);
+ return &II;
+ }
+
+ // Only active lanes matter when simplifying the operation.
+ LHS = stripInactiveLanes(LHS, Pg);
+ RHS = stripInactiveLanes(RHS, Pg);
+
+ if (LHS->getType() != RHS->getType()) {
+ // We can do more for wide compares, but not using simplifyCmpInst.
+ const APInt *LHSVal, *RHSVal;
+ if (!match(LHS, m_APInt(LHSVal)) || !match(RHS, m_APInt(RHSVal)))
+ return std::nullopt;
+
+ // Consider cmpge.wide(..., <vscale x 4 x i32> LHS, <vscale x 2 x i64> RHS),
+ // we must reconstruct the constants because LHS has the wrong element type,
+ // and RHS the wrong element count.
+ Type *WideVT = VectorType::get(RHS->getType()->getScalarType(),
+ cast<VectorType>(LHS->getType()));
+ assert(Opc == Instruction::ICmp && "Only wide integer compares exist!");
+ if (ICmpInst::isSigned(CmpPred)) {
+ LHS = ConstantInt::get(WideVT, LHSVal->getSExtValue());
+ RHS = ConstantInt::get(WideVT, RHSVal->getSExtValue());
+ } else {
+ LHS = ConstantInt::get(WideVT, LHSVal->getZExtValue());
+ RHS = ConstantInt::get(WideVT, RHSVal->getZExtValue());
+ }
+ }
+
+ // TODO: Allow fast-math flags for calls to compare intrinsics.
+ Value *SimpleII = simplifyCmpInst(CmpPred, LHS, RHS, DL);
+
+ // No simplification happened.
+ if (!SimpleII)
+ return std::nullopt;
+
+ assert(IInfo.resultIsZeroInitialized() && "Expected a zeroing operation!");
+
+ if (match(SimpleII, m_ZeroInt()))
+ return IC.replaceInstUsesWith(II, SimpleII);
+
+ // Inactive lanes must be zero'd.
+ SimpleII = IC.Builder.CreateLogicalAnd(Pg, SimpleII);
+ return IC.replaceInstUsesWith(II, SimpleII);
+}
+
// Use SVE intrinsic info to eliminate redundant operands and/or canonicalise
// to operations with less strict inactive lane requirements.
static std::optional<Instruction *>
@@ -2004,11 +2131,21 @@ simplifySVEIntrinsic(InstCombiner &IC, IntrinsicInst &II,
}
}
+ if (!IInfo.hasMatchingIROpode())
+ return std::nullopt;
+
+ //
// Operation specific simplifications.
- if (IInfo.hasMatchingIROpode() &&
- Instruction::isBinaryOp(IInfo.getMatchingIROpode()))
+ //
+
+ unsigned Opc = IInfo.getMatchingIROpode();
+
+ if (Instruction::isBinaryOp(Opc))
return simplifySVEIntrinsicBinOp(IC, II, IInfo);
+ if (Opc == Instruction::FCmp || Opc == Instruction::ICmp)
+ return simplifySVEIntrinsicCompare(IC, II, IInfo);
+
return std::nullopt;
}
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
index 11e6e0089293a..d9a5434c35269 100644
--- a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-opts-cmpne.ll
@@ -275,8 +275,7 @@ define <vscale x 16 x i8> @zext_cmpne_i8(<vscale x 16 x i8> %vec) #0 {
define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(<vscale x 16 x i8> %vec) #0 {
; CHECK-LABEL: define <vscale x 16 x i8> @zext_cmpne_zero_lhs_i8(
; CHECK-SAME: <vscale x 16 x i8> [[VEC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[TMP1:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> zeroinitializer, <vscale x 16 x i8> [[VEC]])
-; CHECK-NEXT: [[ZEXT:%.*]] = zext <vscale x 16 x i1> [[TMP1]] to <vscale x 16 x i8>
+; CHECK-NEXT: [[ZEXT:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.umin.u.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> [[VEC]], <vscale x 16 x i8> splat (i8 1))
; CHECK-NEXT: ret <vscale x 16 x i8> [[ZEXT]]
;
%cmp = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpne.nxv16i8(<vscale x 16 x i1> splat (i1 true), <vscale x 16 x i8> zeroinitializer, <vscale x 16 x i8> %vec)
diff --git a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll
index 9efa9dee1b6cb..b355c169ae9f6 100644
--- a/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll
+++ b/llvm/test/Transforms/InstCombine/AArch64/sve-intrinsic-simplify-cmp.ll
@@ -9,7 +9,7 @@ target triple = "aarch64-unknown-linux-gnu"
define <vscale x 4 x i1> @cannonicalise_constant_to_rhs(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @cannonicalise_constant_to_rhs(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i32> [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> [[A]])
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> [[A]], <vscale x 4 x i32> splat (i32 303))
; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> %a)
@@ -39,7 +39,7 @@ define <vscale x 4 x i1> @cannot_cannonicalise_constant_to_rhs_operand_type_mism
define <vscale x 4 x i1> @cannonicalise_fp_constant_to_rhs(<vscale x 4 x i1> %pg, <vscale x 4 x float> %a) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @cannonicalise_fp_constant_to_rhs(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x float> [[A:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> [[A]])
+; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> [[A]], <vscale x 4 x float> splat (float 5.000000e+00))
; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> %a)
@@ -60,10 +60,7 @@ define <vscale x 4 x i1> @cannot_cannonicalise_fp_constant_to_rhs_not_commutativ
define <vscale x 4 x i1> @constant_icmp_after_striping_inactive_lanes(<vscale x 4 x i1> %pg, <vscale x 4 x i32> %a, <vscale x 4 x i32> %b) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmp_after_striping_inactive_lanes(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i32> [[A:%.*]], <vscale x 4 x i32> [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[A_DUP:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> [[A]], <vscale x 4 x i1> [[PG]], i32 3)
-; CHECK-NEXT: [[B_DUP:%.*]] = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> [[B]], <vscale x 4 x i1> [[PG]], i32 2)
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> [[A_DUP]], <vscale x 4 x i32> [[B_DUP]])
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> [[PG]]
;
%a.dup = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> %a, <vscale x 4 x i1> %pg, i32 3)
%b.dup = call <vscale x 4 x i32> @llvm.aarch64.sve.dup.nxv4i32(<vscale x 4 x i32> %b, <vscale x 4 x i1> %pg, i32 2)
@@ -74,8 +71,7 @@ define <vscale x 4 x i1> @constant_icmp_after_striping_inactive_lanes(<vscale x
define <vscale x 16 x i1> @constant_icmp_due_to_range_of_type(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a) #0 {
; CHECK-LABEL: define <vscale x 16 x i1> @constant_icmp_due_to_range_of_type(
; CHECK-SAME: <vscale x 16 x i1> [[PG:%.*]], <vscale x 16 x i8> [[A:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.nxv16i8(<vscale x 16 x i1> [[PG]], <vscale x 16 x i8> [[A]], <vscale x 16 x i8> splat (i8 127))
-; CHECK-NEXT: ret <vscale x 16 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 16 x i1> zeroinitializer
;
%r = call <vscale x 16 x i1> @llvm.aarch64.sve.cmpgt.nxv16i8(<vscale x 16 x i1> %pg, <vscale x 16 x i8> %a, <vscale x 16 x i8> splat (i8 127))
ret <vscale x 16 x i1> %r
@@ -109,8 +105,7 @@ define <vscale x 16 x i1> @constant_icmp_wide_due_to_range_of_type(<vscale x 16
define <vscale x 4 x i1> @non_constant_simplification(<vscale x 4 x i1> %pg, <vscale x 4 x i1> %a) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @non_constant_simplification(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]], <vscale x 4 x i1> [[A:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[A_EXT:%.*]] = zext <vscale x 4 x i1> [[A]] to <vscale x 4 x i32>
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> [[A_EXT]], <vscale x 4 x i32> zeroinitializer)
+; CHECK-NEXT: [[R:%.*]] = select <vscale x 4 x i1> [[PG]], <vscale x 4 x i1> [[A]], <vscale x 4 x i1> zeroinitializer
; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
;
%a.ext = zext <vscale x 4 x i1> %a to <vscale x 4 x i32>
@@ -125,8 +120,7 @@ define <vscale x 4 x i1> @non_constant_simplification(<vscale x 4 x i1> %pg, <vs
define <vscale x 4 x i1> @constant_fcmpeq(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpeq(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpeq.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> zeroinitializer
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpeq.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
ret <vscale x 4 x i1> %r
@@ -135,8 +129,7 @@ define <vscale x 4 x i1> @constant_fcmpeq(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_fcmpge(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpge(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> [[PG]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpge.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
ret <vscale x 4 x i1> %r
@@ -145,8 +138,7 @@ define <vscale x 4 x i1> @constant_fcmpge(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_fcmpgt(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpgt(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpgt.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float -5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> zeroinitializer
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpgt.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float -5.0), <vscale x 4 x float> splat (float 3.0))
ret <vscale x 4 x i1> %r
@@ -155,8 +147,7 @@ define <vscale x 4 x i1> @constant_fcmpgt(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_fcmpne(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpne(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> [[PG]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpne.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
ret <vscale x 4 x i1> %r
@@ -165,8 +156,7 @@ define <vscale x 4 x i1> @constant_fcmpne(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_fcmpuo(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_fcmpuo(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpuo.nxv4f32(<vscale x 4 x i1> [[PG]], <vscale x 4 x float> splat (float 5.000000e+00), <vscale x 4 x float> splat (float 3.000000e+00))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> zeroinitializer
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.fcmpuo.nxv4f32(<vscale x 4 x i1> %pg, <vscale x 4 x float> splat (float 5.0), <vscale x 4 x float> splat (float 3.0))
ret <vscale x 4 x i1> %r
@@ -175,8 +165,7 @@ define <vscale x 4 x i1> @constant_fcmpuo(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmpeq(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpeq(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> zeroinitializer
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
ret <vscale x 4 x i1> %r
@@ -185,8 +174,7 @@ define <vscale x 4 x i1> @constant_icmpeq(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmpeq_wide(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpeq_wide(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> [[PG]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpeq.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
ret <vscale x 4 x i1> %r
@@ -195,8 +183,7 @@ define <vscale x 4 x i1> @constant_icmpeq_wide(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmpge(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpge(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 -303), <vscale x 4 x i32> splat (i32 777))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> zeroinitializer
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 -303), <vscale x 4 x i32> splat (i32 777))
ret <vscale x 4 x i1> %r
@@ -205,8 +192,7 @@ define <vscale x 4 x i1> @constant_icmpge(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmpge_wide(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpge_wide(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> [[PG]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpge.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
ret <vscale x 4 x i1> %r
@@ -215,8 +201,7 @@ define <vscale x 4 x i1> @constant_icmpge_wide(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmpgt(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpgt(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 -777))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> [[PG]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 -777))
ret <vscale x 4 x i1> %r
@@ -225,8 +210,7 @@ define <vscale x 4 x i1> @constant_icmpgt(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmpgt_wide(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpgt_wide(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> [[PG]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpgt.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
ret <vscale x 4 x i1> %r
@@ -235,8 +219,7 @@ define <vscale x 4 x i1> @constant_icmpgt_wide(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmphi(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphi(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> zeroinitializer
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
ret <vscale x 4 x i1> %r
@@ -245,8 +228,7 @@ define <vscale x 4 x i1> @constant_icmphi(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmphi_wide(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphi_wide(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> zeroinitializer
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphi.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
ret <vscale x 4 x i1> %r
@@ -255,8 +237,7 @@ define <vscale x 4 x i1> @constant_icmphi_wide(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmphs(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphs(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> zeroinitializer
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
ret <vscale x 4 x i1> %r
@@ -265,8 +246,7 @@ define <vscale x 4 x i1> @constant_icmphs(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmphs_wide(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmphs_wide(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> [[PG]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmphs.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
ret <vscale x 4 x i1> %r
@@ -275,8 +255,7 @@ define <vscale x 4 x i1> @constant_icmphs_wide(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmple_wide(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmple_wide(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmple.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 -303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> [[PG]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmple.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 -303), <vscale x 2 x i64> splat (i64 303))
ret <vscale x 4 x i1> %r
@@ -285,8 +264,7 @@ define <vscale x 4 x i1> @constant_icmple_wide(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmplo_wide(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmplo_wide(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplo.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 777), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> zeroinitializer
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplo.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 777), <vscale x 2 x i64> splat (i64 303))
ret <vscale x 4 x i1> %r
@@ -295,8 +273,7 @@ define <vscale x 4 x i1> @constant_icmplo_wide(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmpls_wide(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpls_wide(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpls.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 777))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> [[PG]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpls.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 777))
ret <vscale x 4 x i1> %r
@@ -305,8 +282,7 @@ define <vscale x 4 x i1> @constant_icmpls_wide(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmplt_wide(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmplt_wide(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplt.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> zeroinitializer
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmplt.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 -303))
ret <vscale x 4 x i1> %r
@@ -315,8 +291,7 @@ define <vscale x 4 x i1> @constant_icmplt_wide(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmpne(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpne(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> [[PG]]
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 4 x i32> splat (i32 777))
ret <vscale x 4 x i1> %r
@@ -325,8 +300,7 @@ define <vscale x 4 x i1> @constant_icmpne(<vscale x 4 x i1> %pg) #0 {
define <vscale x 4 x i1> @constant_icmpne_wide(<vscale x 4 x i1> %pg) #0 {
; CHECK-LABEL: define <vscale x 4 x i1> @constant_icmpne_wide(
; CHECK-SAME: <vscale x 4 x i1> [[PG:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[R:%.*]] = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> [[PG]], <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
-; CHECK-NEXT: ret <vscale x 4 x i1> [[R]]
+; CHECK-NEXT: ret <vscale x 4 x i1> zeroinitializer
;
%r = call <vscale x 4 x i1> @llvm.aarch64.sve.cmpne.wide.nxv4i32(<vscale x 4 x i1> %pg, <vscale x 4 x i32> splat (i32 303), <vscale x 2 x i64> splat (i64 303))
ret <vscale x 4 x i1> %r
>From 966502960bb37eab7aebb277684fabb837d5b566 Mon Sep 17 00:00:00 2001
From: Paul Walker <paul.walker at arm.com>
Date: Tue, 28 Jul 2026 14:27:42 +0000
Subject: [PATCH 3/3] Make wide icmp handling more explicit.
---
.../AArch64/AArch64TargetTransformInfo.cpp | 25 +++++++------------
1 file changed, 9 insertions(+), 16 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index d1f8365839777..07d2ba315356f 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -2029,21 +2029,14 @@ simplifySVEIntrinsicCompare(InstCombiner &IC, IntrinsicInst &II,
Value *LHS = II.getOperand(1);
Value *RHS = II.getOperand(2);
CmpInst::Predicate CmpPred = IInfo.getCmpPredicate();
- const DataLayout &DL = II.getDataLayout();
-
- // Canonicalise integer constants to the RHS.
- if (Opc == Instruction::ICmp && ICmpInst::isCommutative(CmpPred) &&
- isa<Constant>(LHS) && !isa<Constant>(RHS) &&
- LHS->getType() == RHS->getType()) {
- IC.replaceOperand(II, 1, RHS);
- IC.replaceOperand(II, 2, LHS);
- return &II;
- }
+ bool IsWideICmp =
+ Opc == Instruction::ICmp && LHS->getType() != RHS->getType();
+ assert((IsWideICmp || LHS->getType() == RHS->getType()) &&
+ "Unexpected wide compare!");
- // Canonicalise floating-point constants to the RHS.
- if (Opc == Instruction::FCmp && FCmpInst::isCommutative(CmpPred) &&
- isa<Constant>(LHS) && !isa<Constant>(RHS)) {
- assert(LHS->getType() == RHS->getType() && "Unexpected wide compare!");
+ // Canonicalise constants to the RHS.
+ if ((ICmpInst::isCommutative(CmpPred) || FCmpInst::isCommutative(CmpPred)) &&
+ isa<Constant>(LHS) && !isa<Constant>(RHS) && !IsWideICmp) {
IC.replaceOperand(II, 1, RHS);
IC.replaceOperand(II, 2, LHS);
return &II;
@@ -2053,7 +2046,7 @@ simplifySVEIntrinsicCompare(InstCombiner &IC, IntrinsicInst &II,
LHS = stripInactiveLanes(LHS, Pg);
RHS = stripInactiveLanes(RHS, Pg);
- if (LHS->getType() != RHS->getType()) {
+ if (IsWideICmp) {
// We can do more for wide compares, but not using simplifyCmpInst.
const APInt *LHSVal, *RHSVal;
if (!match(LHS, m_APInt(LHSVal)) || !match(RHS, m_APInt(RHSVal)))
@@ -2064,7 +2057,6 @@ simplifySVEIntrinsicCompare(InstCombiner &IC, IntrinsicInst &II,
// and RHS the wrong element count.
Type *WideVT = VectorType::get(RHS->getType()->getScalarType(),
cast<VectorType>(LHS->getType()));
- assert(Opc == Instruction::ICmp && "Only wide integer compares exist!");
if (ICmpInst::isSigned(CmpPred)) {
LHS = ConstantInt::get(WideVT, LHSVal->getSExtValue());
RHS = ConstantInt::get(WideVT, RHSVal->getSExtValue());
@@ -2075,6 +2067,7 @@ simplifySVEIntrinsicCompare(InstCombiner &IC, IntrinsicInst &II,
}
// TODO: Allow fast-math flags for calls to compare intrinsics.
+ const DataLayout &DL = II.getDataLayout();
Value *SimpleII = simplifyCmpInst(CmpPred, LHS, RHS, DL);
// No simplification happened.
More information about the llvm-commits
mailing list