[llvm] [AArch64] Use SVE rev for full reverse shuffles with SVE128 (PR #224589)
David Green via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 06:12:33 PDT 2026
https://github.com/davemgreen updated https://github.com/llvm/llvm-project/pull/224589
>From 5202c82643878ca613090f880b2eadc3d1dd7796 Mon Sep 17 00:00:00 2001
From: David Green <david.green at arm.com>
Date: Fri, 18 Sep 2026 10:38:15 +0100
Subject: [PATCH 1/3] Test!
---
.../sve-fixed-length-shuffle-reverse.ll | 158 ++++++++++++++++++
1 file changed, 158 insertions(+)
create mode 100644 llvm/test/CodeGen/AArch64/sve-fixed-length-shuffle-reverse.ll
diff --git a/llvm/test/CodeGen/AArch64/sve-fixed-length-shuffle-reverse.ll b/llvm/test/CodeGen/AArch64/sve-fixed-length-shuffle-reverse.ll
new file mode 100644
index 0000000000000..6dffe8d267353
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-fixed-length-shuffle-reverse.ll
@@ -0,0 +1,158 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=aarch64--linux-gnu -mattr=+sve | FileCheck %s
+
+define <2 x i64> @testrev_v2i64_vscale1(<2 x i64> %x) vscale_range(1,1) {
+; CHECK-LABEL: testrev_v2i64_vscale1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <2 x i64> %x, <2 x i64> poison, <2 x i32> <i32 1, i32 0>
+ ret <2 x i64> %r
+}
+
+define <2 x i64> @testrev_v2i64(<2 x i64> %x) {
+; CHECK-LABEL: testrev_v2i64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <2 x i64> %x, <2 x i64> poison, <2 x i32> <i32 1, i32 0>
+ ret <2 x i64> %r
+}
+
+define <2 x double> @testrev_v2f64_vscale1(<2 x double> %x) vscale_range(1,1) {
+; CHECK-LABEL: testrev_v2f64_vscale1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <2 x double> %x, <2 x double> poison, <2 x i32> <i32 1, i32 0>
+ ret <2 x double> %r
+}
+
+define <2 x double> @testrev_v2f64(<2 x double> %x) {
+; CHECK-LABEL: testrev_v2f64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <2 x double> %x, <2 x double> poison, <2 x i32> <i32 1, i32 0>
+ ret <2 x double> %r
+}
+
+define <4 x i32> @testrev_v4i32_vscale1(<4 x i32> %x) vscale_range(1,1) {
+; CHECK-LABEL: testrev_v4i32_vscale1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.4s, v0.4s
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <4 x i32> %x, <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+ ret <4 x i32> %r
+}
+
+define <4 x i32> @testrev_v4i32(<4 x i32> %x) {
+; CHECK-LABEL: testrev_v4i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.4s, v0.4s
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <4 x i32> %x, <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+ ret <4 x i32> %r
+}
+
+define <4 x float> @testrev_v4f32_vscale1(<4 x float> %x) vscale_range(1,1) {
+; CHECK-LABEL: testrev_v4f32_vscale1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.4s, v0.4s
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <4 x float> %x, <4 x float> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+ ret <4 x float> %r
+}
+
+define <4 x float> @testrev_v4f32(<4 x float> %x) {
+; CHECK-LABEL: testrev_v4f32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.4s, v0.4s
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <4 x float> %x, <4 x float> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+ ret <4 x float> %r
+}
+
+define <8 x i16> @testrev_v8i16_vscale1(<8 x i16> %x) vscale_range(1,1) {
+; CHECK-LABEL: testrev_v8i16_vscale1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.8h, v0.8h
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <8 x i16> %x, <8 x i16> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+ ret <8 x i16> %r
+}
+
+define <8 x i16> @testrev_v8i16(<8 x i16> %x) {
+; CHECK-LABEL: testrev_v8i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.8h, v0.8h
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <8 x i16> %x, <8 x i16> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+ ret <8 x i16> %r
+}
+
+define <8 x half> @testrev_v8f16_vscale1(<8 x half> %x) vscale_range(1,1) {
+; CHECK-LABEL: testrev_v8f16_vscale1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.8h, v0.8h
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <8 x half> %x, <8 x half> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+ ret <8 x half> %r
+}
+
+define <8 x half> @testrev_v8f16(<8 x half> %x) {
+; CHECK-LABEL: testrev_v8f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.8h, v0.8h
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <8 x half> %x, <8 x half> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+ ret <8 x half> %r
+}
+
+define <8 x bfloat> @testrev_v8bf16_vscale1(<8 x bfloat> %x) vscale_range(1,1) {
+; CHECK-LABEL: testrev_v8bf16_vscale1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.8h, v0.8h
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <8 x bfloat> %x, <8 x bfloat> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+ ret <8 x bfloat> %r
+}
+
+define <8 x bfloat> @testrev_v8bf16(<8 x bfloat> %x) {
+; CHECK-LABEL: testrev_v8bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.8h, v0.8h
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <8 x bfloat> %x, <8 x bfloat> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+ ret <8 x bfloat> %r
+}
+
+define <16 x i8> @testrev_v16i8_vscale1(<16 x i8> %x) vscale_range(1,1) {
+; CHECK-LABEL: testrev_v16i8_vscale1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.16b, v0.16b
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <16 x i8> %x, <16 x i8> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+ ret <16 x i8> %r
+}
+
+define <16 x i8> @testrev_v16i8(<16 x i8> %x) {
+; CHECK-LABEL: testrev_v16i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: rev64 v0.16b, v0.16b
+; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: ret
+ %r = shufflevector <16 x i8> %x, <16 x i8> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+ ret <16 x i8> %r
+}
>From bbb478c6a1a600807e775eb914033c858e9b91a4 Mon Sep 17 00:00:00 2001
From: David Green <david.green at arm.com>
Date: Fri, 18 Sep 2026 10:56:52 +0100
Subject: [PATCH 2/3] [AArch64] Use SVE rev for full reverse shuffles with
SVE128
When the vscale_range is always 1, we can make use of the SVE rev instruction
to perform a full 128bit vector reverse.
---
.../Target/AArch64/AArch64ISelLowering.cpp | 20 ++++++++++---
.../sve-fixed-length-shuffle-reverse.ll | 30 +++++++++++--------
2 files changed, 34 insertions(+), 16 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index c4942e12e17ac..2c47b32df440a 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -16195,11 +16195,23 @@ SDValue AArch64TargetLowering::LowerVECTOR_SHUFFLE(SDValue Op,
DAG.getNode(AArch64ISD::NVCAST, DL, BSVT, V1)));
}
- if (((NumElts == 8 && EltSize == 16) || (NumElts == 16 && EltSize == 8)) &&
+ if (((NumElts == 8 && EltSize == 16) || (NumElts == 16 && EltSize == 8) ||
+ (NumElts == 4 && EltSize == 32)) &&
ShuffleVectorInst::isReverseMask(ShuffleMask, ShuffleMask.size())) {
- SDValue Rev = DAG.getNode(AArch64ISD::REV64, DL, VT, V1);
- return DAG.getNode(AArch64ISD::EXT, DL, VT, Rev, Rev,
- DAG.getConstant(8, DL, MVT::i32));
+ // For sve128 we can use a REV full vector reverse.
+ if (Subtarget->isSVEorStreamingSVEAvailable() &&
+ Subtarget->getMaxSVEVectorSizeInBits() == 128) {
+ EVT ContainerVT = getContainerForFixedLengthVector(DAG, VT);
+ V1 = convertToScalableVector(DAG, ContainerVT, V1);
+ SDValue Rev = DAG.getNode(ISD::VECTOR_REVERSE, DL, ContainerVT, V1);
+ return convertFromScalableVector(DAG, VT, Rev);
+ }
+
+ if (EltSize != 32) {
+ SDValue Rev = DAG.getNode(AArch64ISD::REV64, DL, VT, V1);
+ return DAG.getNode(AArch64ISD::EXT, DL, VT, Rev, Rev,
+ DAG.getConstant(8, DL, MVT::i32));
+ }
}
// Check for slide-with-zeros pattern before EXT (slide is also valid EXT)
diff --git a/llvm/test/CodeGen/AArch64/sve-fixed-length-shuffle-reverse.ll b/llvm/test/CodeGen/AArch64/sve-fixed-length-shuffle-reverse.ll
index 6dffe8d267353..0bd3afe63dbc0 100644
--- a/llvm/test/CodeGen/AArch64/sve-fixed-length-shuffle-reverse.ll
+++ b/llvm/test/CodeGen/AArch64/sve-fixed-length-shuffle-reverse.ll
@@ -40,8 +40,9 @@ define <2 x double> @testrev_v2f64(<2 x double> %x) {
define <4 x i32> @testrev_v4i32_vscale1(<4 x i32> %x) vscale_range(1,1) {
; CHECK-LABEL: testrev_v4i32_vscale1:
; CHECK: // %bb.0:
-; CHECK-NEXT: rev64 v0.4s, v0.4s
-; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: rev z0.s, z0.s
+; CHECK-NEXT: // kill: def $q0 killed $q0 killed $z0
; CHECK-NEXT: ret
%r = shufflevector <4 x i32> %x, <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
ret <4 x i32> %r
@@ -60,8 +61,9 @@ define <4 x i32> @testrev_v4i32(<4 x i32> %x) {
define <4 x float> @testrev_v4f32_vscale1(<4 x float> %x) vscale_range(1,1) {
; CHECK-LABEL: testrev_v4f32_vscale1:
; CHECK: // %bb.0:
-; CHECK-NEXT: rev64 v0.4s, v0.4s
-; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: rev z0.s, z0.s
+; CHECK-NEXT: // kill: def $q0 killed $q0 killed $z0
; CHECK-NEXT: ret
%r = shufflevector <4 x float> %x, <4 x float> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
ret <4 x float> %r
@@ -80,8 +82,9 @@ define <4 x float> @testrev_v4f32(<4 x float> %x) {
define <8 x i16> @testrev_v8i16_vscale1(<8 x i16> %x) vscale_range(1,1) {
; CHECK-LABEL: testrev_v8i16_vscale1:
; CHECK: // %bb.0:
-; CHECK-NEXT: rev64 v0.8h, v0.8h
-; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: rev z0.h, z0.h
+; CHECK-NEXT: // kill: def $q0 killed $q0 killed $z0
; CHECK-NEXT: ret
%r = shufflevector <8 x i16> %x, <8 x i16> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
ret <8 x i16> %r
@@ -100,8 +103,9 @@ define <8 x i16> @testrev_v8i16(<8 x i16> %x) {
define <8 x half> @testrev_v8f16_vscale1(<8 x half> %x) vscale_range(1,1) {
; CHECK-LABEL: testrev_v8f16_vscale1:
; CHECK: // %bb.0:
-; CHECK-NEXT: rev64 v0.8h, v0.8h
-; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: rev z0.h, z0.h
+; CHECK-NEXT: // kill: def $q0 killed $q0 killed $z0
; CHECK-NEXT: ret
%r = shufflevector <8 x half> %x, <8 x half> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
ret <8 x half> %r
@@ -120,8 +124,9 @@ define <8 x half> @testrev_v8f16(<8 x half> %x) {
define <8 x bfloat> @testrev_v8bf16_vscale1(<8 x bfloat> %x) vscale_range(1,1) {
; CHECK-LABEL: testrev_v8bf16_vscale1:
; CHECK: // %bb.0:
-; CHECK-NEXT: rev64 v0.8h, v0.8h
-; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: rev z0.h, z0.h
+; CHECK-NEXT: // kill: def $q0 killed $q0 killed $z0
; CHECK-NEXT: ret
%r = shufflevector <8 x bfloat> %x, <8 x bfloat> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
ret <8 x bfloat> %r
@@ -140,8 +145,9 @@ define <8 x bfloat> @testrev_v8bf16(<8 x bfloat> %x) {
define <16 x i8> @testrev_v16i8_vscale1(<16 x i8> %x) vscale_range(1,1) {
; CHECK-LABEL: testrev_v16i8_vscale1:
; CHECK: // %bb.0:
-; CHECK-NEXT: rev64 v0.16b, v0.16b
-; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: rev z0.b, z0.b
+; CHECK-NEXT: // kill: def $q0 killed $q0 killed $z0
; CHECK-NEXT: ret
%r = shufflevector <16 x i8> %x, <16 x i8> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
ret <16 x i8> %r
>From 8b1a7b3b4db8fd667e15575cdb08b3ea4ca833ba Mon Sep 17 00:00:00 2001
From: David Green <david.green at arm.com>
Date: Mon, 21 Sep 2026 14:12:09 +0100
Subject: [PATCH 3/3] Use getSVEVectorSizeInBits, even if the name is less
clear
---
.../Target/AArch64/AArch64ISelLowering.cpp | 2 +-
llvm/test/CodeGen/Thumb2/mve-vctp.ll | 112 +++++++++++++-----
2 files changed, 85 insertions(+), 29 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 2c47b32df440a..885502e50db72 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -16200,7 +16200,7 @@ SDValue AArch64TargetLowering::LowerVECTOR_SHUFFLE(SDValue Op,
ShuffleVectorInst::isReverseMask(ShuffleMask, ShuffleMask.size())) {
// For sve128 we can use a REV full vector reverse.
if (Subtarget->isSVEorStreamingSVEAvailable() &&
- Subtarget->getMaxSVEVectorSizeInBits() == 128) {
+ Subtarget->getSVEVectorSizeInBits() == 128) {
EVT ContainerVT = getContainerForFixedLengthVector(DAG, VT);
V1 = convertToScalableVector(DAG, ContainerVT, V1);
SDValue Rev = DAG.getNode(ISD::VECTOR_REVERSE, DL, ContainerVT, V1);
diff --git a/llvm/test/CodeGen/Thumb2/mve-vctp.ll b/llvm/test/CodeGen/Thumb2/mve-vctp.ll
index 02f7de5a07244..6359d3271dc10 100644
--- a/llvm/test/CodeGen/Thumb2/mve-vctp.ll
+++ b/llvm/test/CodeGen/Thumb2/mve-vctp.ll
@@ -85,18 +85,26 @@ entry:
ret <4 x i32> %s
}
-define arm_aapcs_vfpcc <4 x i32> @vcmp_uge_v4i32(i32 %n, <4 x i32> %a, <4 x i32> %b) {
-; CHECK-LABEL: vcmp_uge_v4i32:
+define arm_aapcs_vfpcc <4 x i32> @vcmp_ugt_v4i32(i32 %n, <4 x i32> %a, <4 x i32> %b) {
+; CHECK-LABEL: vcmp_ugt_v4i32:
; CHECK: @ %bb.0: @ %entry
-; CHECK-NEXT: vctp.32 r0
-; CHECK-NEXT: vpst
-; CHECK-NEXT: vmovt q1, q0
-; CHECK-NEXT: vmov q0, q1
+; CHECK-NEXT: vdup.32 q2, r0
+; CHECK-NEXT: adr r0, .LCPI5_0
+; CHECK-NEXT: vldrw.u32 q3, [r0]
+; CHECK-NEXT: vcmp.u32 hi, q2, q3
+; CHECK-NEXT: vpsel q0, q0, q1
; CHECK-NEXT: bx lr
+; CHECK-NEXT: .p2align 4
+; CHECK-NEXT: @ %bb.1:
+; CHECK-NEXT: .LCPI5_0:
+; CHECK-NEXT: .long 0 @ 0x0
+; CHECK-NEXT: .long 1 @ 0x1
+; CHECK-NEXT: .long 2 @ 0x2
+; CHECK-NEXT: .long 3 @ 0x3
entry:
%i = insertelement <4 x i32> undef, i32 %n, i32 0
%ns = shufflevector <4 x i32> %i, <4 x i32> undef, <4 x i32> zeroinitializer
- %c = icmp uge <4 x i32> %ns, <i32 0, i32 1, i32 2, i32 3>
+ %c = icmp ugt <4 x i32> %ns, <i32 0, i32 1, i32 2, i32 3>
%s = select <4 x i1> %c, <4 x i32> %a, <4 x i32> %b
ret <4 x i32> %s
}
@@ -135,19 +143,30 @@ entry:
ret <8 x i16> %s
}
-define arm_aapcs_vfpcc <8 x i16> @vcmp_uge_v8i16(i16 %n, <8 x i16> %a, <8 x i16> %b) {
-; CHECK-LABEL: vcmp_uge_v8i16:
+define arm_aapcs_vfpcc <8 x i16> @vcmp_ugt_v8i16(i16 %n, <8 x i16> %a, <8 x i16> %b) {
+; CHECK-LABEL: vcmp_ugt_v8i16:
; CHECK: @ %bb.0: @ %entry
-; CHECK-NEXT: uxth r0, r0
-; CHECK-NEXT: vctp.16 r0
-; CHECK-NEXT: vpst
-; CHECK-NEXT: vmovt q1, q0
-; CHECK-NEXT: vmov q0, q1
+; CHECK-NEXT: vdup.16 q2, r0
+; CHECK-NEXT: adr r0, .LCPI8_0
+; CHECK-NEXT: vldrw.u32 q3, [r0]
+; CHECK-NEXT: vcmp.u16 hi, q2, q3
+; CHECK-NEXT: vpsel q0, q0, q1
; CHECK-NEXT: bx lr
+; CHECK-NEXT: .p2align 4
+; CHECK-NEXT: @ %bb.1:
+; CHECK-NEXT: .LCPI8_0:
+; CHECK-NEXT: .short 0 @ 0x0
+; CHECK-NEXT: .short 1 @ 0x1
+; CHECK-NEXT: .short 2 @ 0x2
+; CHECK-NEXT: .short 3 @ 0x3
+; CHECK-NEXT: .short 4 @ 0x4
+; CHECK-NEXT: .short 5 @ 0x5
+; CHECK-NEXT: .short 6 @ 0x6
+; CHECK-NEXT: .short 7 @ 0x7
entry:
%i = insertelement <8 x i16> undef, i16 %n, i32 0
%ns = shufflevector <8 x i16> %i, <8 x i16> undef, <8 x i32> zeroinitializer
- %c = icmp uge <8 x i16> %ns, <i16 0, i16 1, i16 2, i16 3, i16 4, i16 5, i16 6, i16 7>
+ %c = icmp ugt <8 x i16> %ns, <i16 0, i16 1, i16 2, i16 3, i16 4, i16 5, i16 6, i16 7>
%s = select <8 x i1> %c, <8 x i16> %a, <8 x i16> %b
ret <8 x i16> %s
}
@@ -170,6 +189,42 @@ entry:
ret <16 x i8> %s
}
+define arm_aapcs_vfpcc <16 x i8> @vcmp_ugt_v16i8(i8 %n, <16 x i8> %a, <16 x i8> %b) {
+; CHECK-LABEL: vcmp_ugt_v16i8:
+; CHECK: @ %bb.0: @ %entry
+; CHECK-NEXT: vdup.8 q2, r0
+; CHECK-NEXT: adr r0, .LCPI10_0
+; CHECK-NEXT: vldrw.u32 q3, [r0]
+; CHECK-NEXT: vcmp.u8 hi, q2, q3
+; CHECK-NEXT: vpsel q0, q0, q1
+; CHECK-NEXT: bx lr
+; CHECK-NEXT: .p2align 4
+; CHECK-NEXT: @ %bb.1:
+; CHECK-NEXT: .LCPI10_0:
+; CHECK-NEXT: .byte 0 @ 0x0
+; CHECK-NEXT: .byte 1 @ 0x1
+; CHECK-NEXT: .byte 2 @ 0x2
+; CHECK-NEXT: .byte 3 @ 0x3
+; CHECK-NEXT: .byte 4 @ 0x4
+; CHECK-NEXT: .byte 5 @ 0x5
+; CHECK-NEXT: .byte 6 @ 0x6
+; CHECK-NEXT: .byte 7 @ 0x7
+; CHECK-NEXT: .byte 8 @ 0x8
+; CHECK-NEXT: .byte 9 @ 0x9
+; CHECK-NEXT: .byte 10 @ 0xa
+; CHECK-NEXT: .byte 11 @ 0xb
+; CHECK-NEXT: .byte 12 @ 0xc
+; CHECK-NEXT: .byte 13 @ 0xd
+; CHECK-NEXT: .byte 14 @ 0xe
+; CHECK-NEXT: .byte 15 @ 0xf
+entry:
+ %i = insertelement <16 x i8> undef, i8 %n, i32 0
+ %ns = shufflevector <16 x i8> %i, <16 x i8> undef, <16 x i32> zeroinitializer
+ %c = icmp ugt <16 x i8> %ns, <i8 0, i8 1, i8 2, i8 3, i8 4, i8 5, i8 6, i8 7, i8 8, i8 9, i8 10, i8 11, i8 12, i8 13, i8 14, i8 15>
+ %s = select <16 x i1> %c, <16 x i8> %a, <16 x i8> %b
+ ret <16 x i8> %s
+}
+
define arm_aapcs_vfpcc <16 x i8> @vcmp_uge_v16i8(i8 %n, <16 x i8> %a, <16 x i8> %b) {
; CHECK-LABEL: vcmp_uge_v16i8:
; CHECK: @ %bb.0: @ %entry
@@ -204,24 +259,25 @@ entry:
ret <2 x i64> %s
}
-define arm_aapcs_vfpcc <2 x i64> @vcmp_uge_v2i64(i64 %n, <2 x i64> %a, <2 x i64> %b) {
-; CHECK-LABEL: vcmp_uge_v2i64:
+define arm_aapcs_vfpcc <2 x i64> @vcmp_ugt_v2i64(i64 %n, <2 x i64> %a, <2 x i64> %b) {
+; CHECK-LABEL: vcmp_ugt_v2i64:
; CHECK: @ %bb.0: @ %entry
-; CHECK-NEXT: vctp.64 r0
-; CHECK-NEXT: vpst
-; CHECK-NEXT: vmovt q1, q0
-; CHECK-NEXT: vmov q0, q1
+; CHECK-NEXT: rsbs r3, r0, #0
+; CHECK-NEXT: mov.w r2, #0
+; CHECK-NEXT: sbcs.w r3, r2, r1
+; CHECK-NEXT: csetm r3, lo
+; CHECK-NEXT: rsbs.w r0, r0, #1
+; CHECK-NEXT: sbcs.w r0, r2, r1
+; CHECK-NEXT: bfi r2, r3, #0, #8
+; CHECK-NEXT: csetm r0, lo
+; CHECK-NEXT: bfi r2, r0, #8, #8
+; CHECK-NEXT: vmsr p0, r2
+; CHECK-NEXT: vpsel q0, q0, q1
; CHECK-NEXT: bx lr
entry:
%i = insertelement <2 x i64> undef, i64 %n, i32 0
%ns = shufflevector <2 x i64> %i, <2 x i64> undef, <2 x i32> zeroinitializer
- %c = icmp uge <2 x i64> %ns, <i64 0, i64 1>
+ %c = icmp ugt <2 x i64> %ns, <i64 0, i64 1>
%s = select <2 x i1> %c, <2 x i64> %a, <2 x i64> %b
ret <2 x i64> %s
}
-
-
-declare <16 x i1> @llvm.arm.mve.vctp8(i32)
-declare <8 x i1> @llvm.arm.mve.vctp16(i32)
-declare <4 x i1> @llvm.arm.mve.vctp32(i32)
-declare <2 x i1> @llvm.arm.mve.vctp64(i32)
More information about the llvm-commits
mailing list