[llvm] [LLVM][CodeGen][SVE] Add isel patterns for predicate "and a, (op b, c)" sequences. (PR #206751)
Paul Walker via llvm-commits
llvm-commits at lists.llvm.org
Wed Jul 1 03:34:54 PDT 2026
https://github.com/paulwalker-arm updated https://github.com/llvm/llvm-project/pull/206751
>From 917c8cdcb2b24c1273124f1f397a798860a79606 Mon Sep 17 00:00:00 2001
From: Paul Walker <paul.walker at arm.com>
Date: Mon, 29 Jun 2026 17:13:45 +0100
Subject: [PATCH 1/2] [LLVM][CodeGen][SVE] Add isel patterns for predicate "and
a, (op b, c)" sequences.
SVE predicate logical operations are predicated and thus effectively
work as is their operation is unpredicated but then anded with the
general predicate.
---
.../lib/Target/AArch64/AArch64SVEInstrInfo.td | 48 ++++---
.../AArch64/intrinsic-vector-match-sve2.ll | 15 +-
llvm/test/CodeGen/AArch64/sve-pred-log.ll | 129 +++++-------------
.../AArch64/sve-split-int-pred-reduce.ll | 17 ++-
4 files changed, 81 insertions(+), 128 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td b/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td
index ec7cb3361635c..2862b535f5303 100644
--- a/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td
+++ b/llvm/lib/Target/AArch64/AArch64SVEInstrInfo.td
@@ -532,6 +532,28 @@ def AArch64bic : PatFrags<(ops node:$op1, node:$op2),
(and node:$op1, (xor node:$op2, (SVEAllActive))),
(AArch64bic_node node:$op1, node:$op2)]>;
+def AArch64and_z : PatFrags<(ops node:$pg, node:$op1, node:$op2),
+ [(int_aarch64_sve_and_z node:$pg, node:$op1, node:$op2),
+ (and node:$pg, (and node:$op1, node:$op2))]>;
+def AArch64bic_z : PatFrags<(ops node:$pg, node:$op1, node:$op2),
+ [(int_aarch64_sve_bic_z node:$pg, node:$op1, node:$op2),
+ (and node:$pg, (AArch64bic node:$op1, node:$op2))]>;
+def AArch64eor_z : PatFrags<(ops node:$pg, node:$op1, node:$op2),
+ [(int_aarch64_sve_eor_z node:$pg, node:$op1, node:$op2),
+ (and node:$pg, (xor node:$op1, node:$op2))]>;
+def AArch64nand_z : PatFrags<(ops node:$pg, node:$op1, node:$op2),
+ [(int_aarch64_sve_nand_z node:$pg, node:$op1, node:$op2),
+ (and node:$pg, (xor (and node:$op1, node:$op2), immAllOnesV))]>;
+def AArch64nor_z : PatFrags<(ops node:$pg, node:$op1, node:$op2),
+ [(int_aarch64_sve_nor_z node:$pg, node:$op1, node:$op2),
+ (and node:$pg, (xor (or node:$op1, node:$op2), immAllOnesV))]>;
+def AArch64orn_z : PatFrags<(ops node:$pg, node:$op1, node:$op2),
+ [(int_aarch64_sve_orn_z node:$pg, node:$op1, node:$op2),
+ (and node:$pg, (or node:$op1, (xor node:$op2, immAllOnesV)))]>;
+def AArch64orr_z : PatFrags<(ops node:$pg, node:$op1, node:$op2),
+ [(int_aarch64_sve_orr_z node:$pg, node:$op1, node:$op2),
+ (and node:$pg, (or node:$op1, node:$op2))]>;
+
def AArch64subr : PatFrag<(ops node:$op1, node:$op2),
(sub node:$op2, node:$op1)>;
@@ -1164,17 +1186,17 @@ let Predicates = [HasSVE_or_SME] in {
defm PFIRST : sve_int_pfirst<0b00000, "pfirst", int_aarch64_sve_pfirst>;
defm PNEXT : sve_int_pnext<0b00110, "pnext", int_aarch64_sve_pnext>;
- defm AND_PPzPP : sve_int_pred_log_v2<0b0000, "and", int_aarch64_sve_and_z, and>;
- defm BIC_PPzPP : sve_int_pred_log_v2<0b0001, "bic", int_aarch64_sve_bic_z, AArch64bic>;
- defm EOR_PPzPP : sve_int_pred_log<0b0010, "eor", int_aarch64_sve_eor_z, xor>;
+ defm AND_PPzPP : sve_int_pred_log_v2<0b0000, "and", AArch64and_z, and>;
+ defm BIC_PPzPP : sve_int_pred_log_v2<0b0001, "bic", AArch64bic_z, AArch64bic>;
+ defm EOR_PPzPP : sve_int_pred_log<0b0010, "eor", AArch64eor_z, xor>;
defm SEL_PPPP : sve_int_pred_log_v2<0b0011, "sel", vselect, or>;
defm ANDS_PPzPP : sve_int_pred_log<0b0100, "ands", null_frag>;
defm BICS_PPzPP : sve_int_pred_log<0b0101, "bics", null_frag>;
defm EORS_PPzPP : sve_int_pred_log<0b0110, "eors", null_frag>;
- defm ORR_PPzPP : sve_int_pred_log<0b1000, "orr", int_aarch64_sve_orr_z>;
- defm ORN_PPzPP : sve_int_pred_log<0b1001, "orn", int_aarch64_sve_orn_z>;
- defm NOR_PPzPP : sve_int_pred_log<0b1010, "nor", int_aarch64_sve_nor_z>;
- defm NAND_PPzPP : sve_int_pred_log<0b1011, "nand", int_aarch64_sve_nand_z>;
+ defm ORR_PPzPP : sve_int_pred_log<0b1000, "orr", AArch64orr_z>;
+ defm ORN_PPzPP : sve_int_pred_log<0b1001, "orn", AArch64orn_z>;
+ defm NOR_PPzPP : sve_int_pred_log<0b1010, "nor", AArch64nor_z>;
+ defm NAND_PPzPP : sve_int_pred_log<0b1011, "nand", AArch64nand_z>;
defm ORRS_PPzPP : sve_int_pred_log<0b1100, "orrs", null_frag>;
defm ORNS_PPzPP : sve_int_pred_log<0b1101, "orns", null_frag>;
defm NORS_PPzPP : sve_int_pred_log<0b1110, "nors", null_frag>;
@@ -3052,18 +3074,6 @@ let Predicates = [HasSVE_or_SME] in {
foreach VT2 = [ nxv16i8, nxv8i16, nxv4i32, nxv2i64, nxv8f16, nxv4f32, nxv2f64, nxv8bf16 ] in
def : Pat<(VT (AArch64NvCast (VT2 ZPR:$src))), (VT ZPR:$src)>;
- def : Pat<(nxv16i1 (and PPR:$Ps1, PPR:$Ps2)),
- (AND_PPzPP (PTRUE_B 31), PPR:$Ps1, PPR:$Ps2)>;
- def : Pat<(nxv8i1 (and PPR:$Ps1, PPR:$Ps2)),
- (AND_PPzPP (PTRUE_H 31), PPR:$Ps1, PPR:$Ps2)>;
- def : Pat<(nxv4i1 (and PPR:$Ps1, PPR:$Ps2)),
- (AND_PPzPP (PTRUE_S 31), PPR:$Ps1, PPR:$Ps2)>;
- def : Pat<(nxv2i1 (and PPR:$Ps1, PPR:$Ps2)),
- (AND_PPzPP (PTRUE_D 31), PPR:$Ps1, PPR:$Ps2)>;
- // Emulate .Q operation using a PTRUE_D when the other lanes don't matter.
- def : Pat<(nxv1i1 (and PPR:$Ps1, PPR:$Ps2)),
- (AND_PPzPP (PTRUE_D 31), PPR:$Ps1, PPR:$Ps2)>;
-
// Add more complex addressing modes here as required
multiclass pred_load<ValueType Ty, ValueType PredTy, SDPatternOperator Load,
Instruction RegRegInst, Instruction RegImmInst, ComplexPattern AddrCP> {
diff --git a/llvm/test/CodeGen/AArch64/intrinsic-vector-match-sve2.ll b/llvm/test/CodeGen/AArch64/intrinsic-vector-match-sve2.ll
index 0bf9e0de3055b..de88e10141fbf 100644
--- a/llvm/test/CodeGen/AArch64/intrinsic-vector-match-sve2.ll
+++ b/llvm/test/CodeGen/AArch64/intrinsic-vector-match-sve2.ll
@@ -23,8 +23,7 @@ define <vscale x 16 x i1> @match_nxv16i8_v2i8(<vscale x 16 x i8> %op1, <2 x i8>
; CHECK-NEXT: mov z1.b, w8
; CHECK-NEXT: cmpeq p3.b, p1/z, z0.b, z2.b
; CHECK-NEXT: cmpeq p2.b, p1/z, z0.b, z1.b
-; CHECK-NEXT: sel p1.b, p3, p3.b, p2.b
-; CHECK-NEXT: and p0.b, p1/z, p1.b, p0.b
+; CHECK-NEXT: orr p0.b, p0/z, p3.b, p2.b
; CHECK-NEXT: ret
%r = tail call <vscale x 16 x i1> @llvm.experimental.vector.match(<vscale x 16 x i8> %op1, <2 x i8> %op2, <vscale x 16 x i1> %mask)
ret <vscale x 16 x i1> %r
@@ -55,8 +54,7 @@ define <vscale x 16 x i1> @match_nxv16i8_v4i8(<vscale x 16 x i8> %op1, <4 x i8>
; CHECK-NEXT: mov p2.b, p3/m, p3.b
; CHECK-NEXT: sel p2.b, p2, p2.b, p4.b
; CHECK-NEXT: ldr p4, [sp, #7, mul vl] // 2-byte Reload
-; CHECK-NEXT: mov p1.b, p2/m, p2.b
-; CHECK-NEXT: and p0.b, p1/z, p1.b, p0.b
+; CHECK-NEXT: orr p0.b, p0/z, p2.b, p1.b
; CHECK-NEXT: addvl sp, sp, #1
; CHECK-NEXT: ldr x29, [sp], #16 // 8-byte Folded Reload
; CHECK-NEXT: ret
@@ -336,8 +334,7 @@ define <vscale x 16 x i1> @match_nxv16i8_v32i8(<vscale x 16 x i8> %op1, <32 x i8
; CHECK-NEXT: sel p2.b, p2, p2.b, p3.b
; CHECK-NEXT: sel p2.b, p2, p2.b, p4.b
; CHECK-NEXT: ldr p4, [sp, #7, mul vl] // 2-byte Reload
-; CHECK-NEXT: mov p1.b, p2/m, p2.b
-; CHECK-NEXT: and p0.b, p1/z, p1.b, p0.b
+; CHECK-NEXT: orr p0.b, p0/z, p2.b, p1.b
; CHECK-NEXT: addvl sp, sp, #1
; CHECK-NEXT: ldr x29, [sp], #16 // 8-byte Folded Reload
; CHECK-NEXT: ret
@@ -473,8 +470,7 @@ define <vscale x 4 x i1> @match_nxv4xi32_v4i32(<vscale x 4 x i32> %op1, <4 x i32
; CHECK-NEXT: mov p2.b, p3/m, p3.b
; CHECK-NEXT: sel p2.b, p2, p2.b, p4.b
; CHECK-NEXT: ldr p4, [sp, #7, mul vl] // 2-byte Reload
-; CHECK-NEXT: mov p1.b, p2/m, p2.b
-; CHECK-NEXT: and p0.b, p1/z, p1.b, p0.b
+; CHECK-NEXT: orr p0.b, p0/z, p2.b, p1.b
; CHECK-NEXT: addvl sp, sp, #1
; CHECK-NEXT: ldr x29, [sp], #16 // 8-byte Folded Reload
; CHECK-NEXT: ret
@@ -491,8 +487,7 @@ define <vscale x 2 x i1> @match_nxv2xi64_v2i64(<vscale x 2 x i64> %op1, <2 x i64
; CHECK-NEXT: mov z1.d, d1
; CHECK-NEXT: cmpeq p2.d, p1/z, z0.d, z2.d
; CHECK-NEXT: cmpeq p3.d, p1/z, z0.d, z1.d
-; CHECK-NEXT: sel p1.b, p3, p3.b, p2.b
-; CHECK-NEXT: and p0.b, p1/z, p1.b, p0.b
+; CHECK-NEXT: orr p0.b, p0/z, p3.b, p2.b
; CHECK-NEXT: ret
%r = tail call <vscale x 2 x i1> @llvm.experimental.vector.match(<vscale x 2 x i64> %op1, <2 x i64> %op2, <vscale x 2 x i1> %mask)
ret <vscale x 2 x i1> %r
diff --git a/llvm/test/CodeGen/AArch64/sve-pred-log.ll b/llvm/test/CodeGen/AArch64/sve-pred-log.ll
index 418dc503695a4..d499311e16b5f 100644
--- a/llvm/test/CodeGen/AArch64/sve-pred-log.ll
+++ b/llvm/test/CodeGen/AArch64/sve-pred-log.ll
@@ -49,8 +49,7 @@ define <vscale x 1 x i1> @and_nxv1i1(<vscale x 1 x i1> %a, <vscale x 1 x i1> %b)
define <vscale x 16 x i1> @and_and_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16 x i1> %a, <vscale x 16 x i1> %b) {
; CHECK-LABEL: and_and_nxv16i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: and p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: and p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = and <vscale x 16 x i1> %a, %b
%res = and <vscale x 16 x i1> %pg, %op
@@ -60,8 +59,7 @@ define <vscale x 16 x i1> @and_and_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16
define <vscale x 8 x i1> @and_and_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1> %a, <vscale x 8 x i1> %b) {
; CHECK-LABEL: and_and_nxv8i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: and p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: and p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = and <vscale x 8 x i1> %a, %b
%res = and <vscale x 8 x i1> %pg, %op
@@ -71,8 +69,7 @@ define <vscale x 8 x i1> @and_and_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1
define <vscale x 4 x i1> @and_and_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1> %a, <vscale x 4 x i1> %b) {
; CHECK-LABEL: and_and_nxv4i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: and p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: and p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = and <vscale x 4 x i1> %a, %b
%res = and <vscale x 4 x i1> %pg, %op
@@ -82,8 +79,7 @@ define <vscale x 4 x i1> @and_and_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1
define <vscale x 2 x i1> @and_and_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1> %a, <vscale x 2 x i1> %b) {
; CHECK-LABEL: and_and_nxv2i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: and p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: and p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = and <vscale x 2 x i1> %a, %b
%res = and <vscale x 2 x i1> %pg, %op
@@ -93,8 +89,7 @@ define <vscale x 2 x i1> @and_and_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1
define <vscale x 1 x i1> @and_and_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1> %a, <vscale x 1 x i1> %b) {
; CHECK-LABEL: and_and_nxv1i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: and p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: and p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = and <vscale x 1 x i1> %a, %b
%res = and <vscale x 1 x i1> %pg, %op
@@ -104,8 +99,7 @@ define <vscale x 1 x i1> @and_and_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1
define <vscale x 16 x i1> @and_bic_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16 x i1> %a, <vscale x 16 x i1> %b) {
; CHECK-LABEL: and_bic_nxv16i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: bic p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: bic p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%not.b = xor <vscale x 16 x i1> %b, splat (i1 true)
%op = and <vscale x 16 x i1> %a, %not.b
@@ -116,8 +110,7 @@ define <vscale x 16 x i1> @and_bic_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16
define <vscale x 8 x i1> @and_bic_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1> %a, <vscale x 8 x i1> %b) {
; CHECK-LABEL: and_bic_nxv8i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: bic p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: bic p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%not.b = xor <vscale x 8 x i1> %b, splat (i1 true)
%op = and <vscale x 8 x i1> %a, %not.b
@@ -128,8 +121,7 @@ define <vscale x 8 x i1> @and_bic_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1
define <vscale x 4 x i1> @and_bic_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1> %a, <vscale x 4 x i1> %b) {
; CHECK-LABEL: and_bic_nxv4i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: bic p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: bic p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%not.b = xor <vscale x 4 x i1> %b, splat (i1 true)
%op = and <vscale x 4 x i1> %a, %not.b
@@ -140,8 +132,7 @@ define <vscale x 4 x i1> @and_bic_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1
define <vscale x 2 x i1> @and_bic_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1> %a, <vscale x 2 x i1> %b) {
; CHECK-LABEL: and_bic_nxv2i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: bic p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: bic p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%not.b = xor <vscale x 2 x i1> %b, splat (i1 true)
%op = and <vscale x 2 x i1> %a, %not.b
@@ -152,8 +143,7 @@ define <vscale x 2 x i1> @and_bic_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1
define <vscale x 1 x i1> @and_bic_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1> %a, <vscale x 1 x i1> %b) {
; CHECK-LABEL: and_bic_nxv1i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: bic p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: bic p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%not.b = xor <vscale x 1 x i1> %b, splat (i1 true)
%op = and <vscale x 1 x i1> %a, %not.b
@@ -164,9 +154,7 @@ define <vscale x 1 x i1> @and_bic_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1
define <vscale x 16 x i1> @and_eor_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16 x i1> %a, <vscale x 16 x i1> %b) {
; CHECK-LABEL: and_eor_nxv16i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: ptrue p3.b
-; CHECK-NEXT: eor p1.b, p3/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: eor p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = xor <vscale x 16 x i1> %a, %b
%res = and <vscale x 16 x i1> %pg, %op
@@ -176,9 +164,7 @@ define <vscale x 16 x i1> @and_eor_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16
define <vscale x 8 x i1> @and_eor_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1> %a, <vscale x 8 x i1> %b) {
; CHECK-LABEL: and_eor_nxv8i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: ptrue p3.h
-; CHECK-NEXT: eor p1.b, p3/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: eor p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = xor <vscale x 8 x i1> %a, %b
%res = and <vscale x 8 x i1> %pg, %op
@@ -188,9 +174,7 @@ define <vscale x 8 x i1> @and_eor_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1
define <vscale x 4 x i1> @and_eor_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1> %a, <vscale x 4 x i1> %b) {
; CHECK-LABEL: and_eor_nxv4i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: ptrue p3.s
-; CHECK-NEXT: eor p1.b, p3/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: eor p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = xor <vscale x 4 x i1> %a, %b
%res = and <vscale x 4 x i1> %pg, %op
@@ -200,9 +184,7 @@ define <vscale x 4 x i1> @and_eor_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1
define <vscale x 2 x i1> @and_eor_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1> %a, <vscale x 2 x i1> %b) {
; CHECK-LABEL: and_eor_nxv2i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: ptrue p3.d
-; CHECK-NEXT: eor p1.b, p3/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: eor p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = xor <vscale x 2 x i1> %a, %b
%res = and <vscale x 2 x i1> %pg, %op
@@ -212,9 +194,7 @@ define <vscale x 2 x i1> @and_eor_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1
define <vscale x 1 x i1> @and_eor_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1> %a, <vscale x 1 x i1> %b) {
; CHECK-LABEL: and_eor_nxv1i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: ptrue p3.d
-; CHECK-NEXT: eor p1.b, p3/z, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: eor p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = xor <vscale x 1 x i1> %a, %b
%res = and <vscale x 1 x i1> %pg, %op
@@ -224,10 +204,7 @@ define <vscale x 1 x i1> @and_eor_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1
define <vscale x 16 x i1> @and_orn_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16 x i1> %a, <vscale x 16 x i1> %b) {
; CHECK-LABEL: and_orn_nxv16i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: ptrue p3.b
-; CHECK-NEXT: not p2.b, p3/z, p2.b
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: orn p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%not.b = xor <vscale x 16 x i1> %b, splat (i1 true)
%op = or <vscale x 16 x i1> %a, %not.b
@@ -238,10 +215,7 @@ define <vscale x 16 x i1> @and_orn_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16
define <vscale x 8 x i1> @and_orn_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1> %a, <vscale x 8 x i1> %b) {
; CHECK-LABEL: and_orn_nxv8i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: ptrue p3.h
-; CHECK-NEXT: not p2.b, p3/z, p2.b
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: orn p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%not.b = xor <vscale x 8 x i1> %b, splat (i1 true)
%op = or <vscale x 8 x i1> %a, %not.b
@@ -252,10 +226,7 @@ define <vscale x 8 x i1> @and_orn_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1
define <vscale x 4 x i1> @and_orn_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1> %a, <vscale x 4 x i1> %b) {
; CHECK-LABEL: and_orn_nxv4i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: ptrue p3.s
-; CHECK-NEXT: not p2.b, p3/z, p2.b
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: orn p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%not.b = xor <vscale x 4 x i1> %b, splat (i1 true)
%op = or <vscale x 4 x i1> %a, %not.b
@@ -266,10 +237,7 @@ define <vscale x 4 x i1> @and_orn_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1
define <vscale x 2 x i1> @and_orn_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1> %a, <vscale x 2 x i1> %b) {
; CHECK-LABEL: and_orn_nxv2i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: ptrue p3.d
-; CHECK-NEXT: not p2.b, p3/z, p2.b
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: orn p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%not.b = xor <vscale x 2 x i1> %b, splat (i1 true)
%op = or <vscale x 2 x i1> %a, %not.b
@@ -280,19 +248,7 @@ define <vscale x 2 x i1> @and_orn_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1
define <vscale x 1 x i1> @and_orn_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1> %a, <vscale x 1 x i1> %b) {
; CHECK-LABEL: and_orn_nxv1i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: str x29, [sp, #-16]! // 8-byte Folded Spill
-; CHECK-NEXT: addvl sp, sp, #-1
-; CHECK-NEXT: str p4, [sp, #7, mul vl] // 2-byte Spill
-; CHECK-NEXT: .cfi_escape 0x0f, 0x08, 0x8f, 0x10, 0x92, 0x2e, 0x00, 0x38, 0x1e, 0x22 // sp + 16 + 8 * VG
-; CHECK-NEXT: .cfi_offset w29, -16
-; CHECK-NEXT: ptrue p3.d
-; CHECK-NEXT: punpklo p4.h, p3.b
-; CHECK-NEXT: eor p2.b, p3/z, p2.b, p4.b
-; CHECK-NEXT: ldr p4, [sp, #7, mul vl] // 2-byte Reload
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
-; CHECK-NEXT: addvl sp, sp, #1
-; CHECK-NEXT: ldr x29, [sp], #16 // 8-byte Folded Reload
+; CHECK-NEXT: orn p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%not.b = xor <vscale x 1 x i1> %b, splat (i1 true)
%op = or <vscale x 1 x i1> %a, %not.b
@@ -303,8 +259,7 @@ define <vscale x 1 x i1> @and_orn_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1
define <vscale x 16 x i1> @and_orr_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16 x i1> %a, <vscale x 16 x i1> %b) {
; CHECK-LABEL: and_orr_nxv16i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: orr p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = or <vscale x 16 x i1> %a, %b
%res = and <vscale x 16 x i1> %pg, %op
@@ -314,8 +269,7 @@ define <vscale x 16 x i1> @and_orr_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16
define <vscale x 8 x i1> @and_orr_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1> %a, <vscale x 8 x i1> %b) {
; CHECK-LABEL: and_orr_nxv8i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: orr p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = or <vscale x 8 x i1> %a, %b
%res = and <vscale x 8 x i1> %pg, %op
@@ -325,8 +279,7 @@ define <vscale x 8 x i1> @and_orr_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1
define <vscale x 4 x i1> @and_orr_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1> %a, <vscale x 4 x i1> %b) {
; CHECK-LABEL: and_orr_nxv4i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: orr p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = or <vscale x 4 x i1> %a, %b
%res = and <vscale x 4 x i1> %pg, %op
@@ -336,8 +289,7 @@ define <vscale x 4 x i1> @and_orr_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1
define <vscale x 2 x i1> @and_orr_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1> %a, <vscale x 2 x i1> %b) {
; CHECK-LABEL: and_orr_nxv2i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: orr p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = or <vscale x 2 x i1> %a, %b
%res = and <vscale x 2 x i1> %pg, %op
@@ -347,8 +299,7 @@ define <vscale x 2 x i1> @and_orr_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1
define <vscale x 1 x i1> @and_orr_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1> %a, <vscale x 1 x i1> %b) {
; CHECK-LABEL: and_orr_nxv1i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: and p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: orr p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%op = or <vscale x 1 x i1> %a, %b
%res = and <vscale x 1 x i1> %pg, %op
@@ -358,8 +309,7 @@ define <vscale x 1 x i1> @and_orr_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1
define <vscale x 16 x i1> @and_nand_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16 x i1> %a, <vscale x 16 x i1> %b) {
; CHECK-LABEL: and_nand_nxv16i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: and p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: bic p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: nand p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%and = and <vscale x 16 x i1> %a, %b
%op = xor <vscale x 16 x i1> %and, splat (i1 true)
@@ -370,8 +320,7 @@ define <vscale x 16 x i1> @and_nand_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16
define <vscale x 8 x i1> @and_nand_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1> %a, <vscale x 8 x i1> %b) {
; CHECK-LABEL: and_nand_nxv8i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: and p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: bic p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: nand p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%and = and <vscale x 8 x i1> %a, %b
%op = xor <vscale x 8 x i1> %and, splat (i1 true)
@@ -382,8 +331,7 @@ define <vscale x 8 x i1> @and_nand_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i
define <vscale x 4 x i1> @and_nand_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1> %a, <vscale x 4 x i1> %b) {
; CHECK-LABEL: and_nand_nxv4i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: and p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: bic p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: nand p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%and = and <vscale x 4 x i1> %a, %b
%op = xor <vscale x 4 x i1> %and, splat (i1 true)
@@ -394,8 +342,7 @@ define <vscale x 4 x i1> @and_nand_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i
define <vscale x 2 x i1> @and_nand_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1> %a, <vscale x 2 x i1> %b) {
; CHECK-LABEL: and_nand_nxv2i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: and p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: bic p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: nand p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%and = and <vscale x 2 x i1> %a, %b
%op = xor <vscale x 2 x i1> %and, splat (i1 true)
@@ -406,8 +353,7 @@ define <vscale x 2 x i1> @and_nand_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i
define <vscale x 1 x i1> @and_nand_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1> %a, <vscale x 1 x i1> %b) {
; CHECK-LABEL: and_nand_nxv1i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: and p2.b, p1/z, p1.b, p2.b
-; CHECK-NEXT: bic p0.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: nand p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%and = and <vscale x 1 x i1> %a, %b
%op = xor <vscale x 1 x i1> %and, splat (i1 true)
@@ -418,8 +364,7 @@ define <vscale x 1 x i1> @and_nand_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i
define <vscale x 16 x i1> @and_nor_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16 x i1> %a, <vscale x 16 x i1> %b) {
; CHECK-LABEL: and_nor_nxv16i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: bic p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: nor p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%or = or <vscale x 16 x i1> %a, %b
%op = xor <vscale x 16 x i1> %or, splat (i1 true)
@@ -430,8 +375,7 @@ define <vscale x 16 x i1> @and_nor_nxv16i1(<vscale x 16 x i1> %pg, <vscale x 16
define <vscale x 8 x i1> @and_nor_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1> %a, <vscale x 8 x i1> %b) {
; CHECK-LABEL: and_nor_nxv8i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: bic p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: nor p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%or = or <vscale x 8 x i1> %a, %b
%op = xor <vscale x 8 x i1> %or, splat (i1 true)
@@ -442,8 +386,7 @@ define <vscale x 8 x i1> @and_nor_nxv8i1(<vscale x 8 x i1> %pg, <vscale x 8 x i1
define <vscale x 4 x i1> @and_nor_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1> %a, <vscale x 4 x i1> %b) {
; CHECK-LABEL: and_nor_nxv4i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: bic p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: nor p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%or = or <vscale x 4 x i1> %a, %b
%op = xor <vscale x 4 x i1> %or, splat (i1 true)
@@ -454,8 +397,7 @@ define <vscale x 4 x i1> @and_nor_nxv4i1(<vscale x 4 x i1> %pg, <vscale x 4 x i1
define <vscale x 2 x i1> @and_nor_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1> %a, <vscale x 2 x i1> %b) {
; CHECK-LABEL: and_nor_nxv2i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: bic p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: nor p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%or = or <vscale x 2 x i1> %a, %b
%op = xor <vscale x 2 x i1> %or, splat (i1 true)
@@ -466,8 +408,7 @@ define <vscale x 2 x i1> @and_nor_nxv2i1(<vscale x 2 x i1> %pg, <vscale x 2 x i1
define <vscale x 1 x i1> @and_nor_nxv1i1(<vscale x 1 x i1> %pg, <vscale x 1 x i1> %a, <vscale x 1 x i1> %b) {
; CHECK-LABEL: and_nor_nxv1i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: sel p1.b, p1, p1.b, p2.b
-; CHECK-NEXT: bic p0.b, p0/z, p0.b, p1.b
+; CHECK-NEXT: nor p0.b, p0/z, p1.b, p2.b
; CHECK-NEXT: ret
%or = or <vscale x 1 x i1> %a, %b
%op = xor <vscale x 1 x i1> %or, splat (i1 true)
diff --git a/llvm/test/CodeGen/AArch64/sve-split-int-pred-reduce.ll b/llvm/test/CodeGen/AArch64/sve-split-int-pred-reduce.ll
index a6cae29081b5e..7c07cd28f3e42 100644
--- a/llvm/test/CodeGen/AArch64/sve-split-int-pred-reduce.ll
+++ b/llvm/test/CodeGen/AArch64/sve-split-int-pred-reduce.ll
@@ -18,12 +18,19 @@ define i1 @andv_nxv32i1(<vscale x 32 x i1> %a) {
define i1 @andv_nxv64i1(<vscale x 64 x i1> %a) {
; CHECK-LABEL: andv_nxv64i1:
; CHECK: // %bb.0:
-; CHECK-NEXT: and p3.b, p1/z, p1.b, p3.b
-; CHECK-NEXT: and p1.b, p0/z, p0.b, p2.b
-; CHECK-NEXT: and p0.b, p1/z, p1.b, p3.b
-; CHECK-NEXT: ptrue p1.b
-; CHECK-NEXT: nots p0.b, p1/z, p0.b
+; CHECK-NEXT: str x29, [sp, #-16]! // 8-byte Folded Spill
+; CHECK-NEXT: addvl sp, sp, #-1
+; CHECK-NEXT: str p4, [sp, #7, mul vl] // 2-byte Spill
+; CHECK-NEXT: .cfi_escape 0x0f, 0x08, 0x8f, 0x10, 0x92, 0x2e, 0x00, 0x38, 0x1e, 0x22 // sp + 16 + 8 * VG
+; CHECK-NEXT: .cfi_offset w29, -16
+; CHECK-NEXT: and p2.b, p0/z, p0.b, p2.b
+; CHECK-NEXT: ptrue p4.b
+; CHECK-NEXT: and p0.b, p2/z, p1.b, p3.b
+; CHECK-NEXT: nots p0.b, p4/z, p0.b
+; CHECK-NEXT: ldr p4, [sp, #7, mul vl] // 2-byte Reload
; CHECK-NEXT: cset w0, eq
+; CHECK-NEXT: addvl sp, sp, #1
+; CHECK-NEXT: ldr x29, [sp], #16 // 8-byte Folded Reload
; CHECK-NEXT: ret
%res = call i1 @llvm.vector.reduce.and.nxv64i1(<vscale x 64 x i1> %a)
ret i1 %res
>From f73653c2999557eed98201edc31bbffe4be1fc82 Mon Sep 17 00:00:00 2001
From: Paul Walker <paul.walker at arm.com>
Date: Wed, 1 Jul 2026 10:34:32 +0000
Subject: [PATCH 2/2] Extend testing to include streaming-sve.
---
llvm/test/CodeGen/AArch64/sve-pred-log.ll | 5 ++++-
1 file changed, 4 insertions(+), 1 deletion(-)
diff --git a/llvm/test/CodeGen/AArch64/sve-pred-log.ll b/llvm/test/CodeGen/AArch64/sve-pred-log.ll
index d499311e16b5f..7d2fea13dfc2e 100644
--- a/llvm/test/CodeGen/AArch64/sve-pred-log.ll
+++ b/llvm/test/CodeGen/AArch64/sve-pred-log.ll
@@ -1,5 +1,8 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -mtriple=aarch64-unknown-linux-gnu -mattr=+sve < %s | FileCheck %s
+; RUN: llc -mattr=+sve < %s | FileCheck %s
+; RUN: llc -mattr=+sme -force-streaming < %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
define <vscale x 16 x i1> @and_nxv16i1(<vscale x 16 x i1> %a, <vscale x 16 x i1> %b) {
; CHECK-LABEL: and_nxv16i1:
More information about the llvm-commits
mailing list