[llvm] [LV][AArch64] Support partial reductions of extended compares (PR #212190)
Adam Scott via llvm-commits
llvm-commits at lists.llvm.org
Fri Jul 31 22:07:30 PDT 2026
https://github.com/as4230 updated https://github.com/llvm/llvm-project/pull/212190
>From 11597be7adf1acf62cf4b1fb39ab55f21935836d Mon Sep 17 00:00:00 2001
From: Adam Scott <adamscott200322 at gmail.com>
Date: Mon, 27 Jul 2026 05:59:30 +0000
Subject: [PATCH 1/3] [LV][AArch64] Add tests for partial reductions of
extended compares NFC
---
.../neon-partial-reduce-dot-product.ll | 126 ++++
.../AArch64/partial-reduce-i1.ll | 698 ++++++++++++++++++
2 files changed, 824 insertions(+)
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll
diff --git a/llvm/test/CodeGen/AArch64/neon-partial-reduce-dot-product.ll b/llvm/test/CodeGen/AArch64/neon-partial-reduce-dot-product.ll
index b5801f8f48057..be64a75bb6e87 100644
--- a/llvm/test/CodeGen/AArch64/neon-partial-reduce-dot-product.ll
+++ b/llvm/test/CodeGen/AArch64/neon-partial-reduce-dot-product.ll
@@ -1650,3 +1650,129 @@ entry:
%partial.reduce = tail call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v16i32(<2 x i32> %acc, <16 x i32> %mult)
ret <2 x i32> %partial.reduce
}
+
+define <2 x i64> @partial_reduce_zext_cmp_i8tov2i64(<2 x i64> %acc, <16 x i8> %a, <16 x i8> %b) {
+; CHECK-NODOT-LABEL: partial_reduce_zext_cmp_i8tov2i64:
+; CHECK-NODOT: // %bb.0:
+; CHECK-NODOT-NEXT: cmeq v1.16b, v1.16b, v2.16b
+; CHECK-NODOT-NEXT: mov w8, #1 // =0x1
+; CHECK-NODOT-NEXT: dup v5.2d, x8
+; CHECK-NODOT-NEXT: ushll2 v2.8h, v1.16b, #0
+; CHECK-NODOT-NEXT: ushll v1.8h, v1.8b, #0
+; CHECK-NODOT-NEXT: ushll v3.4s, v2.4h, #0
+; CHECK-NODOT-NEXT: ushll2 v4.4s, v1.8h, #0
+; CHECK-NODOT-NEXT: ushll v1.4s, v1.4h, #0
+; CHECK-NODOT-NEXT: ushll2 v2.4s, v2.8h, #0
+; CHECK-NODOT-NEXT: ushll v6.2d, v3.2s, #0
+; CHECK-NODOT-NEXT: ushll2 v7.2d, v4.4s, #0
+; CHECK-NODOT-NEXT: ushll v4.2d, v4.2s, #0
+; CHECK-NODOT-NEXT: ushll2 v16.2d, v1.4s, #0
+; CHECK-NODOT-NEXT: ushll v1.2d, v1.2s, #0
+; CHECK-NODOT-NEXT: ushll2 v3.2d, v3.4s, #0
+; CHECK-NODOT-NEXT: ushll2 v17.2d, v2.4s, #0
+; CHECK-NODOT-NEXT: ushll v2.2d, v2.2s, #0
+; CHECK-NODOT-NEXT: and v6.16b, v6.16b, v5.16b
+; CHECK-NODOT-NEXT: and v7.16b, v7.16b, v5.16b
+; CHECK-NODOT-NEXT: and v4.16b, v4.16b, v5.16b
+; CHECK-NODOT-NEXT: and v16.16b, v16.16b, v5.16b
+; CHECK-NODOT-NEXT: and v1.16b, v1.16b, v5.16b
+; CHECK-NODOT-NEXT: and v3.16b, v3.16b, v5.16b
+; CHECK-NODOT-NEXT: and v2.16b, v2.16b, v5.16b
+; CHECK-NODOT-NEXT: add v0.2d, v0.2d, v1.2d
+; CHECK-NODOT-NEXT: add v1.2d, v16.2d, v4.2d
+; CHECK-NODOT-NEXT: add v4.2d, v7.2d, v6.2d
+; CHECK-NODOT-NEXT: and v6.16b, v17.16b, v5.16b
+; CHECK-NODOT-NEXT: add v0.2d, v0.2d, v1.2d
+; CHECK-NODOT-NEXT: add v1.2d, v4.2d, v3.2d
+; CHECK-NODOT-NEXT: add v0.2d, v0.2d, v1.2d
+; CHECK-NODOT-NEXT: add v1.2d, v2.2d, v6.2d
+; CHECK-NODOT-NEXT: add v0.2d, v0.2d, v1.2d
+; CHECK-NODOT-NEXT: ret
+;
+; CHECK-DOT-LABEL: partial_reduce_zext_cmp_i8tov2i64:
+; CHECK-DOT: // %bb.0:
+; CHECK-DOT-NEXT: movi v3.16b, #1
+; CHECK-DOT-NEXT: cmeq v1.16b, v1.16b, v2.16b
+; CHECK-DOT-NEXT: movi v2.2d, #0000000000000000
+; CHECK-DOT-NEXT: and v1.16b, v1.16b, v3.16b
+; CHECK-DOT-NEXT: udot v2.4s, v1.16b, v3.16b
+; CHECK-DOT-NEXT: uaddw v0.2d, v0.2d, v2.2s
+; CHECK-DOT-NEXT: uaddw2 v0.2d, v0.2d, v2.4s
+; CHECK-DOT-NEXT: ret
+;
+; CHECK-DOT-I8MM-LABEL: partial_reduce_zext_cmp_i8tov2i64:
+; CHECK-DOT-I8MM: // %bb.0:
+; CHECK-DOT-I8MM-NEXT: movi v3.16b, #1
+; CHECK-DOT-I8MM-NEXT: cmeq v1.16b, v1.16b, v2.16b
+; CHECK-DOT-I8MM-NEXT: movi v2.2d, #0000000000000000
+; CHECK-DOT-I8MM-NEXT: and v1.16b, v1.16b, v3.16b
+; CHECK-DOT-I8MM-NEXT: udot v2.4s, v1.16b, v3.16b
+; CHECK-DOT-I8MM-NEXT: uaddw v0.2d, v0.2d, v2.2s
+; CHECK-DOT-I8MM-NEXT: uaddw2 v0.2d, v0.2d, v2.4s
+; CHECK-DOT-I8MM-NEXT: ret
+ %cmp = icmp eq <16 x i8> %a, %b
+ %ext = zext <16 x i1> %cmp to <16 x i64>
+ %partial.reduce = tail call <2 x i64> @llvm.vector.partial.reduce.add.v2i64.v16i64(<2 x i64> %acc, <16 x i64> %ext)
+ ret <2 x i64> %partial.reduce
+}
+
+define <4 x i32> @partial_reduce_zext_cmp_i8tov4i32(<4 x i32> %acc, <16 x i8> %a, <16 x i8> %b) {
+; CHECK-NODOT-LABEL: partial_reduce_zext_cmp_i8tov4i32:
+; CHECK-NODOT: // %bb.0:
+; CHECK-NODOT-NEXT: cmeq v1.16b, v1.16b, v2.16b
+; CHECK-NODOT-NEXT: movi v3.4s, #1
+; CHECK-NODOT-NEXT: ushll2 v2.8h, v1.16b, #0
+; CHECK-NODOT-NEXT: ushll v1.8h, v1.8b, #0
+; CHECK-NODOT-NEXT: ushll v4.4s, v2.4h, #0
+; CHECK-NODOT-NEXT: ushll2 v5.4s, v1.8h, #0
+; CHECK-NODOT-NEXT: ushll v1.4s, v1.4h, #0
+; CHECK-NODOT-NEXT: ushll2 v2.4s, v2.8h, #0
+; CHECK-NODOT-NEXT: and v4.16b, v4.16b, v3.16b
+; CHECK-NODOT-NEXT: and v5.16b, v5.16b, v3.16b
+; CHECK-NODOT-NEXT: and v1.16b, v1.16b, v3.16b
+; CHECK-NODOT-NEXT: and v2.16b, v2.16b, v3.16b
+; CHECK-NODOT-NEXT: add v0.4s, v0.4s, v1.4s
+; CHECK-NODOT-NEXT: add v1.4s, v5.4s, v4.4s
+; CHECK-NODOT-NEXT: add v0.4s, v0.4s, v1.4s
+; CHECK-NODOT-NEXT: add v0.4s, v0.4s, v2.4s
+; CHECK-NODOT-NEXT: ret
+;
+; CHECK-DOT-LABEL: partial_reduce_zext_cmp_i8tov4i32:
+; CHECK-DOT: // %bb.0:
+; CHECK-DOT-NEXT: movi v3.16b, #1
+; CHECK-DOT-NEXT: cmeq v1.16b, v1.16b, v2.16b
+; CHECK-DOT-NEXT: and v1.16b, v1.16b, v3.16b
+; CHECK-DOT-NEXT: udot v0.4s, v1.16b, v3.16b
+; CHECK-DOT-NEXT: ret
+;
+; CHECK-DOT-I8MM-LABEL: partial_reduce_zext_cmp_i8tov4i32:
+; CHECK-DOT-I8MM: // %bb.0:
+; CHECK-DOT-I8MM-NEXT: movi v3.16b, #1
+; CHECK-DOT-I8MM-NEXT: cmeq v1.16b, v1.16b, v2.16b
+; CHECK-DOT-I8MM-NEXT: and v1.16b, v1.16b, v3.16b
+; CHECK-DOT-I8MM-NEXT: udot v0.4s, v1.16b, v3.16b
+; CHECK-DOT-I8MM-NEXT: ret
+ %cmp = icmp eq <16 x i8> %a, %b
+ %ext = zext <16 x i1> %cmp to <16 x i32>
+ %partial.reduce = tail call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> %acc, <16 x i32> %ext)
+ ret <4 x i32> %partial.reduce
+}
+
+define <2 x i64> @partial_reduce_zext_cmp_i32tov2i64(<2 x i64> %acc, <4 x i32> %a, <4 x i32> %b) {
+; CHECK-COMMON-LABEL: partial_reduce_zext_cmp_i32tov2i64:
+; CHECK-COMMON: // %bb.0:
+; CHECK-COMMON-NEXT: cmeq v1.4s, v1.4s, v2.4s
+; CHECK-COMMON-NEXT: mov w8, #1 // =0x1
+; CHECK-COMMON-NEXT: dup v2.2d, x8
+; CHECK-COMMON-NEXT: ushll v3.2d, v1.2s, #0
+; CHECK-COMMON-NEXT: ushll2 v1.2d, v1.4s, #0
+; CHECK-COMMON-NEXT: and v3.16b, v3.16b, v2.16b
+; CHECK-COMMON-NEXT: and v1.16b, v1.16b, v2.16b
+; CHECK-COMMON-NEXT: add v0.2d, v0.2d, v3.2d
+; CHECK-COMMON-NEXT: add v0.2d, v0.2d, v1.2d
+; CHECK-COMMON-NEXT: ret
+ %cmp = icmp eq <4 x i32> %a, %b
+ %ext = zext <4 x i1> %cmp to <4 x i64>
+ %partial.reduce = tail call <2 x i64> @llvm.vector.partial.reduce.add.v2i64.v4i64(<2 x i64> %acc, <4 x i64> %ext)
+ ret <2 x i64> %partial.reduce
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll
new file mode 100644
index 0000000000000..e66d856e17a19
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll
@@ -0,0 +1,698 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^scalar.ph:" --version 4
+; RUN: opt -passes=loop-vectorize -force-vector-interleave=1 -enable-epilogue-vectorization=false -S < %s | FileCheck %s --check-prefixes=CHECK
+
+target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128"
+target triple = "aarch64-none-unknown-elf"
+
+define i64 @count_matches_i8_i64_dotprod(ptr %a) #0 {
+; CHECK-LABEL: define i64 @count_matches_i8_i64_dotprod(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP5:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 10)
+; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i1> [[TMP1]] to <16 x i64>
+; CHECK-NEXT: [[TMP5]] = add <16 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP3]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP5]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP4]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %add, %for.body ]
+ %gep = getelementptr inbounds nuw i8, ptr %a, i64 %iv
+ %load = load i8, ptr %gep, align 1
+ %cmp = icmp eq i8 %load, 10
+ %ext = zext i1 %cmp to i64
+ %add = add i64 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %add
+}
+
+define i64 @count_matches_i8_i64_sve(ptr %a) #1 {
+; CHECK-LABEL: define i64 @count_matches_i8_i64_sve(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 1024, [[TMP1]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 1024, [[TMP2]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 1024, [[N_MOD_VF]]
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <vscale x 16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP7:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 16 x i8>, ptr [[TMP3]], align 1
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq <vscale x 16 x i8> [[WIDE_LOAD]], splat (i8 10)
+; CHECK-NEXT: [[TMP5:%.*]] = zext <vscale x 16 x i1> [[TMP4]] to <vscale x 16 x i64>
+; CHECK-NEXT: [[TMP7]] = add <vscale x 16 x i64> [[VEC_PHI]], [[TMP5]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP6]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP8:%.*]] = call i64 @llvm.vector.reduce.add.nxv16i64(<vscale x 16 x i64> [[TMP7]])
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 1024, [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label [[EXIT:%.*]], label [[SCALAR_PH]]
+; CHECK: scalar.ph:
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %add, %for.body ]
+ %gep = getelementptr inbounds nuw i8, ptr %a, i64 %iv
+ %load = load i8, ptr %gep, align 1
+ %cmp = icmp eq i8 %load, 10
+ %ext = zext i1 %cmp to i64
+ %add = add i64 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %add
+}
+
+define i64 @count_matches_i32_i64(ptr %a) #0 {
+; CHECK-LABEL: define i64 @count_matches_i32_i64(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP5:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <4 x i32> [[WIDE_LOAD]], splat (i32 10)
+; CHECK-NEXT: [[TMP2:%.*]] = zext <4 x i1> [[TMP1]] to <4 x i64>
+; CHECK-NEXT: [[TMP5]] = add <4 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP3]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vector.reduce.add.v4i64(<4 x i64> [[TMP5]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP4]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %add, %for.body ]
+ %gep = getelementptr inbounds nuw i32, ptr %a, i64 %iv
+ %load = load i32, ptr %gep, align 4
+ %cmp = icmp eq i32 %load, 10
+ %ext = zext i1 %cmp to i64
+ %add = add i64 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %add
+}
+
+define i64 @count_matches_sext_i8_i64(ptr %a) #0 {
+; CHECK-LABEL: define i64 @count_matches_sext_i8_i64(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP5:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 10)
+; CHECK-NEXT: [[TMP2:%.*]] = sext <16 x i1> [[TMP1]] to <16 x i64>
+; CHECK-NEXT: [[TMP5]] = add <16 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP3]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP5]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP4]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %add, %for.body ]
+ %gep = getelementptr inbounds nuw i8, ptr %a, i64 %iv
+ %load = load i8, ptr %gep, align 1
+ %cmp = icmp eq i8 %load, 10
+ %ext = sext i1 %cmp to i64
+ %add = add i64 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %add
+}
+
+define i64 @count_i1_not_compare(ptr %a) #0 {
+; CHECK-LABEL: define i64 @count_i1_not_compare(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP3:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = trunc <16 x i8> [[WIDE_LOAD]] to <16 x i1>
+; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i1> [[TMP1]] to <16 x i64>
+; CHECK-NEXT: [[TMP3]] = add <16 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP4]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP5:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP3]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP5]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %add, %for.body ]
+ %gep = getelementptr inbounds nuw i8, ptr %a, i64 %iv
+ %load = load i8, ptr %gep, align 1
+ %bit = trunc i8 %load to i1
+ %ext = zext i1 %bit to i64
+ %add = add i64 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %add
+}
+
+define i16 @count_matches_i8_i16(ptr %a) #0 {
+; CHECK-LABEL: define i16 @count_matches_i8_i16(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i16> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP5:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 10)
+; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i1> [[TMP1]] to <16 x i16>
+; CHECK-NEXT: [[TMP5]] = add <16 x i16> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP3]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP4:%.*]] = call i16 @llvm.vector.reduce.add.v16i16(<16 x i16> [[TMP5]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i16 [[TMP4]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i16 [ 0, %entry ], [ %add, %for.body ]
+ %gep = getelementptr inbounds nuw i8, ptr %a, i64 %iv
+ %load = load i8, ptr %gep, align 1
+ %cmp = icmp eq i8 %load, 10
+ %ext = zext i1 %cmp to i16
+ %add = add i16 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i16 %add
+}
+
+define i64 @count_matches_sub_i8_i64(ptr %a) #0 {
+; CHECK-LABEL: define i64 @count_matches_sub_i8_i64(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP4:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 10)
+; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i1> [[TMP1]] to <16 x i64>
+; CHECK-NEXT: [[TMP4]] = sub <16 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP3]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP5:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP4]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP5]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %sub, %for.body ]
+ %gep = getelementptr inbounds nuw i8, ptr %a, i64 %iv
+ %load = load i8, ptr %gep, align 1
+ %cmp = icmp eq i8 %load, 10
+ %ext = zext i1 %cmp to i64
+ %sub = sub i64 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %sub
+}
+
+define i64 @count_matches_f32_i64(ptr %a) #0 {
+; CHECK-LABEL: define i64 @count_matches_f32_i64(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP5:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw float, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x float>, ptr [[TMP0]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = fcmp oeq <4 x float> [[WIDE_LOAD]], splat (float 1.000000e+00)
+; CHECK-NEXT: [[TMP2:%.*]] = zext <4 x i1> [[TMP1]] to <4 x i64>
+; CHECK-NEXT: [[TMP5]] = add <4 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP3]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vector.reduce.add.v4i64(<4 x i64> [[TMP5]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP4]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %add, %for.body ]
+ %gep = getelementptr inbounds nuw float, ptr %a, i64 %iv
+ %load = load float, ptr %gep, align 4
+ %cmp = fcmp oeq float %load, 1.000000e+00
+ %ext = zext i1 %cmp to i64
+ %add = add i64 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %add
+}
+
+define i64 @count_and_of_compares_i8_i64(ptr %a, ptr %b) #0 {
+; CHECK-LABEL: define i64 @count_and_of_compares_i8_i64(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP8:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 120)
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD1]], splat (i8 121)
+; CHECK-NEXT: [[TMP4:%.*]] = and <16 x i1> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = zext <16 x i1> [[TMP4]] to <16 x i64>
+; CHECK-NEXT: [[TMP8]] = add <16 x i64> [[VEC_PHI]], [[TMP5]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP6]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP8]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP7]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc = phi i64 [ 0, %entry ], [ %acc.next, %loop ]
+ %gep.a = getelementptr inbounds i8, ptr %a, i64 %iv
+ %la = load i8, ptr %gep.a, align 1
+ %gep.b = getelementptr inbounds i8, ptr %b, i64 %iv
+ %lb = load i8, ptr %gep.b, align 1
+ %ca = icmp eq i8 %la, 120
+ %cb = icmp eq i8 %lb, 121
+ %both = and i1 %ca, %cb
+ %z = zext i1 %both to i64
+ %acc.next = add i64 %acc, %z
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret i64 %acc.next
+}
+
+define i64 @count_or_of_compares_i8_i64(ptr %a) #0 {
+; CHECK-LABEL: define i64 @count_or_of_compares_i8_i64(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP7:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 10)
+; CHECK-NEXT: [[TMP2:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 13)
+; CHECK-NEXT: [[TMP3:%.*]] = or <16 x i1> [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[TMP4:%.*]] = zext <16 x i1> [[TMP3]] to <16 x i64>
+; CHECK-NEXT: [[TMP7]] = add <16 x i64> [[VEC_PHI]], [[TMP4]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP5]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP6:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP7]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP6]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc = phi i64 [ 0, %entry ], [ %acc.next, %loop ]
+ %gep.a = getelementptr inbounds i8, ptr %a, i64 %iv
+ %la = load i8, ptr %gep.a, align 1
+ %c1 = icmp eq i8 %la, 10
+ %c2 = icmp eq i8 %la, 13
+ %either = or i1 %c1, %c2
+ %z = zext i1 %either to i64
+ %acc.next = add i64 %acc, %z
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret i64 %acc.next
+}
+
+define i64 @count_and_of_mixed_width_compares(ptr %a, ptr %b) #0 {
+; CHECK-LABEL: define i64 @count_and_of_mixed_width_compares(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP6:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <16 x i16>, ptr [[TMP1]], align 2
+; CHECK-NEXT: [[TMP2:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 120)
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq <16 x i16> [[WIDE_LOAD1]], splat (i16 300)
+; CHECK-NEXT: [[TMP4:%.*]] = and <16 x i1> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = zext <16 x i1> [[TMP4]] to <16 x i64>
+; CHECK-NEXT: [[TMP6]] = add <16 x i64> [[VEC_PHI]], [[TMP5]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP7]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP8:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP6]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP8]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %acc = phi i64 [ 0, %entry ], [ %acc.next, %loop ]
+ %gep.a = getelementptr inbounds i8, ptr %a, i64 %iv
+ %la = load i8, ptr %gep.a, align 1
+ %gep.b = getelementptr inbounds i16, ptr %b, i64 %iv
+ %lb = load i16, ptr %gep.b, align 2
+ %ca = icmp eq i8 %la, 120
+ %cb = icmp eq i16 %lb, 300
+ %both = and i1 %ca, %cb
+ %z = zext i1 %both to i64
+ %acc.next = add i64 %acc, %z
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 1024
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret i64 %acc.next
+}
+
+define i64 @count_matches_i8_i64_no_dotprod(ptr %a) #2 {
+; CHECK-LABEL: define i64 @count_matches_i8_i64_no_dotprod(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR2:[0-9]+]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP3:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 10)
+; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i1> [[TMP1]] to <16 x i64>
+; CHECK-NEXT: [[TMP3]] = add <16 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP4]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP5:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP3]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP5]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %add, %for.body ]
+ %gep = getelementptr inbounds nuw i8, ptr %a, i64 %iv
+ %load = load i8, ptr %gep, align 1
+ %cmp = icmp eq i8 %load, 10
+ %ext = zext i1 %cmp to i64
+ %add = add i64 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %add
+}
+
+define i64 @count_nested_and_of_compares_i8_i64(ptr %a) #0 {
+; CHECK-LABEL: define i64 @count_nested_and_of_compares_i8_i64(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP9:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], splat (i8 10)
+; CHECK-NEXT: [[TMP2:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], splat (i8 13)
+; CHECK-NEXT: [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], splat (i8 32)
+; CHECK-NEXT: [[TMP4:%.*]] = and <16 x i1> [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[TMP5:%.*]] = and <16 x i1> [[TMP4]], [[TMP3]]
+; CHECK-NEXT: [[TMP6:%.*]] = zext <16 x i1> [[TMP5]] to <16 x i64>
+; CHECK-NEXT: [[TMP9]] = add <16 x i64> [[VEC_PHI]], [[TMP6]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP7]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP8:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP9]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP8]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %add, %for.body ]
+ %gep = getelementptr inbounds nuw i8, ptr %a, i64 %iv
+ %load = load i8, ptr %gep, align 1
+ %cmp1 = icmp ne i8 %load, 10
+ %cmp2 = icmp ne i8 %load, 13
+ %cmp3 = icmp ne i8 %load, 32
+ %and1 = and i1 %cmp1, %cmp2
+ %and2 = and i1 %and1, %cmp3
+ %ext = zext i1 %and2 to i64
+ %add = add i64 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %add
+}
+
+define i64 @count_xor_of_compares_i8_i64(ptr %a, ptr %b) #0 {
+; CHECK-LABEL: define i64 @count_xor_of_compares_i8_i64(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP8:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 10)
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD1]], splat (i8 13)
+; CHECK-NEXT: [[TMP4:%.*]] = xor <16 x i1> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = zext <16 x i1> [[TMP4]] to <16 x i64>
+; CHECK-NEXT: [[TMP8]] = add <16 x i64> [[VEC_PHI]], [[TMP5]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP6]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP16:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP8]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP7]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %add, %for.body ]
+ %gep.a = getelementptr inbounds nuw i8, ptr %a, i64 %iv
+ %load.a = load i8, ptr %gep.a, align 1
+ %gep.b = getelementptr inbounds nuw i8, ptr %b, i64 %iv
+ %load.b = load i8, ptr %gep.b, align 1
+ %cmp.a = icmp eq i8 %load.a, 10
+ %cmp.b = icmp eq i8 %load.b, 13
+ %xor = xor i1 %cmp.a, %cmp.b
+ %ext = zext i1 %xor to i64
+ %add = add i64 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %add
+}
+
+define i64 @count_pointer_compare_i64(ptr %a, ptr %t) #0 {
+; CHECK-LABEL: define i64 @count_pointer_compare_i64(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[T:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x ptr> poison, ptr [[T]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <2 x ptr> [[BROADCAST_SPLATINSERT]], <2 x ptr> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <2 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP3:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw ptr, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x ptr>, ptr [[TMP0]], align 8
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <2 x ptr> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[TMP2:%.*]] = zext <2 x i1> [[TMP1]] to <2 x i64>
+; CHECK-NEXT: [[TMP3]] = add <2 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 2
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP4]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP17:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP5:%.*]] = call i64 @llvm.vector.reduce.add.v2i64(<2 x i64> [[TMP3]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP5]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %add, %for.body ]
+ %gep = getelementptr inbounds nuw ptr, ptr %a, i64 %iv
+ %load = load ptr, ptr %gep, align 8
+ %cmp = icmp eq ptr %load, %t
+ %ext = zext i1 %cmp to i64
+ %add = add i64 %count, %ext
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %add
+}
+
+attributes #0 = { "target-features"="+neon,+dotprod" }
+attributes #1 = { "target-features"="+sve" }
+attributes #2 = { "target-features"="+neon" }
>From b7932b6b5a40a13736a0e69e729df6762699c664 Mon Sep 17 00:00:00 2001
From: Adam Scott <adamscott200322 at gmail.com>
Date: Mon, 27 Jul 2026 06:45:43 +0000
Subject: [PATCH 2/3] [LV] Cost partial reductions of extended compares from
the compare's width
---
.../AArch64/AArch64TargetTransformInfo.cpp | 12 ++++
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 5 +-
.../Transforms/Vectorize/VPlanTransforms.cpp | 9 +--
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 45 ++++++++++++++
llvm/lib/Transforms/Vectorize/VPlanUtils.h | 7 +++
.../AArch64/partial-reduce-i1.ll | 61 ++++++++++---------
6 files changed, 104 insertions(+), 35 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 03aa42b31e6aa..bfabe4aa465ed 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -6516,6 +6516,18 @@ InstructionCost AArch64TTIImpl::getPartialReductionCost(
return Cost;
}
+ // i8 -> i64 has no native dot product. LowerPARTIAL_REDUCE_MLA accumulates
+ // in two steps via i32.
+ // partial_reduce_[us]mla acc, lhs, rhs
+ // <=> movi tmp, #0
+ // [us]dot tmp, lhs, rhs
+ // [us]addw acc, acc, tmp
+ // [us]addw2 acc, acc, tmp
+ if (AccumLT.second.getScalarType() == MVT::i64 &&
+ InputLT.second.getScalarType() == MVT::i8 && !IsUSDot &&
+ IsSupported(false, ST->hasDotProd()))
+ return Cost * 4 + INegCost;
+
// f16 -> f32 is natively supported for fdot using either
// SVE or NEON instruction.
if (Opcode == Instruction::FAdd && !IsSub &&
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 5127c5963b02e..4cb66e642e9ff 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -3536,7 +3536,10 @@ InstructionCost VPExpressionRecipe::computeCost(ElementCount VF,
if (RedR->isPartialReduction())
return Ctx.TTI.getPartialReductionCost(
- Opcode, getOperand(0)->getScalarType(), nullptr, RedTy, VF,
+ Opcode,
+ vputils::getExtendSrcTypeForPartialReduction(ExtR->getOpcode(),
+ getOperand(0)),
+ nullptr, RedTy, VF,
TargetTransformInfo::getPartialReductionExtendKind(ExtR->getOpcode()),
TargetTransformInfo::PR_None, std::nullopt, Ctx.CostKind,
RedTy->isFloatingPointTy()
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index e5035c4e7767f..13ab1b2454f99 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -5083,10 +5083,11 @@ matchExtendedReductionOperand(VPWidenRecipe *UpdateR, VPValue *Op) {
// optimizeExtendsForPartialReduction.
Op = CastSource;
} else {
- return ExtendedReductionOperand{
- UpdateR,
- /*ExtendA=*/{CastSource->getScalarType(), *OuterExtKind},
- /*ExtendB=*/{}};
+ Type *SrcTy = vputils::getExtendSrcTypeForPartialReduction(
+ CastRecipe->getOpcode(), CastSource);
+ return ExtendedReductionOperand{UpdateR,
+ /*ExtendA=*/{SrcTy, *OuterExtKind},
+ /*ExtendB=*/{}};
}
}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 0bcd4dd8aba5b..973fec055d956 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -762,6 +762,51 @@ VPInstruction *vputils::findComputeReductionResult(VPReductionPHIRecipe *PhiR) {
cast<VPSingleDefRecipe>(SelR));
}
+/// Returns the type at which the i1 vector \p Mask is already materialized
+/// in a register, or nullptr if it has no such type. A compare's mask exists
+/// at the width of the compared operands and a logical combination of masks
+/// exists at the width its operands share.
+static Type *getMaskMaterializationType(VPValue *Mask, unsigned Depth = 0) {
+ if (Depth > 3)
+ return nullptr;
+
+ VPValue *A, *B;
+ if (match(Mask, m_Cmp(m_VPValue(A), m_VPValue()))) {
+ Type *CmpTy = A->getScalarType();
+ if (CmpTy->isFloatingPointTy())
+ return IntegerType::get(CmpTy->getContext(),
+ CmpTy->getPrimitiveSizeInBits());
+ if (CmpTy->isIntegerTy() && !CmpTy->isIntegerTy(1))
+ return CmpTy;
+ // Pointer compares and compares of i1 values have no suitable type.
+ return nullptr;
+ }
+
+ if (match(Mask, m_Binary<Instruction::And>(m_VPValue(A), m_VPValue(B))) ||
+ match(Mask, m_Binary<Instruction::Or>(m_VPValue(A), m_VPValue(B))) ||
+ match(Mask, m_Binary<Instruction::Xor>(m_VPValue(A), m_VPValue(B)))) {
+ Type *TyA = getMaskMaterializationType(A, Depth + 1);
+ Type *TyB = getMaskMaterializationType(B, Depth + 1);
+ return TyA == TyB ? TyA : nullptr;
+ }
+
+ return nullptr;
+}
+
+Type *vputils::getExtendSrcTypeForPartialReduction(unsigned ExtOpcode,
+ VPValue *ExtSrc) {
+ Type *SrcTy = ExtSrc->getScalarType();
+ if (!SrcTy->isIntegerTy(1))
+ return SrcTy;
+
+ // A sext of a mask needs codegen support that does not exist yet.
+ if (ExtOpcode == Instruction::ZExt)
+ if (Type *MatTy = getMaskMaterializationType(ExtSrc))
+ return MatTy;
+
+ return SrcTy;
+}
+
bool vputils::isUsedByLoadStoreAddress(const VPValue *V) {
SmallPtrSet<const VPValue *, 4> Seen;
SmallVector<const VPValue *> WorkList = {V};
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.h b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
index 8c0690a82cf95..6219f53d45fb6 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
@@ -178,6 +178,13 @@ bool isUsedByLoadStoreAddress(const VPValue *V);
/// inserted for predicated reductions or tail folding.
VPInstruction *findComputeReductionResult(VPReductionPHIRecipe *PhiR);
+/// Returns the type an extend with opcode \p ExtOpcode of \p ExtSrc should be
+/// costed as extending from when forming a partial reduction. Normally that is
+/// just the type of \p ExtSrc. A compare is the exception as its i1 result
+/// says nothing about width. The type of what it compared is used instead,
+/// which is the width its mask really occupies. Only the cost depends on this.
+Type *getExtendSrcTypeForPartialReduction(unsigned ExtOpcode, VPValue *ExtSrc);
+
/// Finds the incoming alias-mask within the vector preheader.
VPValue *findIncomingAliasMask(const VPlan &Plan);
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll
index e66d856e17a19..eb47f62716d0d 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll
@@ -13,17 +13,17 @@ define i64 @count_matches_i8_i64_dotprod(ptr %a) #0 {
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP5:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <2 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 10)
; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i1> [[TMP1]] to <16 x i64>
-; CHECK-NEXT: [[TMP5]] = add <16 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[PARTIAL_REDUCE]] = call <2 x i64> @llvm.vector.partial.reduce.add.v2i64.v16i64(<2 x i64> [[VEC_PHI]], <16 x i64> [[TMP2]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
; CHECK-NEXT: br i1 [[TMP3]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: middle.block:
-; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP5]])
+; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vector.reduce.add.v2i64(<2 x i64> [[PARTIAL_REDUCE]])
; CHECK-NEXT: br label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: ret i64 [[TMP4]]
@@ -62,17 +62,17 @@ define i64 @count_matches_i8_i64_sve(ptr %a) #1 {
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <vscale x 16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP7:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <vscale x 2 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 16 x i8>, ptr [[TMP3]], align 1
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq <vscale x 16 x i8> [[WIDE_LOAD]], splat (i8 10)
; CHECK-NEXT: [[TMP5:%.*]] = zext <vscale x 16 x i1> [[TMP4]] to <vscale x 16 x i64>
-; CHECK-NEXT: [[TMP7]] = add <vscale x 16 x i64> [[VEC_PHI]], [[TMP5]]
+; CHECK-NEXT: [[PARTIAL_REDUCE]] = call <vscale x 2 x i64> @llvm.vector.partial.reduce.add.nxv2i64.nxv16i64(<vscale x 2 x i64> [[VEC_PHI]], <vscale x 16 x i64> [[TMP5]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP6]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: middle.block:
-; CHECK-NEXT: [[TMP8:%.*]] = call i64 @llvm.vector.reduce.add.nxv16i64(<vscale x 16 x i64> [[TMP7]])
+; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.vector.reduce.add.nxv2i64(<vscale x 2 x i64> [[PARTIAL_REDUCE]])
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 1024, [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label [[EXIT:%.*]], label [[SCALAR_PH]]
; CHECK: scalar.ph:
@@ -105,17 +105,17 @@ define i64 @count_matches_i32_i64(ptr %a) #0 {
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP5:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <2 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i32, ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <4 x i32> [[WIDE_LOAD]], splat (i32 10)
; CHECK-NEXT: [[TMP2:%.*]] = zext <4 x i1> [[TMP1]] to <4 x i64>
-; CHECK-NEXT: [[TMP5]] = add <4 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[PARTIAL_REDUCE]] = call <2 x i64> @llvm.vector.partial.reduce.add.v2i64.v4i64(<2 x i64> [[VEC_PHI]], <4 x i64> [[TMP2]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
; CHECK-NEXT: br i1 [[TMP3]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: middle.block:
-; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vector.reduce.add.v4i64(<4 x i64> [[TMP5]])
+; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vector.reduce.add.v2i64(<2 x i64> [[PARTIAL_REDUCE]])
; CHECK-NEXT: br label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: ret i64 [[TMP4]]
@@ -234,17 +234,17 @@ define i16 @count_matches_i8_i16(ptr %a) #0 {
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i16> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP5:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <8 x i16> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 10)
; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i1> [[TMP1]] to <16 x i16>
-; CHECK-NEXT: [[TMP5]] = add <16 x i16> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[PARTIAL_REDUCE]] = call <8 x i16> @llvm.vector.partial.reduce.add.v8i16.v16i16(<8 x i16> [[VEC_PHI]], <16 x i16> [[TMP2]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
; CHECK-NEXT: br i1 [[TMP3]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
; CHECK: middle.block:
-; CHECK-NEXT: [[TMP4:%.*]] = call i16 @llvm.vector.reduce.add.v16i16(<16 x i16> [[TMP5]])
+; CHECK-NEXT: [[TMP4:%.*]] = call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[PARTIAL_REDUCE]])
; CHECK-NEXT: br label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: ret i16 [[TMP4]]
@@ -277,17 +277,18 @@ define i64 @count_matches_sub_i8_i64(ptr %a) #0 {
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP4:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <2 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 10)
; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i1> [[TMP1]] to <16 x i64>
-; CHECK-NEXT: [[TMP4]] = sub <16 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[PARTIAL_REDUCE]] = call <2 x i64> @llvm.vector.partial.reduce.add.v2i64.v16i64(<2 x i64> [[VEC_PHI]], <16 x i64> [[TMP2]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
; CHECK-NEXT: br i1 [[TMP3]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
; CHECK: middle.block:
-; CHECK-NEXT: [[TMP5:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP4]])
+; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vector.reduce.add.v2i64(<2 x i64> [[PARTIAL_REDUCE]])
+; CHECK-NEXT: [[TMP5:%.*]] = sub i64 0, [[TMP4]]
; CHECK-NEXT: br label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: ret i64 [[TMP5]]
@@ -320,17 +321,17 @@ define i64 @count_matches_f32_i64(ptr %a) #0 {
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP5:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <2 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw float, ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x float>, ptr [[TMP0]], align 4
; CHECK-NEXT: [[TMP1:%.*]] = fcmp oeq <4 x float> [[WIDE_LOAD]], splat (float 1.000000e+00)
; CHECK-NEXT: [[TMP2:%.*]] = zext <4 x i1> [[TMP1]] to <4 x i64>
-; CHECK-NEXT: [[TMP5]] = add <4 x i64> [[VEC_PHI]], [[TMP2]]
+; CHECK-NEXT: [[PARTIAL_REDUCE]] = call <2 x i64> @llvm.vector.partial.reduce.add.v2i64.v4i64(<2 x i64> [[VEC_PHI]], <4 x i64> [[TMP2]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
; CHECK-NEXT: br i1 [[TMP3]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
; CHECK: middle.block:
-; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vector.reduce.add.v4i64(<4 x i64> [[TMP5]])
+; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.vector.reduce.add.v2i64(<2 x i64> [[PARTIAL_REDUCE]])
; CHECK-NEXT: br label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: ret i64 [[TMP4]]
@@ -363,7 +364,7 @@ define i64 @count_and_of_compares_i8_i64(ptr %a, ptr %b) #0 {
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP8:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <2 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[INDEX]]
@@ -372,12 +373,12 @@ define i64 @count_and_of_compares_i8_i64(ptr %a, ptr %b) #0 {
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD1]], splat (i8 121)
; CHECK-NEXT: [[TMP4:%.*]] = and <16 x i1> [[TMP2]], [[TMP3]]
; CHECK-NEXT: [[TMP5:%.*]] = zext <16 x i1> [[TMP4]] to <16 x i64>
-; CHECK-NEXT: [[TMP8]] = add <16 x i64> [[VEC_PHI]], [[TMP5]]
+; CHECK-NEXT: [[PARTIAL_REDUCE]] = call <2 x i64> @llvm.vector.partial.reduce.add.v2i64.v16i64(<2 x i64> [[VEC_PHI]], <16 x i64> [[TMP5]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
; CHECK-NEXT: br i1 [[TMP6]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
; CHECK: middle.block:
-; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP8]])
+; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.vector.reduce.add.v2i64(<2 x i64> [[PARTIAL_REDUCE]])
; CHECK-NEXT: br label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: ret i64 [[TMP7]]
@@ -414,19 +415,19 @@ define i64 @count_or_of_compares_i8_i64(ptr %a) #0 {
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP7:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <2 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 10)
; CHECK-NEXT: [[TMP2:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], splat (i8 13)
; CHECK-NEXT: [[TMP3:%.*]] = or <16 x i1> [[TMP1]], [[TMP2]]
; CHECK-NEXT: [[TMP4:%.*]] = zext <16 x i1> [[TMP3]] to <16 x i64>
-; CHECK-NEXT: [[TMP7]] = add <16 x i64> [[VEC_PHI]], [[TMP4]]
+; CHECK-NEXT: [[PARTIAL_REDUCE]] = call <2 x i64> @llvm.vector.partial.reduce.add.v2i64.v16i64(<2 x i64> [[VEC_PHI]], <16 x i64> [[TMP4]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
; CHECK-NEXT: br i1 [[TMP5]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
; CHECK: middle.block:
-; CHECK-NEXT: [[TMP6:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP7]])
+; CHECK-NEXT: [[TMP6:%.*]] = call i64 @llvm.vector.reduce.add.v2i64(<2 x i64> [[PARTIAL_REDUCE]])
; CHECK-NEXT: br label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: ret i64 [[TMP6]]
@@ -555,7 +556,7 @@ define i64 @count_nested_and_of_compares_i8_i64(ptr %a) #0 {
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP9:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <2 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], splat (i8 10)
@@ -564,12 +565,12 @@ define i64 @count_nested_and_of_compares_i8_i64(ptr %a) #0 {
; CHECK-NEXT: [[TMP4:%.*]] = and <16 x i1> [[TMP1]], [[TMP2]]
; CHECK-NEXT: [[TMP5:%.*]] = and <16 x i1> [[TMP4]], [[TMP3]]
; CHECK-NEXT: [[TMP6:%.*]] = zext <16 x i1> [[TMP5]] to <16 x i64>
-; CHECK-NEXT: [[TMP9]] = add <16 x i64> [[VEC_PHI]], [[TMP6]]
+; CHECK-NEXT: [[PARTIAL_REDUCE]] = call <2 x i64> @llvm.vector.partial.reduce.add.v2i64.v16i64(<2 x i64> [[VEC_PHI]], <16 x i64> [[TMP6]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
; CHECK-NEXT: br i1 [[TMP7]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
; CHECK: middle.block:
-; CHECK-NEXT: [[TMP8:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP9]])
+; CHECK-NEXT: [[TMP8:%.*]] = call i64 @llvm.vector.reduce.add.v2i64(<2 x i64> [[PARTIAL_REDUCE]])
; CHECK-NEXT: br label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: ret i64 [[TMP8]]
@@ -606,7 +607,7 @@ define i64 @count_xor_of_compares_i8_i64(ptr %a, ptr %b) #0 {
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP8:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <2 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[INDEX]]
@@ -615,12 +616,12 @@ define i64 @count_xor_of_compares_i8_i64(ptr %a, ptr %b) #0 {
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD1]], splat (i8 13)
; CHECK-NEXT: [[TMP4:%.*]] = xor <16 x i1> [[TMP2]], [[TMP3]]
; CHECK-NEXT: [[TMP5:%.*]] = zext <16 x i1> [[TMP4]] to <16 x i64>
-; CHECK-NEXT: [[TMP8]] = add <16 x i64> [[VEC_PHI]], [[TMP5]]
+; CHECK-NEXT: [[PARTIAL_REDUCE]] = call <2 x i64> @llvm.vector.partial.reduce.add.v2i64.v16i64(<2 x i64> [[VEC_PHI]], <16 x i64> [[TMP5]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
; CHECK-NEXT: br i1 [[TMP6]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP16:![0-9]+]]
; CHECK: middle.block:
-; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP8]])
+; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.vector.reduce.add.v2i64(<2 x i64> [[PARTIAL_REDUCE]])
; CHECK-NEXT: br label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: ret i64 [[TMP7]]
>From b13dc2f8c4a4211c291788708e71e0c3b63042b1 Mon Sep 17 00:00:00 2001
From: Adam Scott <adamscott200322 at gmail.com>
Date: Sat, 1 Aug 2026 05:07:04 +0000
Subject: [PATCH 3/3] [LV] Bail out of partial reduction chains with a scale
factor of 1
---
.../Transforms/Vectorize/VPlanTransforms.cpp | 15 ++++---
.../AArch64/partial-reduce-i1.ll | 45 +++++++++++++++++++
2 files changed, 55 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 13ab1b2454f99..404f3b65b30df 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -5205,11 +5205,16 @@ getScaledReductions(VPReductionPHIRecipe *RedPhiR) {
if (!PHISize.hasKnownScalarFactor(ExtSrcSize))
return std::nullopt;
- VPPartialReductionChain Link(
- {UpdateR, *ExtendedOp, RK,
- PrevValue == UpdateR->getOperand(0) ? 0U : 1U,
- static_cast<unsigned>(PHISize.getKnownScalarFactor(ExtSrcSize)),
- Blend});
+ // A source as wide as the accumulator gives a scale factor of 1, leaving
+ // the PHI VF no narrower than the input VF, so there is nothing to scale.
+ unsigned ScaleFactor =
+ static_cast<unsigned>(PHISize.getKnownScalarFactor(ExtSrcSize));
+ if (ScaleFactor < 2)
+ return std::nullopt;
+
+ VPPartialReductionChain Link({UpdateR, *ExtendedOp, RK,
+ PrevValue == UpdateR->getOperand(0) ? 0U : 1U,
+ ScaleFactor, Blend});
Chain.push_back(Link);
CurrentValue = PrevValue;
}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll
index eb47f62716d0d..12caf70158c36 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-i1.ll
@@ -694,6 +694,51 @@ exit:
ret i64 %add
}
+define i64 @count_invariant_cmp_i64_i64(ptr %dst, i64 %x) #0 {
+; CHECK-LABEL: define i64 @count_invariant_cmp_i64_i64(
+; CHECK-SAME: ptr [[DST:%.*]], i64 [[X:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: [[TMP0:%.*]] = icmp eq i64 [[X]], 0
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <16 x i1> poison, i1 [[TMP0]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <16 x i1> [[BROADCAST_SPLATINSERT]], <16 x i1> poison, <16 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP1:%.*]] = zext <16 x i1> [[BROADCAST_SPLAT]] to <16 x i64>
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP2:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2]] = add <16 x i64> [[VEC_PHI]], [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i8, ptr [[DST]], i64 [[INDEX]]
+; CHECK-NEXT: store <16 x i8> zeroinitializer, ptr [[TMP3]], align 1
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP4]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP18:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[TMP5:%.*]] = call i64 @llvm.vector.reduce.add.v16i64(<16 x i64> [[TMP2]])
+; CHECK-NEXT: br label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i64 [[TMP5]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %count = phi i64 [ 0, %entry ], [ %add, %for.body ]
+ %cmp = icmp eq i64 %x, 0
+ %ext = zext i1 %cmp to i64
+ %add = add i64 %count, %ext
+ %gep = getelementptr i8, ptr %dst, i64 %iv
+ store i8 0, ptr %gep, align 1
+ %iv.next = add nuw i64 %iv, 1
+ %done = icmp eq i64 %iv.next, 1024
+ br i1 %done, label %exit, label %for.body
+
+exit:
+ ret i64 %add
+}
+
attributes #0 = { "target-features"="+neon,+dotprod" }
attributes #1 = { "target-features"="+sve" }
attributes #2 = { "target-features"="+neon" }
More information about the llvm-commits
mailing list