[llvm] [LV] Enable partial reductions with EVL tail folding (PR #215515)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Aug 11 03:44:33 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
@llvm/pr-subscribers-backend-risc-v
Author: Pengcheng Wang (wangpc-pp)
<details>
<summary>Changes</summary>
Partial reductions were disabled whenever the loop was tail-folded
with EVL, because VPPartialReductionRecipe had no EVL variant (see
the guard added in llvm/llvm-project#<!-- -->167863). However, a partial
reduction over a full-width vector operand is already correct under
EVL: the reduction update is masked by the EVL header mask via a
select that zeroes the inactive tail lanes, which is the reduction
identity for add. This is the same mechanism used for data-style
tail folding.
This PR moves createPartialReductions out of the !foldTailWithEVL()
guard so it runs for EVL-tail-folded loops as well.
This lets RISC-V, whose default tail-folding style is DataWithEVL,
form `vdot4a/vdot4au/vdot4asu` for i8 dot-product loops with the
`Zvdot4a8i` extension.
Assisted-by: TRAE CLI (Opus 4.8)
---
Patch is 80.45 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/215515.diff
3 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorize.cpp (+11-9)
- (added) llvm/test/Transforms/LoopVectorize/RISCV/partial-reduce-dot-product-predicated.ll (+117)
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/partial-reduce-dot-product.ll (+264-679)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 2c23ecd67917e..76397952f6e22 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -6865,17 +6865,19 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
RUN_VPLAN_PASS(VPlanTransforms::removeBranchOnConst, *Plan, false);
- // Create partial reduction recipes for scaled reductions and transform
- // recipes to abstract recipes if it is legal and beneficial and clamp the
- // range for better cost estimation.
- // TODO: Enable following transform when the EVL-version of extended-reduction
- // and mulacc-reduction are implemented.
- if (!CM.foldTailWithEVL()) {
- RUN_VPLAN_PASS(VPlanTransforms::createPartialReductions, *Plan, CostCtx,
- Range);
+ // Create partial reduction recipes for scaled reductions if it is legal and
+ // beneficial and clamp the range for better cost estimation. This is also
+ // valid when tail-folding with EVL, as the reduction update is masked to
+ // zero the inactive lanes before the (full-width) partial reduction.
+ RUN_VPLAN_PASS(VPlanTransforms::createPartialReductions, *Plan, CostCtx,
+ Range);
+ // Transform recipes to abstract recipes if it is legal and beneficial and
+ // clamp the range for better cost estimation.
+ // TODO: Enable with EVL tail folding when the EVL-version of
+ // extended-reduction and mulacc-reduction are implemented.
+ if (!CM.foldTailWithEVL())
RUN_VPLAN_PASS(VPlanTransforms::convertToAbstractRecipes, *Plan, CostCtx,
Range);
- }
// Interleave memory: for each Interleave Group we marked earlier as relevant
// for this VPlan, replace the Recipes widening its memory instructions with a
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/partial-reduce-dot-product-predicated.ll b/llvm/test/Transforms/LoopVectorize/RISCV/partial-reduce-dot-product-predicated.ll
new file mode 100644
index 0000000000000..3fa3d759283d8
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/partial-reduce-dot-product-predicated.ll
@@ -0,0 +1,117 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^scalar.ph:" --version 4
+; RUN: opt -passes=loop-vectorize -mattr=+v,+experimental-zvdot4a8i -tail-folding-policy=dont-fold-tail -S < %s | FileCheck %s --check-prefixes=NOTAILFOLD
+; RUN: opt -passes=loop-vectorize -mattr=+v,+experimental-zvdot4a8i -S < %s | FileCheck %s --check-prefixes=TAILFOLD
+
+; A data-dependent predicated dot-product reduction. This used to crash the
+; loop vectorizer under EVL tail folding (llvm/llvm-project#167861): the scaled
+; reduction PHI was fed by a plain vp.reduce.add of a mismatched type instead
+; of a partial reduction. Check that a partial reduction is formed with the
+; predication folded into the reduction input, both with and without EVL tail
+; folding.
+
+target triple = "riscv64-none-unknown-elf"
+
+define i32 @pred_dot(ptr %a, ptr %b, ptr %c, i64 %n) {
+; NOTAILFOLD-LABEL: define i32 @pred_dot(
+; NOTAILFOLD-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; NOTAILFOLD-NEXT: entry:
+; NOTAILFOLD-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; NOTAILFOLD-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
+; NOTAILFOLD-NEXT: [[TMP2:%.*]] = call i64 @llvm.umax.i64(i64 [[TMP1]], i64 8)
+; NOTAILFOLD-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP2]]
+; NOTAILFOLD-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
+; NOTAILFOLD: vector.ph:
+; NOTAILFOLD-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 2
+; NOTAILFOLD-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP3]]
+; NOTAILFOLD-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; NOTAILFOLD-NEXT: br label [[VECTOR_BODY:%.*]]
+; NOTAILFOLD: vector.body:
+; NOTAILFOLD-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; NOTAILFOLD-NEXT: [[VEC_PHI:%.*]] = phi <vscale x 1 x i32> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
+; NOTAILFOLD-NEXT: [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[C]], i64 [[INDEX]]
+; NOTAILFOLD-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i8>, ptr [[TMP4]], align 1
+; NOTAILFOLD-NEXT: [[TMP5:%.*]] = icmp ne <vscale x 4 x i8> [[WIDE_LOAD]], zeroinitializer
+; NOTAILFOLD-NEXT: [[TMP6:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX]]
+; NOTAILFOLD-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 4 x i8> @llvm.masked.load.nxv4i8.p0(ptr align 1 [[TMP6]], <vscale x 4 x i1> [[TMP5]], <vscale x 4 x i8> poison)
+; NOTAILFOLD-NEXT: [[TMP7:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX]]
+; NOTAILFOLD-NEXT: [[WIDE_MASKED_LOAD1:%.*]] = call <vscale x 4 x i8> @llvm.masked.load.nxv4i8.p0(ptr align 1 [[TMP7]], <vscale x 4 x i1> [[TMP5]], <vscale x 4 x i8> poison)
+; NOTAILFOLD-NEXT: [[TMP8:%.*]] = sext <vscale x 4 x i8> [[WIDE_MASKED_LOAD]] to <vscale x 4 x i32>
+; NOTAILFOLD-NEXT: [[TMP9:%.*]] = sext <vscale x 4 x i8> [[WIDE_MASKED_LOAD1]] to <vscale x 4 x i32>
+; NOTAILFOLD-NEXT: [[TMP10:%.*]] = mul nsw <vscale x 4 x i32> [[TMP8]], [[TMP9]]
+; NOTAILFOLD-NEXT: [[TMP11:%.*]] = select <vscale x 4 x i1> [[TMP5]], <vscale x 4 x i32> [[TMP10]], <vscale x 4 x i32> zeroinitializer
+; NOTAILFOLD-NEXT: [[PARTIAL_REDUCE]] = call <vscale x 1 x i32> @llvm.vector.partial.reduce.add.nxv1i32.nxv4i32(<vscale x 1 x i32> [[VEC_PHI]], <vscale x 4 x i32> [[TMP11]])
+; NOTAILFOLD-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
+; NOTAILFOLD-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; NOTAILFOLD-NEXT: br i1 [[TMP12]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; NOTAILFOLD: middle.block:
+; NOTAILFOLD-NEXT: [[TMP13:%.*]] = call i32 @llvm.vector.reduce.add.nxv1i32(<vscale x 1 x i32> [[PARTIAL_REDUCE]])
+; NOTAILFOLD-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; NOTAILFOLD-NEXT: br i1 [[CMP_N]], label [[FOR_EXIT:%.*]], label [[SCALAR_PH]]
+; NOTAILFOLD: scalar.ph:
+;
+; TAILFOLD-LABEL: define i32 @pred_dot(
+; TAILFOLD-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; TAILFOLD-NEXT: entry:
+; TAILFOLD-NEXT: br label [[VECTOR_PH:%.*]]
+; TAILFOLD: vector.ph:
+; TAILFOLD-NEXT: br label [[VECTOR_BODY:%.*]]
+; TAILFOLD: vector.body:
+; TAILFOLD-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], [[VECTOR_BODY]] ]
+; TAILFOLD-NEXT: [[VEC_PHI:%.*]] = phi <vscale x 1 x i32> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
+; TAILFOLD-NEXT: [[AVL:%.*]] = phi i64 [ [[N]], [[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], [[VECTOR_BODY]] ]
+; TAILFOLD-NEXT: [[TMP0:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 4, i1 true)
+; TAILFOLD-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[C]], i64 [[INDEX]]
+; TAILFOLD-NEXT: [[VP_OP_LOAD:%.*]] = call <vscale x 4 x i8> @llvm.vp.load.nxv4i8.p0(ptr align 1 [[TMP1]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP0]])
+; TAILFOLD-NEXT: [[TMP2:%.*]] = icmp ne <vscale x 4 x i8> [[VP_OP_LOAD]], zeroinitializer
+; TAILFOLD-NEXT: [[TMP3:%.*]] = call <vscale x 4 x i1> @llvm.vp.merge.nxv4i1(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i1> [[TMP2]], <vscale x 4 x i1> zeroinitializer, i32 [[TMP0]])
+; TAILFOLD-NEXT: [[TMP4:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX]]
+; TAILFOLD-NEXT: [[VP_OP_LOAD1:%.*]] = call <vscale x 4 x i8> @llvm.vp.load.nxv4i8.p0(ptr align 1 [[TMP4]], <vscale x 4 x i1> [[TMP2]], i32 [[TMP0]])
+; TAILFOLD-NEXT: [[TMP5:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX]]
+; TAILFOLD-NEXT: [[VP_OP_LOAD2:%.*]] = call <vscale x 4 x i8> @llvm.vp.load.nxv4i8.p0(ptr align 1 [[TMP5]], <vscale x 4 x i1> [[TMP2]], i32 [[TMP0]])
+; TAILFOLD-NEXT: [[TMP6:%.*]] = sext <vscale x 4 x i8> [[VP_OP_LOAD1]] to <vscale x 4 x i32>
+; TAILFOLD-NEXT: [[TMP7:%.*]] = sext <vscale x 4 x i8> [[VP_OP_LOAD2]] to <vscale x 4 x i32>
+; TAILFOLD-NEXT: [[TMP8:%.*]] = mul nsw <vscale x 4 x i32> [[TMP6]], [[TMP7]]
+; TAILFOLD-NEXT: [[TMP9:%.*]] = select <vscale x 4 x i1> [[TMP3]], <vscale x 4 x i32> [[TMP8]], <vscale x 4 x i32> zeroinitializer
+; TAILFOLD-NEXT: [[PARTIAL_REDUCE]] = call <vscale x 1 x i32> @llvm.vector.partial.reduce.add.nxv1i32.nxv4i32(<vscale x 1 x i32> [[VEC_PHI]], <vscale x 4 x i32> [[TMP9]])
+; TAILFOLD-NEXT: [[TMP10:%.*]] = zext i32 [[TMP0]] to i64
+; TAILFOLD-NEXT: [[CURRENT_ITERATION_NEXT]] = add i64 [[TMP10]], [[INDEX]]
+; TAILFOLD-NEXT: [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP10]]
+; TAILFOLD-NEXT: [[TMP11:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
+; TAILFOLD-NEXT: br i1 [[TMP11]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; TAILFOLD: middle.block:
+; TAILFOLD-NEXT: [[TMP12:%.*]] = call i32 @llvm.vector.reduce.add.nxv1i32(<vscale x 1 x i32> [[PARTIAL_REDUCE]])
+; TAILFOLD-NEXT: br label [[FOR_EXIT:%.*]]
+; TAILFOLD: for.exit:
+; TAILFOLD-NEXT: ret i32 [[TMP12]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.inc ]
+ %accum = phi i32 [ 0, %entry ], [ %sum.next, %for.inc ]
+ %gep.c = getelementptr inbounds i8, ptr %c, i64 %iv
+ %load.c = load i8, ptr %gep.c, align 1
+ %tobool = icmp eq i8 %load.c, 0
+ br i1 %tobool, label %for.inc, label %if.then
+
+if.then:
+ %gep.a = getelementptr inbounds i8, ptr %a, i64 %iv
+ %load.a = load i8, ptr %gep.a, align 1
+ %ext.a = sext i8 %load.a to i32
+ %gep.b = getelementptr inbounds i8, ptr %b, i64 %iv
+ %load.b = load i8, ptr %gep.b, align 1
+ %ext.b = sext i8 %load.b to i32
+ %mul = mul nsw i32 %ext.a, %ext.b
+ %add = add nsw i32 %accum, %mul
+ br label %for.inc
+
+for.inc:
+ %sum.next = phi i32 [ %add, %if.then ], [ %accum, %for.body ]
+ %iv.next = add i64 %iv, 1
+ %exitcond.not = icmp eq i64 %iv.next, %n
+ br i1 %exitcond.not, label %for.exit, label %for.body
+
+for.exit:
+ ret i32 %sum.next
+}
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/partial-reduce-dot-product.ll b/llvm/test/Transforms/LoopVectorize/RISCV/partial-reduce-dot-product.ll
index 85075addf1efa..2f2b30e7ab979 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/partial-reduce-dot-product.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/partial-reduce-dot-product.ll
@@ -1,183 +1,78 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^scalar.ph:" --version 4
-; RUN: opt -passes=loop-vectorize -mattr=+v -tail-folding-policy=dont-fold-tail -S < %s | FileCheck %s --check-prefixes=CHECK,V
-; RUN: opt -passes=loop-vectorize -mattr=+v,+experimental-zvdot4a8i -tail-folding-policy=dont-fold-tail -S < %s | FileCheck %s --check-prefixes=CHECK,ZVDOT4A8I
-; RUN: opt -passes=loop-vectorize -mattr=+v -scalable-vectorization=off -tail-folding-policy=dont-fold-tail -S < %s | FileCheck %s --check-prefixes=FIXED,FIXED-V
-; RUN: opt -passes=loop-vectorize -mattr=+v,+experimental-zvdot4a8i -scalable-vectorization=off -tail-folding-policy=dont-fold-tail -S < %s | FileCheck %s --check-prefixes=FIXED,FIXED-ZVDOT4A8I
-; RUN: opt -passes=loop-vectorize -mattr=+v,+experimental-zvdot4a8i -S < %s | FileCheck %s --check-prefixes=CHECK,TAILFOLD
+; RUN: opt -passes=loop-vectorize -mattr=+v -S < %s | FileCheck %s --check-prefixes=CHECK,NODOT
+; RUN: opt -passes=loop-vectorize -mattr=+v,+experimental-zvdot4a8i -S < %s | FileCheck %s --check-prefixes=CHECK,DOT
-; TODO: Remove -tail-folding-policy=dont-fold-tail when partial reductions with EVL tail folding is supported.
+; RISC-V tail-folds with EVL by default. These loops form partial reductions
+; under EVL tail folding, which lower to Zvdot4a8i extension.
target triple = "riscv64-none-unknown-elf"
-define i32 @vdot4a(ptr %a, ptr %b) #0 {
-; V-LABEL: define i32 @vdot4a(
-; V-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) #[[ATTR0:[0-9]+]] {
-; V-NEXT: entry:
-; V-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
-; V-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
-; V-NEXT: [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[TMP1]], i64 8)
-; V-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 1024, [[UMAX]]
-; V-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
-; V: vector.ph:
-; V-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 2
-; V-NEXT: [[N_MOD_VF:%.*]] = urem i64 1024, [[TMP3]]
-; V-NEXT: [[N_VEC:%.*]] = sub i64 1024, [[N_MOD_VF]]
-; V-NEXT: br label [[VECTOR_BODY:%.*]]
-; V: vector.body:
-; V-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; V-NEXT: [[VEC_PHI:%.*]] = phi <vscale x 4 x i32> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP13:%.*]], [[VECTOR_BODY]] ]
-; V-NEXT: [[TMP6:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX]]
-; V-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i8>, ptr [[TMP6]], align 1
-; V-NEXT: [[TMP8:%.*]] = sext <vscale x 4 x i8> [[WIDE_LOAD]] to <vscale x 4 x i32>
-; V-NEXT: [[TMP9:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX]]
-; V-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 4 x i8>, ptr [[TMP9]], align 1
-; V-NEXT: [[TMP11:%.*]] = sext <vscale x 4 x i8> [[WIDE_LOAD1]] to <vscale x 4 x i32>
-; V-NEXT: [[TMP12:%.*]] = mul <vscale x 4 x i32> [[TMP11]], [[TMP8]]
-; V-NEXT: [[TMP13]] = add <vscale x 4 x i32> [[TMP12]], [[VEC_PHI]]
-; V-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
-; V-NEXT: [[TMP14:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; V-NEXT: br i1 [[TMP14]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; V: middle.block:
-; V-NEXT: [[TMP15:%.*]] = call i32 @llvm.vector.reduce.add.nxv4i32(<vscale x 4 x i32> [[TMP13]])
-; V-NEXT: [[CMP_N:%.*]] = icmp eq i64 1024, [[N_VEC]]
-; V-NEXT: br i1 [[CMP_N]], label [[FOR_EXIT:%.*]], label [[SCALAR_PH]]
-; V: scalar.ph:
+define i32 @vdot4a(ptr %a, ptr %b) {
+; NODOT-LABEL: define i32 @vdot4a(
+; NODOT-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) #[[ATTR0:[0-9]+]] {
+; NODOT-NEXT: entry:
+; NODOT-NEXT: br label [[VECTOR_PH:%.*]]
+; NODOT: vector.ph:
+; NODOT-NEXT: br label [[VECTOR_BODY:%.*]]
+; NODOT: vector.body:
+; NODOT-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], [[VECTOR_BODY]] ]
+; NODOT-NEXT: [[VEC_PHI:%.*]] = phi <vscale x 4 x i32> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP7:%.*]], [[VECTOR_BODY]] ]
+; NODOT-NEXT: [[AVL:%.*]] = phi i64 [ 1024, [[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], [[VECTOR_BODY]] ]
+; NODOT-NEXT: [[TMP0:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 4, i1 true)
+; NODOT-NEXT: [[TMP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX]]
+; NODOT-NEXT: [[VP_OP_LOAD:%.*]] = call <vscale x 4 x i8> @llvm.vp.load.nxv4i8.p0(ptr align 1 [[TMP1]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP0]])
+; NODOT-NEXT: [[TMP2:%.*]] = sext <vscale x 4 x i8> [[VP_OP_LOAD]] to <vscale x 4 x i32>
+; NODOT-NEXT: [[TMP3:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX]]
+; NODOT-NEXT: [[VP_OP_LOAD1:%.*]] = call <vscale x 4 x i8> @llvm.vp.load.nxv4i8.p0(ptr align 1 [[TMP3]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP0]])
+; NODOT-NEXT: [[TMP4:%.*]] = sext <vscale x 4 x i8> [[VP_OP_LOAD1]] to <vscale x 4 x i32>
+; NODOT-NEXT: [[TMP5:%.*]] = mul <vscale x 4 x i32> [[TMP4]], [[TMP2]]
+; NODOT-NEXT: [[TMP6:%.*]] = add <vscale x 4 x i32> [[TMP5]], [[VEC_PHI]]
+; NODOT-NEXT: [[TMP7]] = call <vscale x 4 x i32> @llvm.vp.merge.nxv4i32(<vscale x 4 x i1> splat (i1 true), <vscale x 4 x i32> [[TMP6]], <vscale x 4 x i32> [[VEC_PHI]], i32 [[TMP0]])
+; NODOT-NEXT: [[TMP8:%.*]] = zext i32 [[TMP0]] to i64
+; NODOT-NEXT: [[CURRENT_ITERATION_NEXT]] = add nuw i64 [[TMP8]], [[INDEX]]
+; NODOT-NEXT: [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP8]]
+; NODOT-NEXT: [[TMP9:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
+; NODOT-NEXT: br i1 [[TMP9]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; NODOT: middle.block:
+; NODOT-NEXT: [[TMP10:%.*]] = call i32 @llvm.vector.reduce.add.nxv4i32(<vscale x 4 x i32> [[TMP7]])
+; NODOT-NEXT: br label [[FOR_EXIT:%.*]]
+; NODOT: for.exit:
+; NODOT-NEXT: ret i32 [[TMP10]]
;
-; ZVDOT4A8I-LABEL: define i32 @vdot4a(
-; ZVDOT4A8I-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) #[[ATTR0:[0-9]+]] {
-; ZVDOT4A8I-NEXT: entry:
-; ZVDOT4A8I-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
-; ZVDOT4A8I-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
-; ZVDOT4A8I-NEXT: [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[TMP1]], i64 8)
-; ZVDOT4A8I-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 1024, [[UMAX]]
-; ZVDOT4A8I-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
-; ZVDOT4A8I: vector.ph:
-; ZVDOT4A8I-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 2
-; ZVDOT4A8I-NEXT: [[N_MOD_VF:%.*]] = urem i64 1024, [[TMP3]]
-; ZVDOT4A8I-NEXT: [[N_VEC:%.*]] = sub i64 1024, [[N_MOD_VF]]
-; ZVDOT4A8I-NEXT: br label [[VECTOR_BODY:%.*]]
-; ZVDOT4A8I: vector.body:
-; ZVDOT4A8I-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; ZVDOT4A8I-NEXT: [[VEC_PHI:%.*]] = phi <vscale x 1 x i32> [ zeroinitializer, [[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], [[VECTOR_BODY]] ]
-; ZVDOT4A8I-NEXT: [[TMP6:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX]]
-; ZVDOT4A8I-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i8>, ptr [[TMP6]], align 1
-; ZVDOT4A8I-NEXT: [[TMP9:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX]]
-; ZVDOT4A8I-NEXT: [[WIDE_LOAD1:%.*]] = load <vscale x 4 x i8>, ptr [[TMP9]], align 1
-; ZVDOT4A8I-NEXT: [[TMP11:%.*]] = sext <vscale x 4 x i8> [[WIDE_LOAD1]] to <vscale x 4 x i32>
-; ZVDOT4A8I-NEXT: [[TMP8:%.*]] = sext <vscale x 4 x i8> [[WIDE_LOAD]] to <vscale x 4 x i32>
-; ZVDOT4A8I-NEXT: [[TMP12:%.*]] = mul <vscale x 4 x i32> [[TMP11]], [[TMP8]]
-; ZVDOT4A8I-NEXT: [[PARTIAL_REDUCE]] = call <vscale x 1 x i32> @llvm.vector.partial.reduce.add.nxv1i32.nxv4i32(<vscale x 1 x i32> [[VEC_PHI]], <vscale x 4 x i32> [[TMP12]])
-; ZVDOT4A8I-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
-; ZVDOT4A8I-NEXT: [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; ZVDOT4A8I-NEXT: br i1 [[TMP13]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; ZVDOT4A8I: middle.block:
-; ZVDOT4A8I-NEXT: [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.nxv1i32(<vscale x 1 x i32> [[PARTIAL_REDUCE]])
-; ZVDOT4A8I-NEXT: [[CMP_N:%.*]] = icmp eq i64 1024, [[N_VEC]]
-; ZVDOT4A8I-NEXT: br i1 [[CMP_N]], label [[FOR_EXIT:%.*]], label [[SCALAR_PH]]
-; ZVDOT4A8I: scalar.ph:
-;
-; FIXED-V-LABEL: define i32 @vdot4a(
-; FIXED-V-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) #[[ATTR0:[0-9]+]] {
-; FIXED-V-NEXT: entry:
-; FIXED-V-NEXT: br label [[VECTOR_PH:%.*]]
-; FIXED-V: vector.ph:
-; FIXED-V-NEXT: br label [[VECTOR_BODY:%.*]]
-; FIXED-V: vector.body:
-; FIXED-V-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; FIXED-V-NEXT: [[VEC_PHI:%.*]] = phi <8 x i32> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP12:%.*]], [[VECTOR_BODY]] ]
-; FIXED-V-NEXT: [[VEC_PHI1:%.*]] = phi <8 x i32> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP13:%.*]], [[VECTOR_BODY]] ]
-; FIXED-V-NEXT: [[TMP0:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX]]
-; FIXED-V-NEXT: [[TMP2:%.*]] = getelementptr i8, ptr [[TMP0]], i64 8
-; FIXED-V-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i8>, ptr [[TMP0]], align 1
-; FIXED-V-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x i8>, ptr [[TMP2]], align 1
-; FIXED-V-NEXT: [[TMP3:%.*]] = sext <8 x i8> [[WIDE_LOAD]] to <8 x i32>
-; FIXED-V-NEXT: [[TMP4:%.*]] = sext <8 x i8> [[WIDE_LOAD2]] to <8 x i32>
-; FIXED-V-NEXT: [[TMP5:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX]]
-; FIXED-V-NEXT: [[TMP7:%.*]] = getelementptr i8, ptr [[TMP5]], i64 8
-; FIXED-V-NEXT: [[WIDE_LOAD3:%.*]] = load <8 x i8>, ptr [[TMP5]], align 1
-; FIXED-V-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP7]], a...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/215515
More information about the llvm-commits
mailing list