[llvm] [LV] Support tail-folded epilogue loops (PR #208764)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 29 05:53:32 PDT 2026
================
@@ -0,0 +1,671 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; REQUIRES: asserts
+
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -mattr=+sve \
+; RUN: -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width=16 \
+; RUN: -epilogue-vectorization-force-VF=8 %s | FileCheck %s
+
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -mattr=+sve \
+; RUN: -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width="vscale x 16" \
+; RUN: -epilogue-vectorization-force-VF="vscale x 8" %s | FileCheck %s \
+; RUN: --check-prefix=CHECK-VS
+
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize \
+; RUN: -epilogue-tail-folding-policy=prefer-fold-tail --disable-output \
+; RUN: -force-vector-width=4 -epilogue-vectorization-force-VF="vscale x 2" \
+; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s \
+; RUN: --check-prefix=CHECK-INVALID-COSTS
+
+target triple = "aarch64-linux-gnu"
+
+define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
+; CHECK-LABEL: define void @test_epilogue_tf(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ITER_CHECK:.*:]]
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
----------------
fhahn wrote:
I think the minimum iteration count also guards against the case where the trip count computation wraps around to zero. Thinking of something like below. Assume `%n == 0`, I think then we make create an active-lane-mask for the epilogue that is all-zero.
```
define void @i32_iv_tc_wrap(ptr noalias %a, i32 %n) {
entry:
br label %loop
loop:
%iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
%iv.ext = zext i32 %iv to i64
%gep = getelementptr i8, ptr %a, i64 %iv.ext
store i8 0, ptr %gep, align 1
%iv.next = add i32 %iv, 1
%ec = icmp eq i32 %iv.next, %n
br i1 %ec, label %exit, label %loop
exit:
ret void
}
```
https://github.com/llvm/llvm-project/pull/208764
More information about the llvm-commits
mailing list