[llvm] [LV] Add tests showing incorrect handling of large vscale ranges (NFC). (PR #219290)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 1 13:53:19 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/219290
>From 911da58f8fb75a1e66ab06c3dd02e5db907148ed Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 25 Aug 2026 14:38:02 +0100
Subject: [PATCH 1/2] [LV] Add tests for very wide vscale ranges (NFC).
Add tests with very large VScale max (2^30) showing incorrect handling
of wrapping when computing VF * vscale max in computeVF. We incorrectly
determine that no scalar epilogue is needed for the first test function
with a low trip count loop, and vectorize with a scalar epilogue instead
of tail-folding.
---
.../wide-vscale-range-max-runtime-vf.ll | 108 ++++++++++++++++++
1 file changed, 108 insertions(+)
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll b/llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll
new file mode 100644
index 0000000000000..89ca0d5f456b1
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll
@@ -0,0 +1,108 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-unknown-linux-gnu -mattr=+sve -S %s | FileCheck %s
+
+; The maximum runtime VF is 4 * 2^30 = 2^32, which does not divide the trip
+; count of 8, so the tail needs folding.
+; FIXME: This is a low trip count loop, which should be vectorized with
+; tail-folding.
+define void @wide_vscale_range_needs_tail_folding(ptr noalias %A) #0 {
+; CHECK-LABEL: define void @wide_vscale_range_needs_tail_folding(
+; CHECK-SAME: ptr noalias [[A:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 8, [[TMP1]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 8, [[TMP1]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 8, [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add <vscale x 4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <vscale x 4 x i32> [[TMP3]], ptr [[TMP2]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 8, [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[L:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add i32 [[L]], 1
+; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT: [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 8
+; CHECK-NEXT: br i1 [[EC]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep = getelementptr inbounds i32, ptr %A, i64 %iv
+ %l = load i32, ptr %gep, align 4
+ %add = add i32 %l, 1
+ store i32 %add, ptr %gep, align 4
+ %iv.next = add i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 8
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+; Same maximum runtime VF of 2^32, but here it divides the trip count of 2^32,
+; so no tail remains for any VF that may be chosen and the tail does not need
+; folding.
+define void @wide_vscale_range_no_tail_for_any_vf(ptr noalias %A) #1 {
+; CHECK-LABEL: define void @wide_vscale_range_no_tail_for_any_vf(
+; CHECK-SAME: ptr noalias [[A:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add <vscale x 4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <vscale x 4 x i32> [[TMP3]], ptr [[TMP2]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4294967296
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep = getelementptr inbounds i32, ptr %A, i64 %iv
+ %l = load i32, ptr %gep, align 4
+ %add = add i32 %l, 1
+ store i32 %add, ptr %gep, align 4
+ %iv.next = add i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 4294967296
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+attributes #0 = { vscale_range(1,1073741824) }
+attributes #1 = { optsize vscale_range(1,1073741824) }
>From 168c31257e9dc07ffdc3f1f4d51b4ee746cc946b Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 1 Sep 2026 21:52:52 +0100
Subject: [PATCH 2/2] !fixup reword comment, thanks
---
.../LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll b/llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll
index 89ca0d5f456b1..8d9cbe1347286 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll
@@ -1,8 +1,8 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
; RUN: opt -passes=loop-vectorize -mtriple=aarch64-unknown-linux-gnu -mattr=+sve -S %s | FileCheck %s
-; The maximum runtime VF is 4 * 2^30 = 2^32, which does not divide the trip
-; count of 8, so the tail needs folding.
+; The maximum runtime VF is 4 * 2^30 = 2^32. The trip count (8) is not
+; guaranteed to be a multiple of the VF for all values of vscale.
; FIXME: This is a low trip count loop, which should be vectorized with
; tail-folding.
define void @wide_vscale_range_needs_tail_folding(ptr noalias %A) #0 {
More information about the llvm-commits
mailing list