[llvm] [LV] Use getMaxRuntimeElementCount for MaxPowerOf2RuntimeVF. (PR #221219)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 7 06:23:21 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/221219
>From 26daa049d2d026cce08775845f7674ee3ac1f8f1 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 25 Aug 2026 15:22:47 +0100
Subject: [PATCH 1/2] [LV] Use getMaxRuntimeElementCount for
MaxPowerOf2RuntimeVF.
Use new getMaxRuntimeElementCount to compaute MaxPowerOf2RuntimeVF to
avoid overflow by performing computations in uint64_t.
This fixes an overflow in the test added in #219290. We now correctly
determine that a tail is needed for a low trip count loop.
---
.../Transforms/Vectorize/LoopVectorize.cpp | 19 +++++-----
.../wide-vscale-range-max-runtime-vf.ll | 38 ++++++-------------
2 files changed, 21 insertions(+), 36 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 1b52f05cd93e6..84de4bb6c5b80 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -2991,25 +2991,24 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// Avoid tail folding if the trip count is known to be a multiple of any VF
// we choose.
- std::optional<unsigned> MaxPowerOf2RuntimeVF =
+ std::optional<uint64_t> MaxPowerOf2RuntimeVF =
MaxFactors.FixedVF.getFixedValue();
if (MaxFactors.ScalableVF) {
- std::optional<unsigned> MaxVScale = getMaxVScale(*TheFunction, TTI);
- if (MaxVScale) {
- MaxPowerOf2RuntimeVF = std::max<unsigned>(
- *MaxPowerOf2RuntimeVF,
- *MaxVScale * MaxFactors.ScalableVF.getKnownMinValue());
- } else
+ if (std::optional<uint64_t> MaxRuntimeScalableVF =
+ getMaxRuntimeElementCount(MaxFactors.ScalableVF, *TheFunction, TTI))
+ MaxPowerOf2RuntimeVF =
+ std::max(*MaxPowerOf2RuntimeVF, *MaxRuntimeScalableVF);
+ else
MaxPowerOf2RuntimeVF = std::nullopt; // Stick with tail-folding for now.
}
- auto NoScalarEpilogueNeeded = [this, &UserIC](unsigned MaxVF) {
+ auto NoScalarEpilogueNeeded = [this, &UserIC](uint64_t MaxRuntimeVF) {
// Return false if the loop is neither a single-latch-exit loop nor an
// early-exit loop as tail-folding is not supported in that case.
if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
!Legal->hasUncountableEarlyExit())
return false;
- unsigned MaxVFtimesIC = UserIC ? MaxVF * UserIC : MaxVF;
+ uint64_t MaxVFtimesIC = MaxRuntimeVF * uint64_t(std::max(UserIC, 1u));
ScalarEvolution *SE = PSE.getSE();
// Calling getSymbolicMaxBackedgeTakenCount enables support for loops
// with uncountable exits. For countable loops, the symbolic maximum must
@@ -3027,7 +3026,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
};
if (MaxPowerOf2RuntimeVF > 0u) {
- assert((UserVF.isNonZero() || isPowerOf2_32(*MaxPowerOf2RuntimeVF)) &&
+ assert((UserVF.isNonZero() || isPowerOf2_64(*MaxPowerOf2RuntimeVF)) &&
"MaxFixedVF must be a power of 2");
if (NoScalarEpilogueNeeded(*MaxPowerOf2RuntimeVF)) {
// Accept MaxFixedVF if we do not have a tail.
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll b/llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll
index 8d9cbe1347286..29dc9400e775d 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/wide-vscale-range-max-runtime-vf.ll
@@ -3,45 +3,31 @@
; The maximum runtime VF is 4 * 2^30 = 2^32. The trip count (8) is not
; guaranteed to be a multiple of the VF for all values of vscale.
-; FIXME: This is a low trip count loop, which should be vectorized with
-; tail-folding.
define void @wide_vscale_range_needs_tail_folding(ptr noalias %A) #0 {
; CHECK-LABEL: define void @wide_vscale_range_needs_tail_folding(
; CHECK-SAME: ptr noalias [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 8, [[TMP1]]
-; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
-; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 8, [[TMP1]]
-; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 8, [[N_MOD_VF]]
+; CHECK-NEXT: [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 4 x i1> @llvm.get.active.lane.mask.nxv4i1.i64(i64 0, i64 8)
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 4 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VECTOR_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = call <vscale x 4 x i32> @llvm.masked.load.nxv4i32.p0(ptr align 4 [[TMP2]], <vscale x 4 x i1> [[ACTIVE_LANE_MASK]], <vscale x 4 x i32> poison)
; CHECK-NEXT: [[TMP3:%.*]] = add <vscale x 4 x i32> [[WIDE_LOAD]], splat (i32 1)
-; CHECK-NEXT: store <vscale x 4 x i32> [[TMP3]], ptr [[TMP2]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
-; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 8, [[N_VEC]]
-; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK-NEXT: call void @llvm.masked.store.nxv4i32.p0(<vscale x 4 x i32> [[TMP3]], ptr align 4 [[TMP2]], <vscale x 4 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add i64 [[INDEX]], [[TMP1]]
+; CHECK-NEXT: [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 4 x i1> @llvm.get.active.lane.mask.nxv4i1.i64(i64 [[INDEX_NEXT]], i64 8)
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <vscale x 4 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT: [[TMP5:%.*]] = xor i1 [[TMP4]], true
+; CHECK-NEXT: br i1 [[TMP5]], label %[[SCALAR_PH:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[SCALAR_PH]]:
-; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT: [[L:%.*]] = load i32, ptr [[GEP]], align 4
-; CHECK-NEXT: [[ADD:%.*]] = add i32 [[L]], 1
-; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP]], align 4
-; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
-; CHECK-NEXT: [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 8
-; CHECK-NEXT: br i1 [[EC]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
-; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
entry:
>From 1e9a49976734d434102e965b756fe8ccec2f7a29 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Mon, 7 Sep 2026 14:22:36 +0100
Subject: [PATCH 2/2] !fixup adjust std::max, fix build failure after merge
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index c569eea46f573..0cfea4765d0ca 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -2995,7 +2995,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
MaxFactors.FixedVF.getFixedValue();
if (MaxFactors.ScalableVF) {
if (std::optional<uint64_t> MaxRuntimeScalableVF =
- getMaxRuntimeElementCount(MaxFactors.ScalableVF, *TheFunction, TTI))
+ getMaxRuntimeElementCount(MaxFactors.ScalableVF, *TheFunction))
MaxPowerOf2RuntimeVF =
std::max(*MaxPowerOf2RuntimeVF, *MaxRuntimeScalableVF);
else
@@ -3008,7 +3008,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
!Legal->hasUncountableEarlyExit())
return false;
- uint64_t MaxVFtimesIC = MaxRuntimeVF * uint64_t(std::max(UserIC, 1u));
+ uint64_t MaxVFtimesIC = MaxRuntimeVF * std::max<uint64_t>(UserIC, 1);
ScalarEvolution *SE = PSE.getSE();
// Calling getSymbolicMaxBackedgeTakenCount enables support for loops
// with uncountable exits. For countable loops, the symbolic maximum must
More information about the llvm-commits
mailing list