[llvm] [LV] Use uint64_t for MaxVF and MaxTC in isIndvarOverflowCheck (NFC) (PR #219274)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 12:57:42 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-risc-v
Author: Florian Hahn (fhahn)
<details>
<summary>Changes</summary>
isIndvarOverflowCheckKnownFalse scales both VF and the maximum trip count by the maximum value of vscale to decide whether the canonical IV increment can be marked nuw.
Evaluating them in unsigned max silently wrap if max vscale is very large, like 2^31 and VF = vscale x 4.
Compute both in uint64_t instead, which cannot overflow as the minimum value and the maximum vscale are each 32 bits wide. The bail-out for a trip count that does not fit the IV type is now explicit, as the APInt subtraction would otherwise wrap.
---
Full diff: https://github.com/llvm/llvm-project/pull/219274.diff
2 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorize.cpp (+12-10)
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/indvar-overflow-check-scalable-tc.ll (+53-7)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 1f153afd6bedd..d3263470049ff 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1915,7 +1915,7 @@ static bool isIndvarOverflowCheckKnownFalse(
const LoopVectorizationCostModel *Cost,
ElementCount VF, std::optional<unsigned> UF = std::nullopt) {
// Always be conservative if we don't know the exact unroll factor.
- unsigned MaxUF = UF ? *UF
+ uint64_t MaxUF = UF ? *UF
: std::max(Cost->TTI.getMaxInterleaveFactor(VF, false),
Cost->TTI.getMaxInterleaveFactor(VF, true));
@@ -1929,8 +1929,8 @@ static bool isIndvarOverflowCheckKnownFalse(
Cost->PSE, Cost->TheLoop,
/*CanUseConstantMax=*/true, /*CanExcludeZeroTrips=*/false,
/*ComputeUpperBoundOnly=*/true)) {
- unsigned MaxVF = VF.getKnownMinValue();
- unsigned MaxTC = TC->getKnownMinValue();
+ uint64_t MaxVF = VF.getKnownMinValue();
+ uint64_t MaxTC = TC->getKnownMinValue();
if (VF.isScalable() || TC->isScalable()) {
std::optional<unsigned> MaxVScale =
getMaxVScale(*Cost->TheFunction, Cost->TTI);
@@ -1938,15 +1938,17 @@ static bool isIndvarOverflowCheckKnownFalse(
return false;
if (VF.isScalable())
MaxVF *= *MaxVScale;
- if (TC->isScalable()) {
- bool Overflow;
- MaxTC = SaturatingMultiply(MaxTC, *MaxVScale, &Overflow);
- if (Overflow)
- return false;
- }
+ if (TC->isScalable())
+ MaxTC *= *MaxVScale;
}
- return (MaxUIntTripCount - MaxTC).ugt(MaxVF * MaxUF);
+ // Bail out if the maximum trip count is not representable in the induction
+ // variable's type.
+ if (MaxUIntTripCount.ult(MaxTC))
+ return false;
+
+ uint64_t MaxStep = MaxVF * MaxUF;
+ return (MaxUIntTripCount - MaxTC).ugt(MaxStep);
}
return false;
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/indvar-overflow-check-scalable-tc.ll b/llvm/test/Transforms/LoopVectorize/RISCV/indvar-overflow-check-scalable-tc.ll
index 01767ba3357bf..6ec22b0696fa9 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/indvar-overflow-check-scalable-tc.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/indvar-overflow-check-scalable-tc.ll
@@ -48,8 +48,7 @@ exit:
ret void
}
-; VF is scalable, as is TC. TC * MaxVScale overflows,
-; make sure that the IV increment does not have nuw.
+; VF is scalable, as is TC. TC * MaxVScale overflows in i32, but not the i64 induction.
define void @tc_maxvscale_overflow(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc_maxvscale_overflow(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
@@ -68,13 +67,13 @@ define void @tc_maxvscale_overflow(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds nuw float, ptr [[B]], i64 [[INDEX]]
; CHECK-NEXT: call void @llvm.vp.store.nxv4f32.p0(<vscale x 4 x float> [[VP_OP_LOAD]], ptr align 4 [[TMP3]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP0]])
; CHECK-NEXT: [[TMP5:%.*]] = zext i32 [[TMP0]] to i64
-; CHECK-NEXT: [[CURRENT_ITERATION_NEXT]] = add i64 [[TMP5]], [[INDEX]]
+; CHECK-NEXT: [[CURRENT_ITERATION_NEXT]] = add nuw i64 [[TMP5]], [[INDEX]]
; CHECK-NEXT: [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP5]]
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
-; CHECK-NEXT: br i1 [[TMP4]], label %[[SCALAR_PH:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
-; CHECK: [[SCALAR_PH]]:
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
entry:
@@ -95,3 +94,50 @@ loop:
exit:
ret void
}
+
+; VF is scalable, as is TC. TC * MaxVScale (8388609 * 1024) is not
+; representable in the type of the i32 induction variable.
+define void @tc_maxvscale_exceeds_iv_type(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc_maxvscale_exceeds_iv_type(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[VS:%.*]] = tail call i32 @llvm.vscale.i32()
+; CHECK-NEXT: [[N:%.*]] = mul nuw nsw i32 [[VS]], 8388609
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[AVL:%.*]] = phi i32 [ [[N]], %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = call i32 @llvm.experimental.get.vector.length.i32(i32 [[AVL]], i32 4, i1 true)
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw float, ptr [[A]], i32 [[INDEX]]
+; CHECK-NEXT: [[VP_OP_LOAD:%.*]] = call <vscale x 4 x float> @llvm.vp.load.nxv4f32.p0(ptr align 4 [[TMP1]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP0]])
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds nuw float, ptr [[B]], i32 [[INDEX]]
+; CHECK-NEXT: call void @llvm.vp.store.nxv4f32.p0(<vscale x 4 x float> [[VP_OP_LOAD]], ptr align 4 [[TMP2]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP0]])
+; CHECK-NEXT: [[CURRENT_ITERATION_NEXT]] = add i32 [[TMP0]], [[INDEX]]
+; CHECK-NEXT: [[AVL_NEXT]] = sub nuw i32 [[AVL]], [[TMP0]]
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i32 [[AVL_NEXT]], 0
+; CHECK-NEXT: br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %vs = tail call i32 @llvm.vscale.i32()
+ %n = mul nuw nsw i32 %vs, 8388609
+ br label %loop
+
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds nuw float, ptr %a, i32 %iv
+ %la = load float, ptr %gep.a, align 4
+ %gep.b = getelementptr inbounds nuw float, ptr %b, i32 %iv
+ store float %la, ptr %gep.b, align 4
+ %iv.next = add nuw nsw i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, %n
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
``````````
</details>
https://github.com/llvm/llvm-project/pull/219274
More information about the llvm-commits
mailing list