[llvm] 6ee82d6 - [LV] Use uint64_t for MaxVF and MaxTC in isIndvarOverflowCheck. (#219274)

via llvm-commits llvm-commits at lists.llvm.org
Fri Aug 28 08:01:17 PDT 2026


Author: Florian Hahn
Date: 2026-08-28T15:01:11Z
New Revision: 6ee82d6031d4ec563ec203faa4435ac20e99eaf2

URL: https://github.com/llvm/llvm-project/commit/6ee82d6031d4ec563ec203faa4435ac20e99eaf2
DIFF: https://github.com/llvm/llvm-project/commit/6ee82d6031d4ec563ec203faa4435ac20e99eaf2.diff

LOG: [LV] Use uint64_t for MaxVF and MaxTC in isIndvarOverflowCheck. (#219274)

isIndvarOverflowCheckKnownFalse scales both VF and the maximum trip
count by the maximum value of vscale to decide whether the canonical IV
increment can be marked nuw.

Evaluating them in unsigned max silently wrap if max vscale is very
large, like 2^31 and VF = vscale x 4.

Compute both in uint64_t instead, which cannot overflow as the minimum
value and the maximum vscale are each 32 bits wide. The bail-out for a
trip count that does not fit the IV type is now explicit, as the APInt
subtraction would otherwise wrap.

PR: https://github.com/llvm/llvm-project/pull/219274

Added: 
    

Modified: 
    llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
    llvm/test/Transforms/LoopVectorize/RISCV/indvar-overflow-check-scalable-tc.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 1f153afd6bedd..d3263470049ff 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1915,7 +1915,7 @@ static bool isIndvarOverflowCheckKnownFalse(
     const LoopVectorizationCostModel *Cost,
     ElementCount VF, std::optional<unsigned> UF = std::nullopt) {
   // Always be conservative if we don't know the exact unroll factor.
-  unsigned MaxUF = UF ? *UF
+  uint64_t MaxUF = UF ? *UF
                       : std::max(Cost->TTI.getMaxInterleaveFactor(VF, false),
                                  Cost->TTI.getMaxInterleaveFactor(VF, true));
 
@@ -1929,8 +1929,8 @@ static bool isIndvarOverflowCheckKnownFalse(
           Cost->PSE, Cost->TheLoop,
           /*CanUseConstantMax=*/true, /*CanExcludeZeroTrips=*/false,
           /*ComputeUpperBoundOnly=*/true)) {
-    unsigned MaxVF = VF.getKnownMinValue();
-    unsigned MaxTC = TC->getKnownMinValue();
+    uint64_t MaxVF = VF.getKnownMinValue();
+    uint64_t MaxTC = TC->getKnownMinValue();
     if (VF.isScalable() || TC->isScalable()) {
       std::optional<unsigned> MaxVScale =
           getMaxVScale(*Cost->TheFunction, Cost->TTI);
@@ -1938,15 +1938,17 @@ static bool isIndvarOverflowCheckKnownFalse(
         return false;
       if (VF.isScalable())
         MaxVF *= *MaxVScale;
-      if (TC->isScalable()) {
-        bool Overflow;
-        MaxTC = SaturatingMultiply(MaxTC, *MaxVScale, &Overflow);
-        if (Overflow)
-          return false;
-      }
+      if (TC->isScalable())
+        MaxTC *= *MaxVScale;
     }
 
-    return (MaxUIntTripCount - MaxTC).ugt(MaxVF * MaxUF);
+    // Bail out if the maximum trip count is not representable in the induction
+    // variable's type.
+    if (MaxUIntTripCount.ult(MaxTC))
+      return false;
+
+    uint64_t MaxStep = MaxVF * MaxUF;
+    return (MaxUIntTripCount - MaxTC).ugt(MaxStep);
   }
 
   return false;

diff  --git a/llvm/test/Transforms/LoopVectorize/RISCV/indvar-overflow-check-scalable-tc.ll b/llvm/test/Transforms/LoopVectorize/RISCV/indvar-overflow-check-scalable-tc.ll
index 01767ba3357bf..6ec22b0696fa9 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/indvar-overflow-check-scalable-tc.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/indvar-overflow-check-scalable-tc.ll
@@ -48,8 +48,7 @@ exit:
   ret void
 }
 
-; VF is scalable, as is TC. TC * MaxVScale overflows,
-; make sure that the IV increment does not have nuw.
+; VF is scalable, as is TC. TC * MaxVScale overflows in i32, but not the i64 induction.
 define void @tc_maxvscale_overflow(ptr noalias %a, ptr noalias %b) #0 {
 ; CHECK-LABEL: define void @tc_maxvscale_overflow(
 ; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
@@ -68,13 +67,13 @@ define void @tc_maxvscale_overflow(ptr noalias %a, ptr noalias %b) #0 {
 ; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw float, ptr [[B]], i64 [[INDEX]]
 ; CHECK-NEXT:    call void @llvm.vp.store.nxv4f32.p0(<vscale x 4 x float> [[VP_OP_LOAD]], ptr align 4 [[TMP3]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP0]])
 ; CHECK-NEXT:    [[TMP5:%.*]] = zext i32 [[TMP0]] to i64
-; CHECK-NEXT:    [[CURRENT_ITERATION_NEXT]] = add i64 [[TMP5]], [[INDEX]]
+; CHECK-NEXT:    [[CURRENT_ITERATION_NEXT]] = add nuw i64 [[TMP5]], [[INDEX]]
 ; CHECK-NEXT:    [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP5]]
 ; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
-; CHECK-NEXT:    br i1 [[TMP4]], label %[[SCALAR_PH:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
-; CHECK:       [[SCALAR_PH]]:
-; CHECK-NEXT:    br label %[[LOOP:.*]]
-; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
 entry:
@@ -95,3 +94,50 @@ loop:
 exit:
   ret void
 }
+
+; VF is scalable, as is TC. TC * MaxVScale (8388609 * 1024) is not
+; representable in the type of the i32 induction variable.
+define void @tc_maxvscale_exceeds_iv_type(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc_maxvscale_exceeds_iv_type(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[VS:%.*]] = tail call i32 @llvm.vscale.i32()
+; CHECK-NEXT:    [[N:%.*]] = mul nuw nsw i32 [[VS]], 8388609
+; CHECK-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[AVL:%.*]] = phi i32 [ [[N]], %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = call i32 @llvm.experimental.get.vector.length.i32(i32 [[AVL]], i32 4, i1 true)
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw float, ptr [[A]], i32 [[INDEX]]
+; CHECK-NEXT:    [[VP_OP_LOAD:%.*]] = call <vscale x 4 x float> @llvm.vp.load.nxv4f32.p0(ptr align 4 [[TMP1]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP0]])
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw float, ptr [[B]], i32 [[INDEX]]
+; CHECK-NEXT:    call void @llvm.vp.store.nxv4f32.p0(<vscale x 4 x float> [[VP_OP_LOAD]], ptr align 4 [[TMP2]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP0]])
+; CHECK-NEXT:    [[CURRENT_ITERATION_NEXT]] = add i32 [[TMP0]], [[INDEX]]
+; CHECK-NEXT:    [[AVL_NEXT]] = sub nuw i32 [[AVL]], [[TMP0]]
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i32 [[AVL_NEXT]], 0
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %vs = tail call i32 @llvm.vscale.i32()
+  %n = mul nuw nsw i32 %vs, 8388609
+  br label %loop
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %gep.a = getelementptr inbounds nuw float, ptr %a, i32 %iv
+  %la = load float, ptr %gep.a, align 4
+  %gep.b = getelementptr inbounds nuw float, ptr %b, i32 %iv
+  store float %la, ptr %gep.b, align 4
+  %iv.next = add nuw nsw i32 %iv, 1
+  %ec = icmp eq i32 %iv.next, %n
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  ret void
+}


        


More information about the llvm-commits mailing list