[llvm] [LV] Avoid low-trip-count vectorization for countable early exits (PR #225635)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 23 01:13:36 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms

@llvm/pr-subscribers-backend-risc-v

Author: Jack Styles (Stylie777)

<details>
<summary>Changes</summary>

Following #<!-- -->195823, If a loop has a TC that is equal to (VF * IC) + 1, we are able to vectorize the loop to produce 1 scalar and 1 vector iteration. However, for loops that have an early exit condition, this can lead to miscompiles as it will run a full vector iteration, rather than just the iterations intended.

For loops where an early exit condition is present, and the exiting block of the loop is not in the same block as the loop latch, the new transformation should not be applied.

---
Full diff: https://github.com/llvm/llvm-project/pull/225635.diff


3 Files Affected:

- (modified) llvm/lib/Transforms/Vectorize/LoopVectorize.cpp (+2-1) 
- (added) llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count-early-exit.ll (+50) 
- (added) llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count-early-exit.ll (+50) 


``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 9cec144d57daa..8ff02129f8f32 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3086,7 +3086,8 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
     unsigned MaxVFForTC = llvm::bit_floor(TC.getFixedValue());
     if (TC.getFixedValue() - MaxVFForTC <= 1 && MaxVFForTC / EffectiveIC > 1 &&
         MaxVFForTC <= (MaxFactors.FixedVF.getFixedValue() * EffectiveIC) &&
-        !Config.OptForSize) {
+        !Config.OptForSize &&
+        TheLoop->getExitingBlock() == TheLoop->getLoopLatch()) {
       unsigned NumOfInstructions = llvm::sum_of(
           llvm::map_range(TheLoop->blocks(),
                           [](BasicBlock *BB) { return BB->size(); }),
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count-early-exit.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count-early-exit.ll
new file mode 100644
index 0000000000000..dbb1f62098fe8
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count-early-exit.ll
@@ -0,0 +1,50 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve -low-trip-count-loop-body-size-limit=0 < %s | FileCheck %s
+
+define i32 @trip5_early_exit(ptr noalias nocapture noundef %dst, ptr noalias nocapture noundef readonly %src) {
+; CHECK-LABEL: define i32 @trip5_early_exit(
+; CHECK-SAME: ptr noalias noundef captures(none) [[DST:%.*]], ptr noalias noundef readonly captures(none) [[SRC:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[A:%.*]] = icmp eq i64 [[IV]], 3
+; CHECK-NEXT:    br i1 [[A]], label %[[EXIT1:.*]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[GEP_SRC:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP0:%.*]] = load i8, ptr [[GEP_SRC]], align 1
+; CHECK-NEXT:    [[MUL:%.*]] = shl i8 [[TMP0]], 1
+; CHECK-NEXT:    [[GEP_DST:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP1:%.*]] = load i8, ptr [[GEP_DST]], align 1
+; CHECK-NEXT:    [[ADD:%.*]] = add i8 [[MUL]], [[TMP1]]
+; CHECK-NEXT:    store i8 [[ADD]], ptr [[GEP_DST]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT2:.*]], label %[[LOOP]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret i32 1
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret i32 2
+;
+entry:
+  br label %loop
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %a = icmp eq i64 %iv, 3
+  br i1 %a, label %exit1, label %latch
+latch:
+  %gep.src = getelementptr inbounds i8, ptr %src, i64 %iv
+  %0 = load i8, ptr %gep.src, align 1
+  %mul = shl i8 %0, 1
+  %gep.dst = getelementptr inbounds i8, ptr %dst, i64 %iv
+  %1 = load i8, ptr %gep.dst, align 1
+  %add = add i8 %mul, %1
+  store i8 %add, ptr %gep.dst, align 1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, 5
+  br i1 %ec, label %exit2, label %loop
+exit1:
+  ret i32 1
+exit2:
+  ret i32 2
+}
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count-early-exit.ll b/llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count-early-exit.ll
new file mode 100644
index 0000000000000..1168aaed0612f
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count-early-exit.ll
@@ -0,0 +1,50 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=riscv64 -mattr=+v -low-trip-count-loop-body-size-limit=0 < %s | FileCheck %s
+
+define i32 @trip5_early_exit(ptr noalias nocapture noundef %dst, ptr noalias nocapture noundef readonly %src) {
+; CHECK-LABEL: define i32 @trip5_early_exit(
+; CHECK-SAME: ptr noalias noundef captures(none) [[DST:%.*]], ptr noalias noundef readonly captures(none) [[SRC:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[A:%.*]] = icmp eq i64 [[IV]], 3
+; CHECK-NEXT:    br i1 [[A]], label %[[EXIT1:.*]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[GEP_SRC:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP0:%.*]] = load i8, ptr [[GEP_SRC]], align 1
+; CHECK-NEXT:    [[MUL:%.*]] = shl i8 [[TMP0]], 1
+; CHECK-NEXT:    [[GEP_DST:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP1:%.*]] = load i8, ptr [[GEP_DST]], align 1
+; CHECK-NEXT:    [[ADD:%.*]] = add i8 [[MUL]], [[TMP1]]
+; CHECK-NEXT:    store i8 [[ADD]], ptr [[GEP_DST]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT2:.*]], label %[[LOOP]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret i32 1
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret i32 2
+;
+entry:
+  br label %loop
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %a = icmp eq i64 %iv, 3
+  br i1 %a, label %exit1, label %latch
+latch:
+  %gep.src = getelementptr inbounds i8, ptr %src, i64 %iv
+  %0 = load i8, ptr %gep.src, align 1
+  %mul = shl i8 %0, 1
+  %gep.dst = getelementptr inbounds i8, ptr %dst, i64 %iv
+  %1 = load i8, ptr %gep.dst, align 1
+  %add = add i8 %mul, %1
+  store i8 %add, ptr %gep.dst, align 1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, 5
+  br i1 %ec, label %exit2, label %loop
+exit1:
+  ret i32 1
+exit2:
+  ret i32 2
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/225635


More information about the llvm-commits mailing list