[llvm] [LV] Avoid low-trip-count vectorization for countable early exits (PR #225635)

Jack Styles via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 23 01:12:52 PDT 2026


https://github.com/Stylie777 created https://github.com/llvm/llvm-project/pull/225635

Following #195823, If a loop has a TC that is equal to (VF * IC) + 1, we are able to vectorize the loop to produce 1 scalar and 1 vector iteration. However, for loops that have an early exit condition, this can lead to miscompiles as it will run a full vector iteration, rather than just the iterations intended.

For loops where an early exit condition is present, and the exiting block of the loop is not in the same block as the loop latch, the new transformation should not be applied.

>From ee45484c191bac85e10a238863bdfa080ce95992 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 23 Sep 2026 08:10:42 +0100
Subject: [PATCH] [LV] Avoid low-trip-count vectorization for countable early
 exits

Following #195823, If a loop has a TC that is equal to (VF * IC) + 1,
we are able to vectorize the loop to produce 1 scalar and 1 vector
iteration. However, for loops that have an early exit condition, this
can lead to miscompiles as it will run a full vector iteration, rather
than just the iterations intended.

For loops where an early exit condition is present, and the exiting
block of the loop is not in the same block as the loop latch, the
new transformation should not be applied.
---
 .../Transforms/Vectorize/LoopVectorize.cpp    |  3 +-
 .../AArch64/sve-low-trip-count-early-exit.ll  | 50 +++++++++++++++++++
 .../RISCV/low-trip-count-early-exit.ll        | 50 +++++++++++++++++++
 3 files changed, 102 insertions(+), 1 deletion(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count-early-exit.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count-early-exit.ll

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 9cec144d57daa5..8ff02129f8f32e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3086,7 +3086,8 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
     unsigned MaxVFForTC = llvm::bit_floor(TC.getFixedValue());
     if (TC.getFixedValue() - MaxVFForTC <= 1 && MaxVFForTC / EffectiveIC > 1 &&
         MaxVFForTC <= (MaxFactors.FixedVF.getFixedValue() * EffectiveIC) &&
-        !Config.OptForSize) {
+        !Config.OptForSize &&
+        TheLoop->getExitingBlock() == TheLoop->getLoopLatch()) {
       unsigned NumOfInstructions = llvm::sum_of(
           llvm::map_range(TheLoop->blocks(),
                           [](BasicBlock *BB) { return BB->size(); }),
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count-early-exit.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count-early-exit.ll
new file mode 100644
index 00000000000000..dbb1f62098fe8b
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count-early-exit.ll
@@ -0,0 +1,50 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve -low-trip-count-loop-body-size-limit=0 < %s | FileCheck %s
+
+define i32 @trip5_early_exit(ptr noalias nocapture noundef %dst, ptr noalias nocapture noundef readonly %src) {
+; CHECK-LABEL: define i32 @trip5_early_exit(
+; CHECK-SAME: ptr noalias noundef captures(none) [[DST:%.*]], ptr noalias noundef readonly captures(none) [[SRC:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[A:%.*]] = icmp eq i64 [[IV]], 3
+; CHECK-NEXT:    br i1 [[A]], label %[[EXIT1:.*]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[GEP_SRC:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP0:%.*]] = load i8, ptr [[GEP_SRC]], align 1
+; CHECK-NEXT:    [[MUL:%.*]] = shl i8 [[TMP0]], 1
+; CHECK-NEXT:    [[GEP_DST:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP1:%.*]] = load i8, ptr [[GEP_DST]], align 1
+; CHECK-NEXT:    [[ADD:%.*]] = add i8 [[MUL]], [[TMP1]]
+; CHECK-NEXT:    store i8 [[ADD]], ptr [[GEP_DST]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT2:.*]], label %[[LOOP]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret i32 1
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret i32 2
+;
+entry:
+  br label %loop
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %a = icmp eq i64 %iv, 3
+  br i1 %a, label %exit1, label %latch
+latch:
+  %gep.src = getelementptr inbounds i8, ptr %src, i64 %iv
+  %0 = load i8, ptr %gep.src, align 1
+  %mul = shl i8 %0, 1
+  %gep.dst = getelementptr inbounds i8, ptr %dst, i64 %iv
+  %1 = load i8, ptr %gep.dst, align 1
+  %add = add i8 %mul, %1
+  store i8 %add, ptr %gep.dst, align 1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, 5
+  br i1 %ec, label %exit2, label %loop
+exit1:
+  ret i32 1
+exit2:
+  ret i32 2
+}
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count-early-exit.ll b/llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count-early-exit.ll
new file mode 100644
index 00000000000000..1168aaed0612f7
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count-early-exit.ll
@@ -0,0 +1,50 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=loop-vectorize -mtriple=riscv64 -mattr=+v -low-trip-count-loop-body-size-limit=0 < %s | FileCheck %s
+
+define i32 @trip5_early_exit(ptr noalias nocapture noundef %dst, ptr noalias nocapture noundef readonly %src) {
+; CHECK-LABEL: define i32 @trip5_early_exit(
+; CHECK-SAME: ptr noalias noundef captures(none) [[DST:%.*]], ptr noalias noundef readonly captures(none) [[SRC:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[A:%.*]] = icmp eq i64 [[IV]], 3
+; CHECK-NEXT:    br i1 [[A]], label %[[EXIT1:.*]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[GEP_SRC:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP0:%.*]] = load i8, ptr [[GEP_SRC]], align 1
+; CHECK-NEXT:    [[MUL:%.*]] = shl i8 [[TMP0]], 1
+; CHECK-NEXT:    [[GEP_DST:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP1:%.*]] = load i8, ptr [[GEP_DST]], align 1
+; CHECK-NEXT:    [[ADD:%.*]] = add i8 [[MUL]], [[TMP1]]
+; CHECK-NEXT:    store i8 [[ADD]], ptr [[GEP_DST]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT2:.*]], label %[[LOOP]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    ret i32 1
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret i32 2
+;
+entry:
+  br label %loop
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %a = icmp eq i64 %iv, 3
+  br i1 %a, label %exit1, label %latch
+latch:
+  %gep.src = getelementptr inbounds i8, ptr %src, i64 %iv
+  %0 = load i8, ptr %gep.src, align 1
+  %mul = shl i8 %0, 1
+  %gep.dst = getelementptr inbounds i8, ptr %dst, i64 %iv
+  %1 = load i8, ptr %gep.dst, align 1
+  %add = add i8 %mul, %1
+  store i8 %add, ptr %gep.dst, align 1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, 5
+  br i1 %ec, label %exit2, label %loop
+exit1:
+  ret i32 1
+exit2:
+  ret i32 2
+}



More information about the llvm-commits mailing list