[llvm] [LoopVectorize] Don't speculate an early-exit trip count that may cause UB (PR #219893)

Madhur Amilkanthwar via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 7 02:37:20 PDT 2026


https://github.com/madhur13490 updated https://github.com/llvm/llvm-project/pull/219893

>From 0aab51b9e22e3dfa659b88dc4aa36e70bd72c5c2 Mon Sep 17 00:00:00 2001
From: Madhur Amilkanthwar <madhura at nvidia.com>
Date: Fri, 28 Aug 2026 20:54:44 +0530
Subject: [PATCH 1/2] [LoopVectorize] Don't speculate an early-exit trip count
 that may cause UB

For a loop with an uncountable early exit, the trip count is derived from
the latch exit and expanded in the preheader, above the early exit. If it
contains a udiv whose divisor may be zero or poison, that speculates UB.

Bail out in this case, reusing ScalarEvolution::isGuaranteedNotToCauseUB
(made public here). Loops whose divisor is known non-zero and not poison
are still vectorized.

Fixes https://github.com/llvm/llvm-project/issues/219371.
---
 llvm/include/llvm/Analysis/ScalarEvolution.h  |  8 ++--
 .../Vectorize/LoopVectorizationLegality.cpp   | 21 ++++++++-
 .../early-exit-trip-count-may-cause-ub.ll     | 46 ++++---------------
 3 files changed, 33 insertions(+), 42 deletions(-)

diff --git a/llvm/include/llvm/Analysis/ScalarEvolution.h b/llvm/include/llvm/Analysis/ScalarEvolution.h
index 385a8e3f0d3bb..594fc3c478a76 100644
--- a/llvm/include/llvm/Analysis/ScalarEvolution.h
+++ b/llvm/include/llvm/Analysis/ScalarEvolution.h
@@ -1248,6 +1248,11 @@ class ScalarEvolution {
   /// Returns true if \p Op is guaranteed to not be poison.
   LLVM_ABI static bool isGuaranteedNotToBePoison(const SCEV *Op);
 
+  /// Returns true if \p Op is guaranteed not to cause immediate UB. Use this
+  /// to check whether \p Op can be expanded at a point it may not have been
+  /// evaluated at in the original program.
+  LLVM_ABI bool isGuaranteedNotToCauseUB(const SCEV *Op);
+
   /// Test if the given expression is known to be a power of 2.  OrNegative
   /// allows matching negative power of 2s, and OrZero allows matching 0.
   LLVM_ABI bool isKnownToBeAPowerOfTwo(const SCEV *S, bool OrZero = false,
@@ -2454,9 +2459,6 @@ class ScalarEvolution {
   bool isGuaranteedToTransferExecutionTo(const Instruction *A,
                                          const Instruction *B);
 
-  /// Returns true if \p Op is guaranteed not to cause immediate UB.
-  bool isGuaranteedNotToCauseUB(const SCEV *Op);
-
   /// Return true if the SCEV corresponding to \p I is never poison.  Proving
   /// this is more complex than proving that just \p I is never poison, since
   /// SCEV commons expressions across control flow, and you can have cases
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index f1d785b571367..f8ba9cb520cbe 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -1787,12 +1787,29 @@ bool LoopVectorizationLegality::isVectorizableEarlyExitLoop() {
     }
   }
 
-  [[maybe_unused]] const SCEV *SymbolicMaxBTC =
-      PSE.getSymbolicMaxBackedgeTakenCount();
+  const SCEV *SymbolicMaxBTC = PSE.getSymbolicMaxBackedgeTakenCount();
   // Since we have an exact exit count for the latch and the early exit
   // dominates the latch, then this should guarantee a computed SCEV value.
   assert(!isa<SCEVCouldNotCompute>(SymbolicMaxBTC) &&
          "Failed to get symbolic expression for backedge taken count");
+
+  // The symbolic max backedge-taken count is built from the countable exits
+  // only, which the original loop reaches only if no uncountable exit is taken
+  // first. Expanding it in the preheader therefore speculates it above those
+  // uncountable exits, so it must not cause UB on its own. Note that the
+  // sequential umin ScalarEvolution uses to keep the exit counts of multiple
+  // countable exits from short-circuiting does not help here, as uncountable
+  // exits contribute no operand to it.
+  // TODO: Rather than giving up, expand a divisor that may be zero or poison
+  // as freeze + umax(_, 1), like SCEVExpander does for the operands of a
+  // sequential umin.
+  if (!PSE.getSE()->isGuaranteedNotToCauseUB(SymbolicMaxBTC)) {
+    reportVectorizationFailure(
+        "Backedge taken count may cause UB when expanded into the preheader",
+        "PotentiallyUBBackedgeTakenCountEarlyExitLoop", ORE, TheLoop);
+    return false;
+  }
+
   LLVM_DEBUG(dbgs() << "LV: Found an early exit loop with symbolic max "
                        "backedge taken count: "
                     << *SymbolicMaxBTC << '\n');
diff --git a/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll b/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll
index 2f9a37918c65b..d521d766cd017 100644
--- a/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll
+++ b/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll
@@ -10,53 +10,25 @@ declare i32 @llvm.cttz.i32(i32, i1 immarg)
 ; well defined on that path must not be computed unconditionally in the
 ; preheader.
 
-; FIXME: %ct is poison when %x is 0, so %d may be poison. The udiv is currently
-; speculated into the preheader, where it can divide by poison, which is UB. The
-; original loop returns 0 without evaluating %d when %skip is true, so this is a
-; miscompile and the loop must not be vectorized.
+; %ct is poison when %x is 0, so %d may be poison and speculating (63 /u %d)
+; would divide by poison, which is UB. The original loop returns 0 without ever
+; evaluating %d when %skip is true, so do not vectorize.
 define i32 @udiv_by_poison_step(i32 %x, i1 %skip) {
 ; CHECK-LABEL: define i32 @udiv_by_poison_step(
 ; CHECK-SAME: i32 [[X:%.*]], i1 [[SKIP:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
 ; CHECK-NEXT:    [[CT:%.*]] = call i32 @llvm.cttz.i32(i32 [[X]], i1 true)
 ; CHECK-NEXT:    [[D:%.*]] = add nuw nsw i32 [[CT]], 1
-; CHECK-NEXT:    [[TMP0:%.*]] = udiv i32 63, [[D]]
-; CHECK-NEXT:    [[TMP1:%.*]] = add nuw nsw i32 [[TMP0]], 1
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP1]], 4
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP2:%.*]] = and i32 [[TMP1]], 3
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i1> poison, i1 [[SKIP]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP3:%.*]] = mul i32 [[N_VEC]], [[D]]
-; CHECK-NEXT:    [[TMP4:%.*]] = freeze <4 x i1> [[BROADCAST_SPLAT]]
-; CHECK-NEXT:    [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP4]])
-; CHECK-NEXT:    br label %[[LOOP:.*]]
-; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP5]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
-; CHECK:       [[VECTOR_BODY_INTERIM]]:
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP1]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
-; CHECK:       [[VECTOR_EARLY_EXIT]]:
-; CHECK-NEXT:    br label %[[EXIT]]
-; CHECK:       [[SCALAR_PH]]:
-; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i32 [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
 ; CHECK-NEXT:    br label %[[LOOP1:.*]]
 ; CHECK:       [[LOOP1]]:
-; CHECK-NEXT:    [[I:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[INC:%.*]], %[[LATCH:.*]] ]
-; CHECK-NEXT:    br i1 [[SKIP]], label %[[EXIT]], label %[[LATCH]]
+; CHECK-NEXT:    [[I:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[INC:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    br i1 [[SKIP]], label %[[EXIT:.*]], label %[[LATCH]]
 ; CHECK:       [[LATCH]]:
 ; CHECK-NEXT:    [[INC]] = add nuw nsw i32 [[I]], [[D]]
 ; CHECK-NEXT:    [[DONE:%.*]] = icmp uge i32 [[INC]], 64
-; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP0:![0-9]+]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[R:%.*]] = phi i32 [ 0, %[[LOOP1]] ], [ [[INC]], %[[LATCH]] ], [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[VECTOR_EARLY_EXIT]] ]
+; CHECK-NEXT:    [[R:%.*]] = phi i32 [ 0, %[[LOOP1]] ], [ [[INC]], %[[LATCH]] ]
 ; CHECK-NEXT:    ret i32 [[R]]
 ;
 entry:
@@ -106,7 +78,7 @@ define i32 @udiv_by_nonzero_nonpoison_step(i32 noundef %x, i1 %skip) {
 ; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP5]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
 ; CHECK:       [[VECTOR_BODY_INTERIM]]:
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP1]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
@@ -121,7 +93,7 @@ define i32 @udiv_by_nonzero_nonpoison_step(i32 noundef %x, i1 %skip) {
 ; CHECK:       [[LATCH]]:
 ; CHECK-NEXT:    [[INC]] = add nuw nsw i32 [[I]], [[D]]
 ; CHECK-NEXT:    [[DONE:%.*]] = icmp uge i32 [[INC]], 64
-; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP6:![0-9]+]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    [[R:%.*]] = phi i32 [ 0, %[[LOOP]] ], [ [[INC]], %[[LATCH]] ], [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[VECTOR_EARLY_EXIT]] ]
 ; CHECK-NEXT:    ret i32 [[R]]

>From c092d63b40e9ceac6bda8bafec864638aa81718d Mon Sep 17 00:00:00 2001
From: Madhur Amilkanthwar <madhura at nvidia.com>
Date: Mon, 7 Sep 2026 15:06:38 +0530
Subject: [PATCH 2/2] fixup! [LoopVectorize] Don't speculate an early-exit trip
 count that may cause UB

---
 .../early-exit-trip-count-may-cause-ub.ll     | 138 ++++++++++++++++++
 1 file changed, 138 insertions(+)

diff --git a/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll b/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll
index d521d766cd017..bc9fe0e633a94 100644
--- a/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll
+++ b/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll
@@ -2,6 +2,7 @@
 ; RUN: opt -p loop-vectorize -force-vector-width=4 -force-vector-interleave=1 -S %s | FileCheck %s
 
 declare i32 @llvm.cttz.i32(i32, i1 immarg)
+declare void @init_mem(ptr, i64)
 
 ; These loops have two exits: an early exit and the latch. The vectorizer
 ; computes the trip count from the latch and emits that computation in the
@@ -117,6 +118,143 @@ exit:
   ret i32 %r
 }
 
+; Same reasoning, but with a loop-varying early exit: the loop leaves early
+; when a loaded byte is zero. The trip count is still taken from the latch, so
+; %d being poison must still prevent vectorization.
+define i32 @udiv_by_poison_step_early_exit_load(i32 %x) {
+; CHECK-LABEL: define i32 @udiv_by_poison_step_early_exit_load(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[P:%.*]] = alloca [1024 x i8], align 1
+; CHECK-NEXT:    call void @init_mem(ptr [[P]], i64 1024)
+; CHECK-NEXT:    [[CT:%.*]] = call i32 @llvm.cttz.i32(i32 [[X]], i1 true)
+; CHECK-NEXT:    [[D:%.*]] = add nuw nsw i32 [[CT]], 1
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[INC:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[J:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[JNEXT:%.*]], %[[LATCH]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[P]], i32 [[J]]
+; CHECK-NEXT:    [[LD:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-NEXT:    [[FOUND:%.*]] = icmp eq i8 [[LD]], 0
+; CHECK-NEXT:    br i1 [[FOUND]], label %[[EXIT:.*]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[INC]] = add nuw nsw i32 [[I]], [[D]]
+; CHECK-NEXT:    [[JNEXT]] = add nuw nsw i32 [[J]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp uge i32 [[INC]], 64
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP0]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[R:%.*]] = phi i32 [ 0, %[[LOOP]] ], [ [[INC]], %[[LATCH]] ]
+; CHECK-NEXT:    ret i32 [[R]]
+;
+entry:
+  %p = alloca [1024 x i8]
+  call void @init_mem(ptr %p, i64 1024)
+  %ct = call i32 @llvm.cttz.i32(i32 %x, i1 true)
+  %d = add nuw nsw i32 %ct, 1
+  br label %loop
+
+loop:
+  %i = phi i32 [ 0, %entry ], [ %inc, %latch ]
+  %j = phi i32 [ 0, %entry ], [ %jnext, %latch ]
+  %gep = getelementptr inbounds i8, ptr %p, i32 %j
+  %ld = load i8, ptr %gep, align 1
+  %found = icmp eq i8 %ld, 0
+  br i1 %found, label %exit, label %latch
+
+latch:
+  %inc = add nuw nsw i32 %i, %d
+  %jnext = add nuw nsw i32 %j, 1
+  %done = icmp uge i32 %inc, 64
+  br i1 %done, label %exit, label %loop, !llvm.loop !0
+
+exit:
+  %r = phi i32 [ 0, %loop ], [ %inc, %latch ]
+  ret i32 %r
+}
+
+; Same loop-varying early exit, but the divisor is known non-zero and not
+; poison, so the trip count is safe to speculate and the loop is vectorized.
+; This confirms the loop above is only rejected because of the poison divisor.
+define i32 @udiv_by_nonzero_nonpoison_step_early_exit_load(i32 noundef %x) {
+; CHECK-LABEL: define i32 @udiv_by_nonzero_nonpoison_step_early_exit_load(
+; CHECK-SAME: i32 noundef [[X:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[P:%.*]] = alloca [1024 x i8], align 1
+; CHECK-NEXT:    call void @init_mem(ptr [[P]], i64 1024)
+; CHECK-NEXT:    [[CT:%.*]] = call i32 @llvm.cttz.i32(i32 [[X]], i1 false)
+; CHECK-NEXT:    [[D:%.*]] = add nuw nsw i32 [[CT]], 1
+; CHECK-NEXT:    [[TMP0:%.*]] = udiv i32 63, [[D]]
+; CHECK-NEXT:    [[TMP1:%.*]] = add nuw nsw i32 [[TMP0]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP1]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = and i32 [[TMP1]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[TMP3:%.*]] = mul i32 [[N_VEC]], [[D]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[P]], i32 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP4]], align 1
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT:    [[TMP6:%.*]] = freeze <4 x i1> [[TMP5]]
+; CHECK-NEXT:    [[TMP7:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP6]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP1]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i32 [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    [[BC_RESUME_VAL1:%.*]] = phi i32 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[INC:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[J:%.*]] = phi i32 [ [[BC_RESUME_VAL1]], %[[SCALAR_PH]] ], [ [[JNEXT:%.*]], %[[LATCH]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[P]], i32 [[J]]
+; CHECK-NEXT:    [[LD:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-NEXT:    [[FOUND:%.*]] = icmp eq i8 [[LD]], 0
+; CHECK-NEXT:    br i1 [[FOUND]], label %[[EXIT]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[INC]] = add nuw nsw i32 [[I]], [[D]]
+; CHECK-NEXT:    [[JNEXT]] = add nuw nsw i32 [[J]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp uge i32 [[INC]], 64
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[R:%.*]] = phi i32 [ 0, %[[LOOP]] ], [ [[INC]], %[[LATCH]] ], [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[VECTOR_EARLY_EXIT]] ]
+; CHECK-NEXT:    ret i32 [[R]]
+;
+entry:
+  %p = alloca [1024 x i8]
+  call void @init_mem(ptr %p, i64 1024)
+  %ct = call i32 @llvm.cttz.i32(i32 %x, i1 false)
+  %d = add nuw nsw i32 %ct, 1
+  br label %loop
+
+loop:
+  %i = phi i32 [ 0, %entry ], [ %inc, %latch ]
+  %j = phi i32 [ 0, %entry ], [ %jnext, %latch ]
+  %gep = getelementptr inbounds i8, ptr %p, i32 %j
+  %ld = load i8, ptr %gep, align 1
+  %found = icmp eq i8 %ld, 0
+  br i1 %found, label %exit, label %latch
+
+latch:
+  %inc = add nuw nsw i32 %i, %d
+  %jnext = add nuw nsw i32 %j, 1
+  %done = icmp uge i32 %inc, 64
+  br i1 %done, label %exit, label %loop, !llvm.loop !0
+
+exit:
+  %r = phi i32 [ 0, %loop ], [ %inc, %latch ]
+  ret i32 %r
+}
+
 !0 = distinct !{!0, !1, !2}
 !1 = !{!"llvm.loop.vectorize.width", i32 4}
 !2 = !{!"llvm.loop.vectorize.enable"}



More information about the llvm-commits mailing list