[llvm] [NFC][LoopVectorize] Add test for early-exit trip count that may cause UB … (PR #219883)

via llvm-commits llvm-commits at lists.llvm.org
Sun Aug 30 22:33:12 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Madhur Amilkanthwar (madhur13490)

<details>
<summary>Changes</summary>

Pre-commit test for https://github.com/llvm/llvm-project/issues/219371, where a udiv in the trip count is speculated into the preheader even when its divisor may be poison.

---
Full diff: https://github.com/llvm/llvm-project/pull/219883.diff


1 Files Affected:

- (added) llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll (+150) 


``````````diff
diff --git a/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll b/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll
new file mode 100644
index 0000000000000..2f9a37918c65b
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll
@@ -0,0 +1,150 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -p loop-vectorize -force-vector-width=4 -force-vector-interleave=1 -S %s | FileCheck %s
+
+declare i32 @llvm.cttz.i32(i32, i1 immarg)
+
+; These loops have two exits: an early exit and the latch. The vectorizer
+; computes the trip count from the latch and emits that computation in the
+; preheader, before the loop runs. But in the original loop the latch is only
+; reached once the early exit has not been taken, so a trip count that is only
+; well defined on that path must not be computed unconditionally in the
+; preheader.
+
+; FIXME: %ct is poison when %x is 0, so %d may be poison. The udiv is currently
+; speculated into the preheader, where it can divide by poison, which is UB. The
+; original loop returns 0 without evaluating %d when %skip is true, so this is a
+; miscompile and the loop must not be vectorized.
+define i32 @udiv_by_poison_step(i32 %x, i1 %skip) {
+; CHECK-LABEL: define i32 @udiv_by_poison_step(
+; CHECK-SAME: i32 [[X:%.*]], i1 [[SKIP:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[CT:%.*]] = call i32 @llvm.cttz.i32(i32 [[X]], i1 true)
+; CHECK-NEXT:    [[D:%.*]] = add nuw nsw i32 [[CT]], 1
+; CHECK-NEXT:    [[TMP0:%.*]] = udiv i32 63, [[D]]
+; CHECK-NEXT:    [[TMP1:%.*]] = add nuw nsw i32 [[TMP0]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP1]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = and i32 [[TMP1]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i1> poison, i1 [[SKIP]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP3:%.*]] = mul i32 [[N_VEC]], [[D]]
+; CHECK-NEXT:    [[TMP4:%.*]] = freeze <4 x i1> [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP4]])
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP1]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i32 [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP1:.*]]
+; CHECK:       [[LOOP1]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[INC:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    br i1 [[SKIP]], label %[[EXIT]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[INC]] = add nuw nsw i32 [[I]], [[D]]
+; CHECK-NEXT:    [[DONE:%.*]] = icmp uge i32 [[INC]], 64
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[R:%.*]] = phi i32 [ 0, %[[LOOP1]] ], [ [[INC]], %[[LATCH]] ], [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[VECTOR_EARLY_EXIT]] ]
+; CHECK-NEXT:    ret i32 [[R]]
+;
+entry:
+  %ct = call i32 @llvm.cttz.i32(i32 %x, i1 true)
+  %d = add nuw nsw i32 %ct, 1
+  br label %loop
+
+loop:
+  %i = phi i32 [ 0, %entry ], [ %inc, %latch ]
+  br i1 %skip, label %exit, label %latch
+
+latch:
+  %inc = add nuw nsw i32 %i, %d
+  %done = icmp uge i32 %inc, 64
+  br i1 %done, label %exit, label %loop, !llvm.loop !0
+
+exit:
+  %r = phi i32 [ 0, %loop ], [ %inc, %latch ]
+  ret i32 %r
+}
+
+; Same loop, but the intrinsic does not return poison for 0 and %x is noundef,
+; so %d is a non-poison value in [1, 34). Here the udiv is safe to speculate,
+; and the loop can be vectorized.
+define i32 @udiv_by_nonzero_nonpoison_step(i32 noundef %x, i1 %skip) {
+; CHECK-LABEL: define i32 @udiv_by_nonzero_nonpoison_step(
+; CHECK-SAME: i32 noundef [[X:%.*]], i1 [[SKIP:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[CT:%.*]] = call i32 @llvm.cttz.i32(i32 [[X]], i1 false)
+; CHECK-NEXT:    [[D:%.*]] = add nuw nsw i32 [[CT]], 1
+; CHECK-NEXT:    [[TMP0:%.*]] = udiv i32 63, [[D]]
+; CHECK-NEXT:    [[TMP1:%.*]] = add nuw nsw i32 [[TMP0]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP1]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = and i32 [[TMP1]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i1> poison, i1 [[SKIP]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP3:%.*]] = mul i32 [[N_VEC]], [[D]]
+; CHECK-NEXT:    [[TMP4:%.*]] = freeze <4 x i1> [[BROADCAST_SPLAT]]
+; CHECK-NEXT:    [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP4]])
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP1]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i32 [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[INC:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    br i1 [[SKIP]], label %[[EXIT]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[INC]] = add nuw nsw i32 [[I]], [[D]]
+; CHECK-NEXT:    [[DONE:%.*]] = icmp uge i32 [[INC]], 64
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[R:%.*]] = phi i32 [ 0, %[[LOOP]] ], [ [[INC]], %[[LATCH]] ], [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[VECTOR_EARLY_EXIT]] ]
+; CHECK-NEXT:    ret i32 [[R]]
+;
+entry:
+  %ct = call i32 @llvm.cttz.i32(i32 %x, i1 false)
+  %d = add nuw nsw i32 %ct, 1
+  br label %loop
+
+loop:
+  %i = phi i32 [ 0, %entry ], [ %inc, %latch ]
+  br i1 %skip, label %exit, label %latch
+
+latch:
+  %inc = add nuw nsw i32 %i, %d
+  %done = icmp uge i32 %inc, 64
+  br i1 %done, label %exit, label %loop, !llvm.loop !0
+
+exit:
+  %r = phi i32 [ 0, %loop ], [ %inc, %latch ]
+  ret i32 %r
+}
+
+!0 = distinct !{!0, !1, !2}
+!1 = !{!"llvm.loop.vectorize.width", i32 4}
+!2 = !{!"llvm.loop.vectorize.enable"}

``````````

</details>


https://github.com/llvm/llvm-project/pull/219883


More information about the llvm-commits mailing list