[llvm] [NFC][LoopVectorize] Add test for early-exit trip count that may cause UB … (PR #219883)
via llvm-commits
llvm-commits at lists.llvm.org
Sun Aug 30 22:33:12 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: Madhur Amilkanthwar (madhur13490)
<details>
<summary>Changes</summary>
Pre-commit test for https://github.com/llvm/llvm-project/issues/219371, where a udiv in the trip count is speculated into the preheader even when its divisor may be poison.
---
Full diff: https://github.com/llvm/llvm-project/pull/219883.diff
1 Files Affected:
- (added) llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll (+150)
``````````diff
diff --git a/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll b/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll
new file mode 100644
index 0000000000000..2f9a37918c65b
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/early-exit-trip-count-may-cause-ub.ll
@@ -0,0 +1,150 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -p loop-vectorize -force-vector-width=4 -force-vector-interleave=1 -S %s | FileCheck %s
+
+declare i32 @llvm.cttz.i32(i32, i1 immarg)
+
+; These loops have two exits: an early exit and the latch. The vectorizer
+; computes the trip count from the latch and emits that computation in the
+; preheader, before the loop runs. But in the original loop the latch is only
+; reached once the early exit has not been taken, so a trip count that is only
+; well defined on that path must not be computed unconditionally in the
+; preheader.
+
+; FIXME: %ct is poison when %x is 0, so %d may be poison. The udiv is currently
+; speculated into the preheader, where it can divide by poison, which is UB. The
+; original loop returns 0 without evaluating %d when %skip is true, so this is a
+; miscompile and the loop must not be vectorized.
+define i32 @udiv_by_poison_step(i32 %x, i1 %skip) {
+; CHECK-LABEL: define i32 @udiv_by_poison_step(
+; CHECK-SAME: i32 [[X:%.*]], i1 [[SKIP:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[CT:%.*]] = call i32 @llvm.cttz.i32(i32 [[X]], i1 true)
+; CHECK-NEXT: [[D:%.*]] = add nuw nsw i32 [[CT]], 1
+; CHECK-NEXT: [[TMP0:%.*]] = udiv i32 63, [[D]]
+; CHECK-NEXT: [[TMP1:%.*]] = add nuw nsw i32 [[TMP0]], 1
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP1]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP2:%.*]] = and i32 [[TMP1]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i1> poison, i1 [[SKIP]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP3:%.*]] = mul i32 [[N_VEC]], [[D]]
+; CHECK-NEXT: [[TMP4:%.*]] = freeze <4 x i1> [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP4]])
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP5]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK: [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[TMP1]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[I:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[INC:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT: br i1 [[SKIP]], label %[[EXIT]], label %[[LATCH]]
+; CHECK: [[LATCH]]:
+; CHECK-NEXT: [[INC]] = add nuw nsw i32 [[I]], [[D]]
+; CHECK-NEXT: [[DONE:%.*]] = icmp uge i32 [[INC]], 64
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[R:%.*]] = phi i32 [ 0, %[[LOOP1]] ], [ [[INC]], %[[LATCH]] ], [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[VECTOR_EARLY_EXIT]] ]
+; CHECK-NEXT: ret i32 [[R]]
+;
+entry:
+ %ct = call i32 @llvm.cttz.i32(i32 %x, i1 true)
+ %d = add nuw nsw i32 %ct, 1
+ br label %loop
+
+loop:
+ %i = phi i32 [ 0, %entry ], [ %inc, %latch ]
+ br i1 %skip, label %exit, label %latch
+
+latch:
+ %inc = add nuw nsw i32 %i, %d
+ %done = icmp uge i32 %inc, 64
+ br i1 %done, label %exit, label %loop, !llvm.loop !0
+
+exit:
+ %r = phi i32 [ 0, %loop ], [ %inc, %latch ]
+ ret i32 %r
+}
+
+; Same loop, but the intrinsic does not return poison for 0 and %x is noundef,
+; so %d is a non-poison value in [1, 34). Here the udiv is safe to speculate,
+; and the loop can be vectorized.
+define i32 @udiv_by_nonzero_nonpoison_step(i32 noundef %x, i1 %skip) {
+; CHECK-LABEL: define i32 @udiv_by_nonzero_nonpoison_step(
+; CHECK-SAME: i32 noundef [[X:%.*]], i1 [[SKIP:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[CT:%.*]] = call i32 @llvm.cttz.i32(i32 [[X]], i1 false)
+; CHECK-NEXT: [[D:%.*]] = add nuw nsw i32 [[CT]], 1
+; CHECK-NEXT: [[TMP0:%.*]] = udiv i32 63, [[D]]
+; CHECK-NEXT: [[TMP1:%.*]] = add nuw nsw i32 [[TMP0]], 1
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP1]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP2:%.*]] = and i32 [[TMP1]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i1> poison, i1 [[SKIP]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP3:%.*]] = mul i32 [[N_VEC]], [[D]]
+; CHECK-NEXT: [[TMP4:%.*]] = freeze <4 x i1> [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP4]])
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP5]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK: [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[TMP1]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[I:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[INC:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT: br i1 [[SKIP]], label %[[EXIT]], label %[[LATCH]]
+; CHECK: [[LATCH]]:
+; CHECK-NEXT: [[INC]] = add nuw nsw i32 [[I]], [[D]]
+; CHECK-NEXT: [[DONE:%.*]] = icmp uge i32 [[INC]], 64
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[R:%.*]] = phi i32 [ 0, %[[LOOP]] ], [ [[INC]], %[[LATCH]] ], [ [[TMP3]], %[[MIDDLE_BLOCK]] ], [ 0, %[[VECTOR_EARLY_EXIT]] ]
+; CHECK-NEXT: ret i32 [[R]]
+;
+entry:
+ %ct = call i32 @llvm.cttz.i32(i32 %x, i1 false)
+ %d = add nuw nsw i32 %ct, 1
+ br label %loop
+
+loop:
+ %i = phi i32 [ 0, %entry ], [ %inc, %latch ]
+ br i1 %skip, label %exit, label %latch
+
+latch:
+ %inc = add nuw nsw i32 %i, %d
+ %done = icmp uge i32 %inc, 64
+ br i1 %done, label %exit, label %loop, !llvm.loop !0
+
+exit:
+ %r = phi i32 [ 0, %loop ], [ %inc, %latch ]
+ ret i32 %r
+}
+
+!0 = distinct !{!0, !1, !2}
+!1 = !{!"llvm.loop.vectorize.width", i32 4}
+!2 = !{!"llvm.loop.vectorize.enable"}
``````````
</details>
https://github.com/llvm/llvm-project/pull/219883
More information about the llvm-commits
mailing list