[llvm] [InstCombine] Fold !umul_ov(X, C1) && X*C1 <u C2 to X <u ceil(C2/C1) (PR #227293)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 29 05:52:35 PDT 2026
https://github.com/RS-Gits updated https://github.com/llvm/llvm-project/pull/227293
>From a5dec8955005e2d66867399c8dd34f2b01b64f85 Mon Sep 17 00:00:00 2001
From: RS-Gits <harish14rs at gmail.com>
Date: Tue, 29 Sep 2026 17:46:55 +0530
Subject: [PATCH] [InstCombine] Fold !umul_ov(X, C1) && X*C1 <u C2 to X <u
ceil(C2/C1)
Fold the "no overflow and below a limit" check on an unsigned multiply by
a constant into a single comparison:
Res, Ov = umul.with.overflow(X, C1)
!Ov && (Res <u C2) --> X <u ceil(C2 / C1)
This is the dual of the existing fold for "Ov || (Res >u C2)". It is only
done for non-zero C1 and C2, and when the compare has no other uses.
Alive2: https://alive2.llvm.org/ce/z/vKSYMY
Fixes #222691
---
.../InstCombine/InstCombineAndOrXor.cpp | 34 ++++
.../InstCombine/icmp_and_umul_no_overflow.ll | 151 ++++++++++++++++++
2 files changed, 185 insertions(+)
create mode 100644 llvm/test/Transforms/InstCombine/icmp_and_umul_no_overflow.ll
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineAndOrXor.cpp b/llvm/lib/Transforms/InstCombine/InstCombineAndOrXor.cpp
index 66c14cfba2056..f8506c9ce4ff5 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineAndOrXor.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineAndOrXor.cpp
@@ -2469,6 +2469,37 @@ Value *InstCombinerImpl::reassociateBooleanAndOr(Value *LHS, Value *X, Value *Y,
return Folded;
}
+/// Fold Res, Overflow = (umul.with.overflow x c1); (and !Overflow (ult Res c2))
+/// --> (ult x ceil(c2/c1)). This is the dual of
+/// foldOrUnsignedUMulOverflowICmp: the product does not overflow and is below
+/// c2 iff x is below c2 divided by c1, rounded up.
+static Value *
+foldAndUnsignedUMulOverflowICmp(BinaryOperator &I,
+ InstCombiner::BuilderTy &Builder) {
+ Value *WOV, *X;
+ const APInt *C1, *C2;
+ if (match(&I,
+ m_c_And(m_Not(m_ExtractValue<1>(
+ m_Value(WOV, m_Intrinsic<Intrinsic::umul_with_overflow>(
+ m_Value(X), m_APInt(C1))))),
+ m_OneUse(m_SpecificCmp(ICmpInst::ICMP_ULT,
+ m_ExtractValue<0>(m_Deferred(WOV)),
+ m_APInt(C2))))) &&
+ // A zero multiplier would divide by zero, and "ult 0" is always false
+ // (left for other folds to simplify).
+ !C1->isZero() && !C2->isZero()) {
+ APInt Quotient, Remainder;
+ APInt::udivrem(*C2, *C1, Quotient, Remainder);
+ // Quotient < C2 whenever a rounding increment is needed (C1 > 1), so this
+ // cannot wrap.
+ if (!Remainder.isZero())
+ ++Quotient;
+ return Builder.CreateICmp(ICmpInst::ICMP_ULT, X,
+ ConstantInt::get(X->getType(), Quotient));
+ }
+ return nullptr;
+}
+
// FIXME: We use commutative matchers (m_c_*) for some, but not all, matches
// here. We should standardize that construct where it is needed or choose some
// other way to ensure that commutated variants of patterns are not missed.
@@ -2853,6 +2884,9 @@ Instruction *InstCombinerImpl::visitAnd(BinaryOperator &I) {
foldBooleanAndOr(Op0, Op1, I, /*IsAnd=*/true, /*IsLogical=*/false))
return replaceInstUsesWith(I, Res);
+ if (Value *Res = foldAndUnsignedUMulOverflowICmp(I, Builder))
+ return replaceInstUsesWith(I, Res);
+
if (match(Op1, m_OneUse(m_LogicalAnd(m_Value(X), m_Value(Y))))) {
bool IsLogical = isa<SelectInst>(Op1);
if (auto *V = reassociateBooleanAndOr(Op0, X, Y, I, /*IsAnd=*/true,
diff --git a/llvm/test/Transforms/InstCombine/icmp_and_umul_no_overflow.ll b/llvm/test/Transforms/InstCombine/icmp_and_umul_no_overflow.ll
new file mode 100644
index 0000000000000..f16ab1a08943f
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/icmp_and_umul_no_overflow.ll
@@ -0,0 +1,151 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -S -passes=instcombine < %s | FileCheck %s
+
+declare void @use.i1(i1 %x)
+
+; !umul_overflow(x, K) && (x * K <u C) --> x <u ceil(C / K)
+
+define i1 @umul_no_overflow_and_ult_const(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_const(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[R:%.*]] = icmp ult i32 [[X]], 14
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 3)
+ %ov = extractvalue { i32, i1 } %m, 1
+ %p = extractvalue { i32, i1 } %m, 0
+ %lt = icmp ult i32 %p, 41
+ %no = xor i1 %ov, true
+ %r = and i1 %lt, %no
+ ret i1 %r
+}
+
+define i1 @umul_no_overflow_and_ult_const_commuted(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_const_commuted(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[R:%.*]] = icmp ult i32 [[X]], 14
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 3)
+ %ov = extractvalue { i32, i1 } %m, 1
+ %p = extractvalue { i32, i1 } %m, 0
+ %lt = icmp ult i32 %p, 41
+ %no = xor i1 %ov, true
+ %r = and i1 %no, %lt
+ ret i1 %r
+}
+
+; C is a multiple of K: no rounding needed.
+define i1 @umul_no_overflow_and_ult_exact(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_exact(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[R:%.*]] = icmp ult i32 [[X]], 8
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 5)
+ %ov = extractvalue { i32, i1 } %m, 1
+ %p = extractvalue { i32, i1 } %m, 0
+ %lt = icmp ult i32 %p, 40
+ %no = xor i1 %ov, true
+ %r = and i1 %lt, %no
+ ret i1 %r
+}
+
+define i1 @umul_no_overflow_and_ult_k1(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_k1(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[LT:%.*]] = icmp ult i32 [[X]], 7
+; CHECK-NEXT: ret i1 [[LT]]
+;
+ %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 1)
+ %ov = extractvalue { i32, i1 } %m, 1
+ %p = extractvalue { i32, i1 } %m, 0
+ %lt = icmp ult i32 %p, 7
+ %no = xor i1 %ov, true
+ %r = and i1 %lt, %no
+ ret i1 %r
+}
+
+; i64 with large constants.
+define i1 @umul_no_overflow_and_ult_i64(i64 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_i64(
+; CHECK-SAME: i64 [[X:%.*]]) {
+; CHECK-NEXT: [[R:%.*]] = icmp ult i64 [[X]], 109802048057794950
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %m = call { i64, i1 } @llvm.umul.with.overflow.i64(i64 %x, i64 168)
+ %ov = extractvalue { i64, i1 } %m, 1
+ %p = extractvalue { i64, i1 } %m, 0
+ %lt = icmp ult i64 %p, -16
+ %no = xor i1 %ov, true
+ %r = and i1 %lt, %no
+ ret i1 %r
+}
+
+; The compare has another use: don't fold.
+define i1 @umul_no_overflow_and_ult_extra_use(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_extra_use(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 [[X]], i32 3)
+; CHECK-NEXT: [[OV:%.*]] = extractvalue { i32, i1 } [[M]], 1
+; CHECK-NEXT: [[P:%.*]] = extractvalue { i32, i1 } [[M]], 0
+; CHECK-NEXT: [[LT:%.*]] = icmp ult i32 [[P]], 41
+; CHECK-NEXT: call void @use.i1(i1 [[LT]])
+; CHECK-NEXT: [[NO:%.*]] = xor i1 [[OV]], true
+; CHECK-NEXT: [[R:%.*]] = and i1 [[LT]], [[NO]]
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 3)
+ %ov = extractvalue { i32, i1 } %m, 1
+ %p = extractvalue { i32, i1 } %m, 0
+ %lt = icmp ult i32 %p, 41
+ call void @use.i1(i1 %lt)
+ %no = xor i1 %ov, true
+ %r = and i1 %lt, %no
+ ret i1 %r
+}
+
+; ule is canonicalized to ult (ule 41 -> ult 42) first, so this still folds.
+define i1 @umul_no_overflow_and_ule_canonicalized_to_ult(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ule_canonicalized_to_ult(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[R:%.*]] = icmp ult i32 [[X]], 14
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 3)
+ %ov = extractvalue { i32, i1 } %m, 1
+ %p = extractvalue { i32, i1 } %m, 0
+ %lt = icmp ule i32 %p, 41
+ %no = xor i1 %ov, true
+ %r = and i1 %lt, %no
+ ret i1 %r
+}
+
+; Non-constant multiplier: not handled.
+define i1 @umul_no_overflow_and_ult_var(i32 %x, i32 %k) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_var(
+; CHECK-SAME: i32 [[X:%.*]], i32 [[K:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 [[X]], i32 [[K]])
+; CHECK-NEXT: [[OV:%.*]] = extractvalue { i32, i1 } [[M]], 1
+; CHECK-NEXT: [[P:%.*]] = extractvalue { i32, i1 } [[M]], 0
+; CHECK-NEXT: [[LT:%.*]] = icmp ult i32 [[P]], 41
+; CHECK-NEXT: [[NO:%.*]] = xor i1 [[OV]], true
+; CHECK-NEXT: [[R:%.*]] = and i1 [[LT]], [[NO]]
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 %k)
+ %ov = extractvalue { i32, i1 } %m, 1
+ %p = extractvalue { i32, i1 } %m, 0
+ %lt = icmp ult i32 %p, 41
+ %no = xor i1 %ov, true
+ %r = and i1 %lt, %no
+ ret i1 %r
+}
More information about the llvm-commits
mailing list