[llvm] [InstCombine] Fold !umul_ov(X, C1) && X*C1 <u C2 to X <u ceil(C2/C1) (PR #227293)

via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 29 05:52:35 PDT 2026


https://github.com/RS-Gits updated https://github.com/llvm/llvm-project/pull/227293

>From a5dec8955005e2d66867399c8dd34f2b01b64f85 Mon Sep 17 00:00:00 2001
From: RS-Gits <harish14rs at gmail.com>
Date: Tue, 29 Sep 2026 17:46:55 +0530
Subject: [PATCH] [InstCombine] Fold !umul_ov(X, C1) && X*C1 <u C2 to X <u
 ceil(C2/C1)

Fold the "no overflow and below a limit" check on an unsigned multiply by
a constant into a single comparison:

  Res, Ov = umul.with.overflow(X, C1)
  !Ov && (Res <u C2)  -->  X <u ceil(C2 / C1)

This is the dual of the existing fold for "Ov || (Res >u C2)". It is only
done for non-zero C1 and C2, and when the compare has no other uses.

Alive2: https://alive2.llvm.org/ce/z/vKSYMY

Fixes #222691
---
 .../InstCombine/InstCombineAndOrXor.cpp       |  34 ++++
 .../InstCombine/icmp_and_umul_no_overflow.ll  | 151 ++++++++++++++++++
 2 files changed, 185 insertions(+)
 create mode 100644 llvm/test/Transforms/InstCombine/icmp_and_umul_no_overflow.ll

diff --git a/llvm/lib/Transforms/InstCombine/InstCombineAndOrXor.cpp b/llvm/lib/Transforms/InstCombine/InstCombineAndOrXor.cpp
index 66c14cfba2056..f8506c9ce4ff5 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineAndOrXor.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineAndOrXor.cpp
@@ -2469,6 +2469,37 @@ Value *InstCombinerImpl::reassociateBooleanAndOr(Value *LHS, Value *X, Value *Y,
   return Folded;
 }
 
+/// Fold Res, Overflow = (umul.with.overflow x c1); (and !Overflow (ult Res c2))
+/// --> (ult x ceil(c2/c1)). This is the dual of
+/// foldOrUnsignedUMulOverflowICmp: the product does not overflow and is below
+/// c2 iff x is below c2 divided by c1, rounded up.
+static Value *
+foldAndUnsignedUMulOverflowICmp(BinaryOperator &I,
+                                InstCombiner::BuilderTy &Builder) {
+  Value *WOV, *X;
+  const APInt *C1, *C2;
+  if (match(&I,
+            m_c_And(m_Not(m_ExtractValue<1>(
+                        m_Value(WOV, m_Intrinsic<Intrinsic::umul_with_overflow>(
+                                         m_Value(X), m_APInt(C1))))),
+                    m_OneUse(m_SpecificCmp(ICmpInst::ICMP_ULT,
+                                           m_ExtractValue<0>(m_Deferred(WOV)),
+                                           m_APInt(C2))))) &&
+      // A zero multiplier would divide by zero, and "ult 0" is always false
+      // (left for other folds to simplify).
+      !C1->isZero() && !C2->isZero()) {
+    APInt Quotient, Remainder;
+    APInt::udivrem(*C2, *C1, Quotient, Remainder);
+    // Quotient < C2 whenever a rounding increment is needed (C1 > 1), so this
+    // cannot wrap.
+    if (!Remainder.isZero())
+      ++Quotient;
+    return Builder.CreateICmp(ICmpInst::ICMP_ULT, X,
+                              ConstantInt::get(X->getType(), Quotient));
+  }
+  return nullptr;
+}
+
 // FIXME: We use commutative matchers (m_c_*) for some, but not all, matches
 // here. We should standardize that construct where it is needed or choose some
 // other way to ensure that commutated variants of patterns are not missed.
@@ -2853,6 +2884,9 @@ Instruction *InstCombinerImpl::visitAnd(BinaryOperator &I) {
           foldBooleanAndOr(Op0, Op1, I, /*IsAnd=*/true, /*IsLogical=*/false))
     return replaceInstUsesWith(I, Res);
 
+  if (Value *Res = foldAndUnsignedUMulOverflowICmp(I, Builder))
+    return replaceInstUsesWith(I, Res);
+
   if (match(Op1, m_OneUse(m_LogicalAnd(m_Value(X), m_Value(Y))))) {
     bool IsLogical = isa<SelectInst>(Op1);
     if (auto *V = reassociateBooleanAndOr(Op0, X, Y, I, /*IsAnd=*/true,
diff --git a/llvm/test/Transforms/InstCombine/icmp_and_umul_no_overflow.ll b/llvm/test/Transforms/InstCombine/icmp_and_umul_no_overflow.ll
new file mode 100644
index 0000000000000..f16ab1a08943f
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/icmp_and_umul_no_overflow.ll
@@ -0,0 +1,151 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -S -passes=instcombine < %s | FileCheck %s
+
+declare void @use.i1(i1 %x)
+
+; !umul_overflow(x, K) && (x * K <u C) --> x <u ceil(C / K)
+
+define i1 @umul_no_overflow_and_ult_const(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_const(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = icmp ult i32 [[X]], 14
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 3)
+  %ov = extractvalue { i32, i1 } %m, 1
+  %p = extractvalue { i32, i1 } %m, 0
+  %lt = icmp ult i32 %p, 41
+  %no = xor i1 %ov, true
+  %r = and i1 %lt, %no
+  ret i1 %r
+}
+
+define i1 @umul_no_overflow_and_ult_const_commuted(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_const_commuted(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = icmp ult i32 [[X]], 14
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 3)
+  %ov = extractvalue { i32, i1 } %m, 1
+  %p = extractvalue { i32, i1 } %m, 0
+  %lt = icmp ult i32 %p, 41
+  %no = xor i1 %ov, true
+  %r = and i1 %no, %lt
+  ret i1 %r
+}
+
+; C is a multiple of K: no rounding needed.
+define i1 @umul_no_overflow_and_ult_exact(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_exact(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = icmp ult i32 [[X]], 8
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 5)
+  %ov = extractvalue { i32, i1 } %m, 1
+  %p = extractvalue { i32, i1 } %m, 0
+  %lt = icmp ult i32 %p, 40
+  %no = xor i1 %ov, true
+  %r = and i1 %lt, %no
+  ret i1 %r
+}
+
+define i1 @umul_no_overflow_and_ult_k1(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_k1(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[LT:%.*]] = icmp ult i32 [[X]], 7
+; CHECK-NEXT:    ret i1 [[LT]]
+;
+  %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 1)
+  %ov = extractvalue { i32, i1 } %m, 1
+  %p = extractvalue { i32, i1 } %m, 0
+  %lt = icmp ult i32 %p, 7
+  %no = xor i1 %ov, true
+  %r = and i1 %lt, %no
+  ret i1 %r
+}
+
+; i64 with large constants.
+define i1 @umul_no_overflow_and_ult_i64(i64 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_i64(
+; CHECK-SAME: i64 [[X:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = icmp ult i64 [[X]], 109802048057794950
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %m = call { i64, i1 } @llvm.umul.with.overflow.i64(i64 %x, i64 168)
+  %ov = extractvalue { i64, i1 } %m, 1
+  %p = extractvalue { i64, i1 } %m, 0
+  %lt = icmp ult i64 %p, -16
+  %no = xor i1 %ov, true
+  %r = and i1 %lt, %no
+  ret i1 %r
+}
+
+; The compare has another use: don't fold.
+define i1 @umul_no_overflow_and_ult_extra_use(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_extra_use(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 [[X]], i32 3)
+; CHECK-NEXT:    [[OV:%.*]] = extractvalue { i32, i1 } [[M]], 1
+; CHECK-NEXT:    [[P:%.*]] = extractvalue { i32, i1 } [[M]], 0
+; CHECK-NEXT:    [[LT:%.*]] = icmp ult i32 [[P]], 41
+; CHECK-NEXT:    call void @use.i1(i1 [[LT]])
+; CHECK-NEXT:    [[NO:%.*]] = xor i1 [[OV]], true
+; CHECK-NEXT:    [[R:%.*]] = and i1 [[LT]], [[NO]]
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 3)
+  %ov = extractvalue { i32, i1 } %m, 1
+  %p = extractvalue { i32, i1 } %m, 0
+  %lt = icmp ult i32 %p, 41
+  call void @use.i1(i1 %lt)
+  %no = xor i1 %ov, true
+  %r = and i1 %lt, %no
+  ret i1 %r
+}
+
+; ule is canonicalized to ult (ule 41 -> ult 42) first, so this still folds.
+define i1 @umul_no_overflow_and_ule_canonicalized_to_ult(i32 %x) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ule_canonicalized_to_ult(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = icmp ult i32 [[X]], 14
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 3)
+  %ov = extractvalue { i32, i1 } %m, 1
+  %p = extractvalue { i32, i1 } %m, 0
+  %lt = icmp ule i32 %p, 41
+  %no = xor i1 %ov, true
+  %r = and i1 %lt, %no
+  ret i1 %r
+}
+
+; Non-constant multiplier: not handled.
+define i1 @umul_no_overflow_and_ult_var(i32 %x, i32 %k) {
+;
+; CHECK-LABEL: define i1 @umul_no_overflow_and_ult_var(
+; CHECK-SAME: i32 [[X:%.*]], i32 [[K:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 [[X]], i32 [[K]])
+; CHECK-NEXT:    [[OV:%.*]] = extractvalue { i32, i1 } [[M]], 1
+; CHECK-NEXT:    [[P:%.*]] = extractvalue { i32, i1 } [[M]], 0
+; CHECK-NEXT:    [[LT:%.*]] = icmp ult i32 [[P]], 41
+; CHECK-NEXT:    [[NO:%.*]] = xor i1 [[OV]], true
+; CHECK-NEXT:    [[R:%.*]] = and i1 [[LT]], [[NO]]
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %m = call { i32, i1 } @llvm.umul.with.overflow.i32(i32 %x, i32 %k)
+  %ov = extractvalue { i32, i1 } %m, 1
+  %p = extractvalue { i32, i1 } %m, 0
+  %lt = icmp ult i32 %p, 41
+  %no = xor i1 %ov, true
+  %r = and i1 %lt, %no
+  ret i1 %r
+}



More information about the llvm-commits mailing list