[llvm] [InstCombine] Fold udiv by constant into icmp and add when maximum quotient is small (PR #209321)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Jul 13 15:51:55 PDT 2026
https://github.com/AZero13 updated https://github.com/llvm/llvm-project/pull/209321
>From 353ba9a3ef114a7154ac0239b35a6f9398791bb2 Mon Sep 17 00:00:00 2001
From: AZero13 <gfunni234 at gmail.com>
Date: Mon, 13 Jul 2026 18:11:35 -0400
Subject: [PATCH] [InstCombine] Fold udiv by constant into icmp and add when
maximum quotient is small
This patch extends InstCombine to optimize udiv operations where the divisor is a constant and the numerator is constrained such that the maximum possible quotient is very small (either 1 or 2).
Previously, InstCombine would only fold Op0 / C -> zext(Op0 >= C) when C was negative (i.e. having the sign bit set), as this syntactically guarantees the quotient is 0 or 1.
This patch generalizes this behavior by using computeConstantRange to dynamically determine the maximum unsigned quotient MaxQ.
If MaxQ == 1, it folds to zext(Op0 >= C) (catching previously missed cases where C is positive, but Op0 is constrained by range metadata or assume()).
If MaxQ == 2, it folds to zext(Op0 >= C) + zext(Op0 >= 2C), which avoids the expensive hardware division and allows targets (like x86 and ARM) to efficiently lower the sequence into optimal conditional accumulations.
This specifically optimizes ubiquitous edge cases like value / INT32_MAX, which previously forced a full division by magic constant expansion, but now lowers flawlessly into native IR comparisons.
Alive2 proof: https://alive2.llvm.org/ce/z/tZG5WS
---
.../InstCombine/InstCombineMulDivRem.cpp | 30 +++++++++++-
llvm/test/Transforms/InstCombine/div.ll | 6 ++-
.../InstCombine/udiv-small-quotient.ll | 48 +++++++++++++++++++
3 files changed, 81 insertions(+), 3 deletions(-)
create mode 100644 llvm/test/Transforms/InstCombine/udiv-small-quotient.ll
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineMulDivRem.cpp b/llvm/lib/Transforms/InstCombine/InstCombineMulDivRem.cpp
index 39598dbbf6c67..87129a09ef1f8 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineMulDivRem.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineMulDivRem.cpp
@@ -1775,13 +1775,41 @@ Instruction *InstCombinerImpl::visitUDiv(BinaryOperator &I) {
}
}
+ Type *Ty = I.getType();
+
// Op0 / C where C is large (negative) --> zext (Op0 >= C)
// TODO: Could use isKnownNegative() to handle non-constant values.
- Type *Ty = I.getType();
if (match(Op1, m_Negative())) {
Value *Cmp = Builder.CreateICmpUGE(Op0, Op1);
return CastInst::CreateZExtOrBitCast(Cmp, Ty);
}
+
+ // Op0 / C --> (Op0 >= C) + (Op0 >= 2C) if MaxQuotient <= 2
+ if (match(Op1, m_APInt(C2)) && !C2->isZero()) {
+ ConstantRange CR = computeConstantRange(Op0, /*ForSigned=*/false,
+ SQ.getWithInstruction(&I));
+ APInt MaxOp0 = CR.getUnsignedMax();
+ APInt MaxQ = MaxOp0.udiv(*C2);
+ if (MaxQ == 1) {
+ Value *Cmp = Builder.CreateICmpUGE(Op0, Op1);
+ return CastInst::CreateZExtOrBitCast(Cmp, Ty);
+ }
+
+ if (MaxQ == 2) {
+ Value *Cmp1 = Builder.CreateICmpUGE(Op0, Op1);
+ Value *Zext1 = Builder.CreateZExtOrBitCast(Cmp1, Ty);
+ bool Overflow;
+ APInt C2x2 = C2->uadd_ov(*C2, Overflow);
+ if (!Overflow) {
+ Value *Cmp2 = Builder.CreateICmpUGE(Op0, ConstantInt::get(Ty, C2x2));
+ Value *Zext2 = Builder.CreateZExtOrBitCast(Cmp2, Ty);
+ BinaryOperator *Add = BinaryOperator::CreateNUWAdd(Zext1, Zext2);
+ Add->setHasNoSignedWrap();
+ return Add;
+ }
+ }
+ }
+
// Op0 / (sext i1 X) --> zext (Op0 == -1) (if X is 0, the div is undefined)
if (match(Op1, m_SExt(m_Value(X))) && X->getType()->isIntOrIntVectorTy(1)) {
Value *Cmp = Builder.CreateICmpEQ(Op0, ConstantInt::getAllOnesValue(Ty));
diff --git a/llvm/test/Transforms/InstCombine/div.ll b/llvm/test/Transforms/InstCombine/div.ll
index 2adf6036d4bcf..7947e677e9561 100644
--- a/llvm/test/Transforms/InstCombine/div.ll
+++ b/llvm/test/Transforms/InstCombine/div.ll
@@ -555,7 +555,8 @@ define <2 x i8> @sdiv_exact_negated_dividend_constant_divisor_vec_overflow(<2 x
define i32 @test35(i32 %A) {
; CHECK-LABEL: @test35(
; CHECK-NEXT: [[AND:%.*]] = and i32 [[A:%.*]], 2147483647
-; CHECK-NEXT: [[MUL:%.*]] = udiv exact i32 [[AND]], 2147483647
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq i32 [[AND]], 2147483647
+; CHECK-NEXT: [[MUL:%.*]] = zext i1 [[TMP1]] to i32
; CHECK-NEXT: ret i32 [[MUL]]
;
%and = and i32 %A, 2147483647
@@ -566,7 +567,8 @@ define i32 @test35(i32 %A) {
define <2 x i32> @test35vec(<2 x i32> %A) {
; CHECK-LABEL: @test35vec(
; CHECK-NEXT: [[AND:%.*]] = and <2 x i32> [[A:%.*]], splat (i32 2147483647)
-; CHECK-NEXT: [[MUL:%.*]] = udiv exact <2 x i32> [[AND]], splat (i32 2147483647)
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <2 x i32> [[AND]], splat (i32 2147483647)
+; CHECK-NEXT: [[MUL:%.*]] = zext <2 x i1> [[TMP1]] to <2 x i32>
; CHECK-NEXT: ret <2 x i32> [[MUL]]
;
%and = and <2 x i32> %A, <i32 2147483647, i32 2147483647>
diff --git a/llvm/test/Transforms/InstCombine/udiv-small-quotient.ll b/llvm/test/Transforms/InstCombine/udiv-small-quotient.ll
new file mode 100644
index 0000000000000..9b4e40f992f40
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/udiv-small-quotient.ll
@@ -0,0 +1,48 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt < %s -passes=instcombine -S | FileCheck %s
+
+define i32 @udiv_max_q1(i32 %x) {
+; CHECK-LABEL: @udiv_max_q1(
+; CHECK-NEXT: [[CMP:%.*]] = icmp ult i32 [[X:%.*]], 100
+; CHECK-NEXT: call void @llvm.assume(i1 [[CMP]])
+; CHECK-NEXT: [[TMP1:%.*]] = icmp samesign ugt i32 [[X]], 59
+; CHECK-NEXT: [[DIV:%.*]] = zext i1 [[TMP1]] to i32
+; CHECK-NEXT: ret i32 [[DIV]]
+;
+ %cmp = icmp ult i32 %x, 100
+ call void @llvm.assume(i1 %cmp)
+ %div = udiv i32 %x, 60
+ ret i32 %div
+}
+
+define i32 @udiv_max_q2(i32 %x) {
+; CHECK-LABEL: @udiv_max_q2(
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ugt i32 [[X:%.*]], 2147483646
+; CHECK-NEXT: [[TMP2:%.*]] = zext i1 [[TMP1]] to i32
+; CHECK-NEXT: [[TMP3:%.*]] = icmp ugt i32 [[X]], -3
+; CHECK-NEXT: [[TMP4:%.*]] = zext i1 [[TMP3]] to i32
+; CHECK-NEXT: [[DIV:%.*]] = add nuw nsw i32 [[TMP2]], [[TMP4]]
+; CHECK-NEXT: ret i32 [[DIV]]
+;
+ %div = udiv i32 %x, 2147483647
+ ret i32 %div
+}
+
+define i32 @udiv_max_q2_with_assume(i32 %x) {
+; CHECK-LABEL: @udiv_max_q2_with_assume(
+; CHECK-NEXT: [[CMP:%.*]] = icmp ult i32 [[X:%.*]], 150
+; CHECK-NEXT: call void @llvm.assume(i1 [[CMP]])
+; CHECK-NEXT: [[TMP1:%.*]] = icmp samesign ugt i32 [[X]], 59
+; CHECK-NEXT: [[TMP2:%.*]] = zext i1 [[TMP1]] to i32
+; CHECK-NEXT: [[TMP3:%.*]] = icmp samesign ugt i32 [[X]], 119
+; CHECK-NEXT: [[TMP4:%.*]] = zext i1 [[TMP3]] to i32
+; CHECK-NEXT: [[DIV:%.*]] = add nuw nsw i32 [[TMP2]], [[TMP4]]
+; CHECK-NEXT: ret i32 [[DIV]]
+;
+ %cmp = icmp ult i32 %x, 150
+ call void @llvm.assume(i1 %cmp)
+ %div = udiv i32 %x, 60
+ ret i32 %div
+}
+
+declare void @llvm.assume(i1)
More information about the llvm-commits
mailing list