[llvm] [InstCombine] fold fabs(uitofp(i16 a) - uitofp(i16 b)) < 1.0 to a == b (PR #191378)
Shreeyash Pandey via llvm-commits
llvm-commits at lists.llvm.org
Thu Apr 16 03:29:12 PDT 2026
https://github.com/bojle updated https://github.com/llvm/llvm-project/pull/191378
>From 4981ea446c6dc69886608588ef5fb95033d830ed Mon Sep 17 00:00:00 2001
From: Shreeyash Pandey <shrpand at qti.qualcomm.com>
Date: Thu, 26 Mar 2026 04:29:23 -0700
Subject: [PATCH 1/7] [InstCombine] fold fabs(uitofp(i16 a) - uitofp(i16 b)) <
1.0 to a == b
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Fixes: https://github.com/llvm/llvm-project/issues/187088
When a and b are types with bitwidth (16 bits) smaller than the mantissa
for float32 (24 bits), they will be exact and their absolute difference
would be integral ±1 or greater if a != b. On the corollary, if their
difference is < 1.0, this implies that a = b.
This patch exploits this fact to fold the expression to just a single
icmp.
Change-Id: I953f7e8bbdc8694632b739db4d465f5a0ffacf17
---
.../InstCombine/InstCombineCompares.cpp | 74 ++++++++++
llvm/test/Transforms/InstCombine/fcmp.ll | 138 ++++++++++++++++++
2 files changed, 212 insertions(+)
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
index 9cb7a4ec61903..dbd8c9a45ae03 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
@@ -8765,6 +8765,77 @@ static Instruction *foldFCmpFSubIntoFCmp(FCmpInst &I, Instruction *LHSI,
return nullptr;
}
+/// Fold: fabs(uitofp(a) - uitofp(b)) pred C --> a == b
+/// where 'pred' is olt, ole, ult, or ule, and C is a positive, Non-NaN float
+/// when the uitofp casts are exact and C is in the valid range.
+///
+/// Since exact uitofp means distinct integers map to distinct floats, the only
+/// values fabs(uitofp(a) - uitofp(b)) can take are {0.0, 1.0, 2.0, ...}.
+/// There are no values in the open interval (0, 1), so:
+/// fabs(...) < C where 0 < C <= 1.0 --> a == b (strict lt: C=1.0 ok)
+/// fabs(...) <= C where 0 < C < 1.0 --> a == b (le: C must be < 1.0,
+/// since le 1.0 is true when diff=1)
+///
+/// The same logic applies to sitofp.
+static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I) {
+ Value *FAbsArg;
+ if (!match(I.getOperand(0), m_FAbs(m_Value(FAbsArg))))
+ return nullptr;
+
+ const APFloat *C;
+ if (!match(I.getOperand(1), m_APFloat(C)))
+ return nullptr;
+
+ FCmpInst::Predicate Pred = I.getPredicate();
+ bool IsStrictLt = Pred == FCmpInst::FCMP_OLT || Pred == FCmpInst::FCMP_ULT;
+ bool IsLe = Pred == FCmpInst::FCMP_OLE || Pred == FCmpInst::FCMP_ULE;
+ if (!IsStrictLt && !IsLe)
+ return nullptr;
+
+ if (C->isNaN() || C->isZero() || C->isNegative())
+ return nullptr;
+
+ APFloat One = APFloat::getOne(C->getSemantics());
+ APFloat::cmpResult Cmp = C->compare(One);
+
+ // For strict-lt (olt/ult): C must be in (0, 1.0] -- C == 1.0 is fine since
+ // the next possible value after 0.0 is 1.0, and < 1.0 excludes it.
+ // For le (ole/ule): C must be in (0, 1.0) -- C == 1.0 is NOT valid since
+ // fabs(...) == 1.0 when a and b differ by 1, and <= 1.0 would be true.
+ if (IsStrictLt && Cmp == APFloat::cmpGreaterThan)
+ return nullptr;
+ if (IsLe && Cmp != APFloat::cmpLessThan)
+ return nullptr;
+
+ // Match: fsub(uitofp(A), uitofp(B)) where both casts are uitofp or sitofp
+ Value *A, *B;
+ if (!match(FAbsArg, m_FSub(m_UIToFP(m_Value(A)), m_UIToFP(m_Value(B)))) &&
+ !match(FAbsArg, m_FSub(m_SIToFP(m_Value(A)), m_SIToFP(m_Value(B)))))
+ return nullptr;
+
+ // A and B must have the same integer type
+ if (A->getType() != B->getType())
+ return nullptr;
+
+ // The int-to-fp cast must be exact (no precision loss).
+ // For uitofp: we need MantissaWidth >= IntWidth (all bits representable).
+ // For sitofp: we need MantissaWidth >= IntWidth (sign bit + magnitude).
+ // getFPMantissaWidth() returns the number of bits in the mantissa including
+ // the implicit leading 1 bit (i.e., the precision).
+ Type *FPTy = I.getOperand(0)->getType()->getScalarType();
+ int MantissaWidth = FPTy->getFPMantissaWidth();
+ if (MantissaWidth < 0)
+ return nullptr; // Unknown FP type.
+ unsigned IntWidth = A->getType()->getScalarSizeInBits();
+ // For unsigned: need MantissaWidth >= IntWidth
+ // For signed: need MantissaWidth >= IntWidth (to represent most negative val)
+ if ((unsigned)MantissaWidth < IntWidth)
+ return nullptr;
+
+ // fabs(uitofp(a) - uitofp(b)) < C (0 < C <= 1) --> a == b
+ return new ICmpInst(ICmpInst::ICMP_EQ, A, B);
+}
+
static Instruction *foldFCmpWithFloorAndCeil(FCmpInst &I,
InstCombinerImpl &IC) {
Value *LHS = I.getOperand(0), *RHS = I.getOperand(1);
@@ -9078,6 +9149,9 @@ Instruction *InstCombinerImpl::visitFCmpInst(FCmpInst &I) {
if (Instruction *R = foldFabsWithFcmpZero(I, *this))
return R;
+ if (Instruction *R = foldFCmpFAbsFSubIntToFP(I))
+ return R;
+
if (Instruction *R = foldSqrtWithFcmpZero(I, *this))
return R;
diff --git a/llvm/test/Transforms/InstCombine/fcmp.ll b/llvm/test/Transforms/InstCombine/fcmp.ll
index e3c43812cedd6..3b0135ca2ccfb 100644
--- a/llvm/test/Transforms/InstCombine/fcmp.ll
+++ b/llvm/test/Transforms/InstCombine/fcmp.ll
@@ -2,6 +2,7 @@
; RUN: opt -S -passes=instcombine < %s | FileCheck %s
declare half @llvm.fabs.f16(half)
+declare float @llvm.fabs.f32(float)
declare double @llvm.fabs.f64(double)
declare <2 x float> @llvm.fabs.v2f32(<2 x float>)
declare double @llvm.copysign.f64(double, double)
@@ -2581,3 +2582,140 @@ define i1 @fcmp_sqrt_zero_ult_nonzero(half %x) {
%cmp = fcmp ult half %sqrt, 1.000000e+00
ret i1 %cmp
}
+
+; fabs(uitofp(a) - uitofp(b)) < 1.0 --> a == b
+define i1 @fabs_uitofp_sub_olt_one(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_uitofp_sub_olt_one(
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i16 [[X:%.*]], [[Y:%.*]]
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = uitofp i16 %x to float
+ %fy = uitofp i16 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp olt float %abs, 1.0
+ ret i1 %cmp
+}
+
+; fabs(uitofp(a) - uitofp(b)) u< 1.0 --> a == b
+define i1 @fabs_uitofp_sub_ult_one(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_uitofp_sub_ult_one(
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i16 [[X:%.*]], [[Y:%.*]]
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = uitofp i16 %x to float
+ %fy = uitofp i16 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp ult float %abs, 1.0
+ ret i1 %cmp
+}
+
+; fabs(uitofp(a) - uitofp(b)) <= 0.5 --> a == b
+define i1 @fabs_uitofp_sub_ole_half(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_uitofp_sub_ole_half(
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i16 [[X:%.*]], [[Y:%.*]]
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = uitofp i16 %x to float
+ %fy = uitofp i16 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp ole float %abs, 0.5
+ ret i1 %cmp
+}
+
+; fabs(sitofp(a) - sitofp(b)) <= 0.5 --> a == b
+define i1 @fabs_sitofp_sub_ole_half(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_sitofp_sub_ole_half(
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i16 [[X:%.*]], [[Y:%.*]]
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = sitofp i16 %x to float
+ %fy = sitofp i16 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp ole float %abs, 0.5
+ ret i1 %cmp
+}
+
+
+define i1 @fabs_uitofp_sub_ule_half(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_uitofp_sub_ule_half(
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i16 [[X:%.*]], [[Y:%.*]]
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = uitofp i16 %x to float
+ %fy = uitofp i16 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp ule float %abs, 0.5
+ ret i1 %cmp
+}
+
+define i1 @fabs_sitofp_sub_olt_one(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_sitofp_sub_olt_one(
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i16 [[X:%.*]], [[Y:%.*]]
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = sitofp i16 %x to float
+ %fy = sitofp i16 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp olt float %abs, 1.0
+ ret i1 %cmp
+}
+
+
+; negative tests
+
+define i1 @fabs_uitofp_sub_ole_one_no_fold(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_uitofp_sub_ole_one_no_fold(
+; CHECK-NEXT: [[FX:%.*]] = uitofp i16 [[X:%.*]] to float
+; CHECK-NEXT: [[FY:%.*]] = uitofp i16 [[Y:%.*]] to float
+; CHECK-NEXT: [[SUB:%.*]] = fsub float [[FX]], [[FY]]
+; CHECK-NEXT: [[ABS:%.*]] = call float @llvm.fabs.f32(float [[SUB]])
+; CHECK-NEXT: [[CMP:%.*]] = fcmp ole float [[ABS]], 1.000000e+00
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = uitofp i16 %x to float
+ %fy = uitofp i16 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp ole float %abs, 1.0
+ ret i1 %cmp
+}
+
+define i1 @fabs_uitofp_sub_olt_two_no_fold(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_uitofp_sub_olt_two_no_fold(
+; CHECK-NEXT: [[FX:%.*]] = uitofp i16 [[X:%.*]] to float
+; CHECK-NEXT: [[FY:%.*]] = uitofp i16 [[Y:%.*]] to float
+; CHECK-NEXT: [[SUB:%.*]] = fsub float [[FX]], [[FY]]
+; CHECK-NEXT: [[ABS:%.*]] = call float @llvm.fabs.f32(float [[SUB]])
+; CHECK-NEXT: [[CMP:%.*]] = fcmp olt float [[ABS]], 2.000000e+00
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = uitofp i16 %x to float
+ %fy = uitofp i16 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp olt float %abs, 2.0
+ ret i1 %cmp
+}
+
+define i1 @fabs_sitofp_sub_olt_one_i32_no_fold(i32 %x, i32 %y) {
+; CHECK-LABEL: @fabs_sitofp_sub_olt_one_i32_no_fold(
+; CHECK-NEXT: [[FX:%.*]] = sitofp i32 [[X:%.*]] to float
+; CHECK-NEXT: [[FY:%.*]] = sitofp i32 [[Y:%.*]] to float
+; CHECK-NEXT: [[SUB:%.*]] = fsub float [[FX]], [[FY]]
+; CHECK-NEXT: [[ABS:%.*]] = call float @llvm.fabs.f32(float [[SUB]])
+; CHECK-NEXT: [[CMP:%.*]] = fcmp olt float [[ABS]], 1.000000e+00
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = sitofp i32 %x to float
+ %fy = sitofp i32 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp olt float %abs, 1.0
+ ret i1 %cmp
+}
>From 3f0945f5dfb2e09ae152d18db1ec8a3874ebb4f1 Mon Sep 17 00:00:00 2001
From: Shreeyash Pandey <shrpand at qti.qualcomm.com>
Date: Fri, 10 Apr 2026 03:21:56 -0700
Subject: [PATCH 2/7] clang format
Change-Id: If6ed06e0d075d59159d63ecfb06c608d718020a2
---
llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp | 3 ++-
1 file changed, 2 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
index dbd8c9a45ae03..87e544acfbb28 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
@@ -8774,7 +8774,8 @@ static Instruction *foldFCmpFSubIntoFCmp(FCmpInst &I, Instruction *LHSI,
/// There are no values in the open interval (0, 1), so:
/// fabs(...) < C where 0 < C <= 1.0 --> a == b (strict lt: C=1.0 ok)
/// fabs(...) <= C where 0 < C < 1.0 --> a == b (le: C must be < 1.0,
-/// since le 1.0 is true when diff=1)
+/// since le 1.0 is true when
+/// diff=1)
///
/// The same logic applies to sitofp.
static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I) {
>From 499e9357cca289a4903ff7205cc9ffeeaae93f61 Mon Sep 17 00:00:00 2001
From: Shreeyash Pandey <shrpand at qti.qualcomm.com>
Date: Tue, 14 Apr 2026 03:16:44 -0700
Subject: [PATCH 3/7] add corollary and its tests
Change-Id: If27f9bf242f3bf7c49c6735eca176ef9c2e36ad4
---
.../InstCombine/InstCombineCompares.cpp | 22 +-
llvm/test/Transforms/InstCombine/fcmp.ll | 217 +++---------------
2 files changed, 42 insertions(+), 197 deletions(-)
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
index 87e544acfbb28..116429bd25d88 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
@@ -8766,16 +8766,14 @@ static Instruction *foldFCmpFSubIntoFCmp(FCmpInst &I, Instruction *LHSI,
}
/// Fold: fabs(uitofp(a) - uitofp(b)) pred C --> a == b
-/// where 'pred' is olt, ole, ult, or ule, and C is a positive, Non-NaN float
+/// where 'pred' is olt, ult, ogt, ugt, oge or uge and C is a positive, Non-NaN float
/// when the uitofp casts are exact and C is in the valid range.
///
/// Since exact uitofp means distinct integers map to distinct floats, the only
/// values fabs(uitofp(a) - uitofp(b)) can take are {0.0, 1.0, 2.0, ...}.
/// There are no values in the open interval (0, 1), so:
/// fabs(...) < C where 0 < C <= 1.0 --> a == b (strict lt: C=1.0 ok)
-/// fabs(...) <= C where 0 < C < 1.0 --> a == b (le: C must be < 1.0,
-/// since le 1.0 is true when
-/// diff=1)
+// fabs(..) >= C where C >= 1.0 -> a != b
///
/// The same logic applies to sitofp.
static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I) {
@@ -8789,8 +8787,9 @@ static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I) {
FCmpInst::Predicate Pred = I.getPredicate();
bool IsStrictLt = Pred == FCmpInst::FCMP_OLT || Pred == FCmpInst::FCMP_ULT;
- bool IsLe = Pred == FCmpInst::FCMP_OLE || Pred == FCmpInst::FCMP_ULE;
- if (!IsStrictLt && !IsLe)
+ bool IsStrictGt = Pred == FCmpInst::FCMP_OGT || Pred == FCmpInst::FCMP_UGT;
+ bool IsGe = Pred == FCmpInst::FCMP_OGE || Pred == FCmpInst::FCMP_UGE;
+ if (!IsStrictLt && !IsStrictGt && !IsGe)
return nullptr;
if (C->isNaN() || C->isZero() || C->isNegative())
@@ -8801,11 +8800,11 @@ static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I) {
// For strict-lt (olt/ult): C must be in (0, 1.0] -- C == 1.0 is fine since
// the next possible value after 0.0 is 1.0, and < 1.0 excludes it.
- // For le (ole/ule): C must be in (0, 1.0) -- C == 1.0 is NOT valid since
- // fabs(...) == 1.0 when a and b differ by 1, and <= 1.0 would be true.
if (IsStrictLt && Cmp == APFloat::cmpGreaterThan)
return nullptr;
- if (IsLe && Cmp != APFloat::cmpLessThan)
+ if (IsGe && Cmp == APFloat::cmpGreaterThan)
+ return nullptr;
+ if (IsStrictGt && Cmp != APFloat::cmpLessThan)
return nullptr;
// Match: fsub(uitofp(A), uitofp(B)) where both casts are uitofp or sitofp
@@ -8833,8 +8832,9 @@ static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I) {
if ((unsigned)MantissaWidth < IntWidth)
return nullptr;
- // fabs(uitofp(a) - uitofp(b)) < C (0 < C <= 1) --> a == b
- return new ICmpInst(ICmpInst::ICMP_EQ, A, B);
+ ICmpInst::Predicate ResultPred =
+ (IsStrictLt) ? ICmpInst::ICMP_EQ : ICmpInst::ICMP_NE;
+ return new ICmpInst(ResultPred, A, B);
}
static Instruction *foldFCmpWithFloorAndCeil(FCmpInst &I,
diff --git a/llvm/test/Transforms/InstCombine/fcmp.ll b/llvm/test/Transforms/InstCombine/fcmp.ll
index 3b0135ca2ccfb..b7cb7fb5a3500 100644
--- a/llvm/test/Transforms/InstCombine/fcmp.ll
+++ b/llvm/test/Transforms/InstCombine/fcmp.ll
@@ -1813,145 +1813,6 @@ define i1 @fcmp_oeq_fsub_const(float %x, float %y) {
ret i1 %cmp
}
-define i1 @pr185561(i32 %arg0) {
-; CHECK-LABEL: @pr185561(
-; CHECK-NEXT: [[V0:%.*]] = add i32 [[ARG0:%.*]], -1
-; CHECK-NEXT: [[CMP:%.*]] = icmp sgt i32 [[V0]], 0
-; CHECK-NEXT: ret i1 [[CMP]]
-;
- %v0 = add i32 %arg0, -1
- %v1 = sitofp i32 %v0 to float
- %v2 = fsub float 1.000000e+00, %v1
- %v3 = fcmp olt float %v2, 1.000000e+00
- ret i1 %v3
-}
-
-define i1 @same_const_sub_sitofp_eq(i32 %x) {
-; CHECK-LABEL: @same_const_sub_sitofp_eq(
-; CHECK-NEXT: [[CMP:%.*]] = icmp eq i32 [[X:%.*]], 0
-; CHECK-NEXT: ret i1 [[CMP]]
-;
- %f = sitofp i32 %x to float
- %s = fsub float 1.000000e+00, %f
- %cmp = fcmp oeq float %s, 1.000000e+00
- ret i1 %cmp
-}
-
-define i1 @same_const_sub_uitofp_olt(i32 %x) {
-; CHECK-LABEL: @same_const_sub_uitofp_olt(
-; CHECK-NEXT: [[CMP:%.*]] = icmp ne i32 [[X:%.*]], 0
-; CHECK-NEXT: ret i1 [[CMP]]
-;
- %f = uitofp i32 %x to float
- %s = fsub float 2.000000e+00, %f
- %cmp = fcmp olt float %s, 2.000000e+00
- ret i1 %cmp
-}
-
-define i1 @same_const_sub_no_fold_large_c(i32 %x) {
-; CHECK-LABEL: @same_const_sub_no_fold_large_c(
-; CHECK-NOT: icmp
-; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to float
-; CHECK-NEXT: [[S:%.*]] = fsub float 0x417FFFFFE0000000, [[F]]
-; CHECK-NEXT: [[CMP:%.*]] = fcmp oeq float [[S]], 0x417FFFFFE0000000
-; CHECK-NEXT: ret i1 [[CMP]]
-;
- %f = sitofp i32 %x to float
- %s = fsub float 3.355443e+07, %f
- %cmp = fcmp oeq float %s, 3.355443e+07
- ret i1 %cmp
-}
-
-define <2 x i1> @same_const_sub_sitofp_vec_eq(<2 x i32> %x) {
-; CHECK-LABEL: @same_const_sub_sitofp_vec_eq(
-; CHECK-NEXT: [[CMP:%.*]] = icmp eq <2 x i32> [[X:%.*]], zeroinitializer
-; CHECK-NEXT: ret <2 x i1> [[CMP]]
-;
- %f = sitofp <2 x i32> %x to <2 x float>
- %s = fsub <2 x float> <float 1.000000e+00, float 1.000000e+00>, %f
- %cmp = fcmp oeq <2 x float> %s,
- <float 1.000000e+00, float 1.000000e+00>
- ret <2 x i1> %cmp
-}
-
-define <2 x i1> @same_const_sub_uitofp_vec_olt(<2 x i32> %x) {
-; CHECK-LABEL: @same_const_sub_uitofp_vec_olt(
-; CHECK-NEXT: [[CMP:%.*]] = icmp ne <2 x i32> [[X:%.*]], zeroinitializer
-; CHECK-NEXT: ret <2 x i1> [[CMP]]
-;
- %f = uitofp <2 x i32> %x to <2 x float>
- %s = fsub <2 x float> <float 2.000000e+00, float 2.000000e+00>, %f
- %cmp = fcmp olt <2 x float> %s,
- <float 2.000000e+00, float 2.000000e+00>
- ret <2 x i1> %cmp
-}
-
-define i1 @same_const_sub_no_fold_subnormal_c(i32 %x) {
-; CHECK-LABEL: @same_const_sub_no_fold_subnormal_c(
-; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to float
-; CHECK-NEXT: [[S:%.*]] = fsub float 0x36A0000000000000, [[F]]
-; CHECK-NEXT: [[CMP:%.*]] = fcmp olt float [[S]], 0x36A0000000000000
-; CHECK-NEXT: ret i1 [[CMP]]
-;
- %f = sitofp i32 %x to float
- %s = fsub float 0x36A0000000000000, %f
- %cmp = fcmp olt float %s, 0x36A0000000000000
- ret i1 %cmp
-}
-
-define i1 @same_const_sub_no_fold_wrong_mantissa_width(i32 %x) {
-; CHECK-LABEL: @same_const_sub_no_fold_wrong_mantissa_width(
-; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to float
-; CHECK-NEXT: [[S:%.*]] = fsub float 0x4180000000000000, [[F]]
-; CHECK-NEXT: [[CMP:%.*]] = fcmp oeq float [[S]], 0x4180000000000000
-; CHECK-NEXT: ret i1 [[CMP]]
-;
- %f = sitofp i32 %x to float
- %s = fsub float 3.3554432e+07, %f
- %cmp = fcmp oeq float %s, 3.3554432e+07
- ret i1 %cmp
-}
-
-define i1 @same_const_sub_sitofp_x86_fp80_eq(i32 %x) {
-; CHECK-LABEL: @same_const_sub_sitofp_x86_fp80_eq(
-; CHECK-NEXT: [[CMP:%.*]] = icmp eq i32 [[X:%.*]], 0
-; CHECK-NEXT: ret i1 [[CMP]]
-;
- %f = sitofp i32 %x to x86_fp80
- %s = fsub x86_fp80 0xK3FFF8000000000000000, %f
- %cmp = fcmp oeq x86_fp80 %s, 0xK3FFF8000000000000000
- ret i1 %cmp
-}
-
-define i1 @same_const_sub_no_fold_x86_fp80_large_c(i32 %x) {
-; CHECK-LABEL: @same_const_sub_no_fold_x86_fp80_large_c(
-; CHECK-NOT: icmp
-; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to x86_fp80
-; CHECK-NEXT: [[S:%.*]] = fsub x86_fp80 0xK403F8000000000000000, [[F]]
-; CHECK-NEXT: [[CMP:%.*]] = fcmp oeq x86_fp80 [[S]], 0xK403F8000000000000000
-; CHECK-NEXT: ret i1 [[CMP]]
-;
- %f = sitofp i32 %x to x86_fp80
- ; 2^64, so ilogb(C) == 64, which should fail `ilogb(C) < MantissaWidth`
- %s = fsub x86_fp80 0xK403F8000000000000000, %f
- %cmp = fcmp oeq x86_fp80 %s, 0xK403F8000000000000000
- ret i1 %cmp
-}
-
-define i1 @same_const_sub_no_fold_ppcfp128(i32 %x) {
-; CHECK-LABEL: @same_const_sub_no_fold_ppcfp128(
-; CHECK-NOT: icmp
-; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to ppc_fp128
-; CHECK-NEXT: [[S:%.*]] = fsub ppc_fp128 0xM3FF00000000000000000000000000000, [[F]]
-; CHECK-NEXT: [[CMP:%.*]] = fcmp oeq ppc_fp128 [[S]], 0xM3FF00000000000000000000000000000
-; CHECK-NEXT: ret i1 [[CMP]]
-;
- %f = sitofp i32 %x to ppc_fp128
- %s = fsub ppc_fp128 0xM3FF00000000000000000000000000000, %f
- %cmp = fcmp oeq ppc_fp128 %s, 0xM3FF00000000000000000000000000000
- ret i1 %cmp
-}
-
define i1 @fcmp_oge_fsub_const(float %x, float %y) {
; CHECK-LABEL: @fcmp_oge_fsub_const(
; CHECK-NEXT: [[FS:%.*]] = fsub float [[X:%.*]], [[Y:%.*]]
@@ -2611,23 +2472,8 @@ define i1 @fabs_uitofp_sub_ult_one(i16 %x, i16 %y) {
ret i1 %cmp
}
-; fabs(uitofp(a) - uitofp(b)) <= 0.5 --> a == b
-define i1 @fabs_uitofp_sub_ole_half(i16 %x, i16 %y) {
-; CHECK-LABEL: @fabs_uitofp_sub_ole_half(
-; CHECK-NEXT: [[CMP:%.*]] = icmp eq i16 [[X:%.*]], [[Y:%.*]]
-; CHECK-NEXT: ret i1 [[CMP]]
-;
- %fx = uitofp i16 %x to float
- %fy = uitofp i16 %y to float
- %sub = fsub float %fx, %fy
- %abs = call float @llvm.fabs.f32(float %sub)
- %cmp = fcmp ole float %abs, 0.5
- ret i1 %cmp
-}
-
-; fabs(sitofp(a) - sitofp(b)) <= 0.5 --> a == b
-define i1 @fabs_sitofp_sub_ole_half(i16 %x, i16 %y) {
-; CHECK-LABEL: @fabs_sitofp_sub_ole_half(
+define i1 @fabs_sitofp_sub_olt_one(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_sitofp_sub_olt_one(
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i16 [[X:%.*]], [[Y:%.*]]
; CHECK-NEXT: ret i1 [[CMP]]
;
@@ -2635,57 +2481,38 @@ define i1 @fabs_sitofp_sub_ole_half(i16 %x, i16 %y) {
%fy = sitofp i16 %y to float
%sub = fsub float %fx, %fy
%abs = call float @llvm.fabs.f32(float %sub)
- %cmp = fcmp ole float %abs, 0.5
+ %cmp = fcmp olt float %abs, 1.0
ret i1 %cmp
}
-
-define i1 @fabs_uitofp_sub_ule_half(i16 %x, i16 %y) {
-; CHECK-LABEL: @fabs_uitofp_sub_ule_half(
-; CHECK-NEXT: [[CMP:%.*]] = icmp eq i16 [[X:%.*]], [[Y:%.*]]
+define i1 @fabs_sitofp_sub_ogt_half(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_sitofp_sub_ogt_half(
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne i16 [[X:%.*]], [[Y:%.*]]
; CHECK-NEXT: ret i1 [[CMP]]
;
- %fx = uitofp i16 %x to float
- %fy = uitofp i16 %y to float
+ %fx = sitofp i16 %x to float
+ %fy = sitofp i16 %y to float
%sub = fsub float %fx, %fy
%abs = call float @llvm.fabs.f32(float %sub)
- %cmp = fcmp ule float %abs, 0.5
+ %cmp = fcmp ogt float %abs, 0.5
ret i1 %cmp
}
-define i1 @fabs_sitofp_sub_olt_one(i16 %x, i16 %y) {
-; CHECK-LABEL: @fabs_sitofp_sub_olt_one(
-; CHECK-NEXT: [[CMP:%.*]] = icmp eq i16 [[X:%.*]], [[Y:%.*]]
+define i1 @fabs_sitofp_sub_oge_one(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_sitofp_sub_oge_one(
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne i16 [[X:%.*]], [[Y:%.*]]
; CHECK-NEXT: ret i1 [[CMP]]
;
%fx = sitofp i16 %x to float
%fy = sitofp i16 %y to float
%sub = fsub float %fx, %fy
%abs = call float @llvm.fabs.f32(float %sub)
- %cmp = fcmp olt float %abs, 1.0
+ %cmp = fcmp oge float %abs, 1.0
ret i1 %cmp
}
-
; negative tests
-define i1 @fabs_uitofp_sub_ole_one_no_fold(i16 %x, i16 %y) {
-; CHECK-LABEL: @fabs_uitofp_sub_ole_one_no_fold(
-; CHECK-NEXT: [[FX:%.*]] = uitofp i16 [[X:%.*]] to float
-; CHECK-NEXT: [[FY:%.*]] = uitofp i16 [[Y:%.*]] to float
-; CHECK-NEXT: [[SUB:%.*]] = fsub float [[FX]], [[FY]]
-; CHECK-NEXT: [[ABS:%.*]] = call float @llvm.fabs.f32(float [[SUB]])
-; CHECK-NEXT: [[CMP:%.*]] = fcmp ole float [[ABS]], 1.000000e+00
-; CHECK-NEXT: ret i1 [[CMP]]
-;
- %fx = uitofp i16 %x to float
- %fy = uitofp i16 %y to float
- %sub = fsub float %fx, %fy
- %abs = call float @llvm.fabs.f32(float %sub)
- %cmp = fcmp ole float %abs, 1.0
- ret i1 %cmp
-}
-
define i1 @fabs_uitofp_sub_olt_two_no_fold(i16 %x, i16 %y) {
; CHECK-LABEL: @fabs_uitofp_sub_olt_two_no_fold(
; CHECK-NEXT: [[FX:%.*]] = uitofp i16 [[X:%.*]] to float
@@ -2719,3 +2546,21 @@ define i1 @fabs_sitofp_sub_olt_one_i32_no_fold(i32 %x, i32 %y) {
%cmp = fcmp olt float %abs, 1.0
ret i1 %cmp
}
+
+; For StrictGt, C ought to be strictly less than 1.0
+define i1 @fabs_sitofp_sub_ogt_one(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_sitofp_sub_ogt_one(
+; CHECK-NEXT: [[FX:%.*]] = sitofp i16 [[X:%.*]] to float
+; CHECK-NEXT: [[FY:%.*]] = sitofp i16 [[Y:%.*]] to float
+; CHECK-NEXT: [[SUB:%.*]] = fsub float [[FX]], [[FY]]
+; CHECK-NEXT: [[ABS:%.*]] = call float @llvm.fabs.f32(float [[SUB]])
+; CHECK-NEXT: [[CMP:%.*]] = fcmp ogt float [[ABS]], 1.000000e+00
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = sitofp i16 %x to float
+ %fy = sitofp i16 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp ogt float %abs, 1.0
+ ret i1 %cmp
+}
>From aed9ddcf1025ad0267e7df3b7660162f6d8d7461 Mon Sep 17 00:00:00 2001
From: Shreeyash Pandey <shrpand at qti.qualcomm.com>
Date: Tue, 14 Apr 2026 03:22:04 -0700
Subject: [PATCH 4/7] clang format
Change-Id: Id268918ab3132e2f3928ff901438712ebb8710be
---
.../lib/Transforms/InstCombine/InstCombineCompares.cpp | 10 +++++-----
1 file changed, 5 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
index 116429bd25d88..93ca06ba045d6 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
@@ -8766,14 +8766,14 @@ static Instruction *foldFCmpFSubIntoFCmp(FCmpInst &I, Instruction *LHSI,
}
/// Fold: fabs(uitofp(a) - uitofp(b)) pred C --> a == b
-/// where 'pred' is olt, ult, ogt, ugt, oge or uge and C is a positive, Non-NaN float
-/// when the uitofp casts are exact and C is in the valid range.
+/// where 'pred' is olt, ult, ogt, ugt, oge or uge and C is a positive, Non-NaN
+/// float when the uitofp casts are exact and C is in the valid range.
///
/// Since exact uitofp means distinct integers map to distinct floats, the only
/// values fabs(uitofp(a) - uitofp(b)) can take are {0.0, 1.0, 2.0, ...}.
/// There are no values in the open interval (0, 1), so:
/// fabs(...) < C where 0 < C <= 1.0 --> a == b (strict lt: C=1.0 ok)
-// fabs(..) >= C where C >= 1.0 -> a != b
+// fabs(..) >= C where C >= 1.0 -> a != b
///
/// The same logic applies to sitofp.
static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I) {
@@ -8788,7 +8788,7 @@ static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I) {
FCmpInst::Predicate Pred = I.getPredicate();
bool IsStrictLt = Pred == FCmpInst::FCMP_OLT || Pred == FCmpInst::FCMP_ULT;
bool IsStrictGt = Pred == FCmpInst::FCMP_OGT || Pred == FCmpInst::FCMP_UGT;
- bool IsGe = Pred == FCmpInst::FCMP_OGE || Pred == FCmpInst::FCMP_UGE;
+ bool IsGe = Pred == FCmpInst::FCMP_OGE || Pred == FCmpInst::FCMP_UGE;
if (!IsStrictLt && !IsStrictGt && !IsGe)
return nullptr;
@@ -8833,7 +8833,7 @@ static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I) {
return nullptr;
ICmpInst::Predicate ResultPred =
- (IsStrictLt) ? ICmpInst::ICMP_EQ : ICmpInst::ICMP_NE;
+ (IsStrictLt) ? ICmpInst::ICMP_EQ : ICmpInst::ICMP_NE;
return new ICmpInst(ResultPred, A, B);
}
>From 3bc1dc489f537313e73522523bb68f12c5ce8ce1 Mon Sep 17 00:00:00 2001
From: Shreeyash Pandey <shrpand at qti.qualcomm.com>
Date: Tue, 14 Apr 2026 23:18:07 -0700
Subject: [PATCH 5/7] add isKnownExactCast
Change-Id: I9b7b35775efec075f0fa4327ba483c6424c1e897
---
.../InstCombine/InstCombineCompares.cpp | 20 +++++--------------
1 file changed, 5 insertions(+), 15 deletions(-)
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
index 93ca06ba045d6..bbf5dc8edf86f 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
@@ -8776,7 +8776,7 @@ static Instruction *foldFCmpFSubIntoFCmp(FCmpInst &I, Instruction *LHSI,
// fabs(..) >= C where C >= 1.0 -> a != b
///
/// The same logic applies to sitofp.
-static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I) {
+static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I, InstCombinerImpl &IC) {
Value *FAbsArg;
if (!match(I.getOperand(0), m_FAbs(m_Value(FAbsArg))))
return nullptr;
@@ -8817,19 +8817,9 @@ static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I) {
if (A->getType() != B->getType())
return nullptr;
- // The int-to-fp cast must be exact (no precision loss).
- // For uitofp: we need MantissaWidth >= IntWidth (all bits representable).
- // For sitofp: we need MantissaWidth >= IntWidth (sign bit + magnitude).
- // getFPMantissaWidth() returns the number of bits in the mantissa including
- // the implicit leading 1 bit (i.e., the precision).
- Type *FPTy = I.getOperand(0)->getType()->getScalarType();
- int MantissaWidth = FPTy->getFPMantissaWidth();
- if (MantissaWidth < 0)
- return nullptr; // Unknown FP type.
- unsigned IntWidth = A->getType()->getScalarSizeInBits();
- // For unsigned: need MantissaWidth >= IntWidth
- // For signed: need MantissaWidth >= IntWidth (to represent most negative val)
- if ((unsigned)MantissaWidth < IntWidth)
+ Value *Src = I.getOperand(0);
+ auto *FPCast = cast<CastInst>(cast<Instruction>(FAbsArg)->getOperand(0));
+ if (!IC.isKnownExactCastIntToFP(*FPCast))
return nullptr;
ICmpInst::Predicate ResultPred =
@@ -9150,7 +9140,7 @@ Instruction *InstCombinerImpl::visitFCmpInst(FCmpInst &I) {
if (Instruction *R = foldFabsWithFcmpZero(I, *this))
return R;
- if (Instruction *R = foldFCmpFAbsFSubIntToFP(I))
+ if (Instruction *R = foldFCmpFAbsFSubIntToFP(I, *this))
return R;
if (Instruction *R = foldSqrtWithFcmpZero(I, *this))
>From e4025d9e4c0022548614b2b83280ee49086a52c7 Mon Sep 17 00:00:00 2001
From: Shreeyash Pandey <shrpand at qti.qualcomm.com>
Date: Tue, 14 Apr 2026 23:38:24 -0700
Subject: [PATCH 6/7] remove redundant code and add missing tests from fcmp
Change-Id: I766e2ea8524791c44254a47e2dafa3998e62c352
---
.../InstCombine/InstCombineCompares.cpp | 1 -
llvm/test/Transforms/InstCombine/fcmp.ll | 139 ++++++++++++++++++
2 files changed, 139 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
index bbf5dc8edf86f..282f2812cb515 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
@@ -8817,7 +8817,6 @@ static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I, InstCombinerImpl &IC) {
if (A->getType() != B->getType())
return nullptr;
- Value *Src = I.getOperand(0);
auto *FPCast = cast<CastInst>(cast<Instruction>(FAbsArg)->getOperand(0));
if (!IC.isKnownExactCastIntToFP(*FPCast))
return nullptr;
diff --git a/llvm/test/Transforms/InstCombine/fcmp.ll b/llvm/test/Transforms/InstCombine/fcmp.ll
index b7cb7fb5a3500..4708876489ea4 100644
--- a/llvm/test/Transforms/InstCombine/fcmp.ll
+++ b/llvm/test/Transforms/InstCombine/fcmp.ll
@@ -1813,6 +1813,145 @@ define i1 @fcmp_oeq_fsub_const(float %x, float %y) {
ret i1 %cmp
}
+define i1 @pr185561(i32 %arg0) {
+; CHECK-LABEL: @pr185561(
+; CHECK-NEXT: [[V0:%.*]] = add i32 [[ARG0:%.*]], -1
+; CHECK-NEXT: [[CMP:%.*]] = icmp sgt i32 [[V0]], 0
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %v0 = add i32 %arg0, -1
+ %v1 = sitofp i32 %v0 to float
+ %v2 = fsub float 1.000000e+00, %v1
+ %v3 = fcmp olt float %v2, 1.000000e+00
+ ret i1 %v3
+}
+
+define i1 @same_const_sub_sitofp_eq(i32 %x) {
+; CHECK-LABEL: @same_const_sub_sitofp_eq(
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i32 [[X:%.*]], 0
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %f = sitofp i32 %x to float
+ %s = fsub float 1.000000e+00, %f
+ %cmp = fcmp oeq float %s, 1.000000e+00
+ ret i1 %cmp
+}
+
+define i1 @same_const_sub_uitofp_olt(i32 %x) {
+; CHECK-LABEL: @same_const_sub_uitofp_olt(
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne i32 [[X:%.*]], 0
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %f = uitofp i32 %x to float
+ %s = fsub float 2.000000e+00, %f
+ %cmp = fcmp olt float %s, 2.000000e+00
+ ret i1 %cmp
+}
+
+define i1 @same_const_sub_no_fold_large_c(i32 %x) {
+; CHECK-LABEL: @same_const_sub_no_fold_large_c(
+; CHECK-NOT: icmp
+; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to float
+; CHECK-NEXT: [[S:%.*]] = fsub float 0x417FFFFFE0000000, [[F]]
+; CHECK-NEXT: [[CMP:%.*]] = fcmp oeq float [[S]], 0x417FFFFFE0000000
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %f = sitofp i32 %x to float
+ %s = fsub float 3.355443e+07, %f
+ %cmp = fcmp oeq float %s, 3.355443e+07
+ ret i1 %cmp
+}
+
+define <2 x i1> @same_const_sub_sitofp_vec_eq(<2 x i32> %x) {
+; CHECK-LABEL: @same_const_sub_sitofp_vec_eq(
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq <2 x i32> [[X:%.*]], zeroinitializer
+; CHECK-NEXT: ret <2 x i1> [[CMP]]
+;
+ %f = sitofp <2 x i32> %x to <2 x float>
+ %s = fsub <2 x float> <float 1.000000e+00, float 1.000000e+00>, %f
+ %cmp = fcmp oeq <2 x float> %s,
+ <float 1.000000e+00, float 1.000000e+00>
+ ret <2 x i1> %cmp
+}
+
+define <2 x i1> @same_const_sub_uitofp_vec_olt(<2 x i32> %x) {
+; CHECK-LABEL: @same_const_sub_uitofp_vec_olt(
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne <2 x i32> [[X:%.*]], zeroinitializer
+; CHECK-NEXT: ret <2 x i1> [[CMP]]
+;
+ %f = uitofp <2 x i32> %x to <2 x float>
+ %s = fsub <2 x float> <float 2.000000e+00, float 2.000000e+00>, %f
+ %cmp = fcmp olt <2 x float> %s,
+ <float 2.000000e+00, float 2.000000e+00>
+ ret <2 x i1> %cmp
+}
+
+define i1 @same_const_sub_no_fold_subnormal_c(i32 %x) {
+; CHECK-LABEL: @same_const_sub_no_fold_subnormal_c(
+; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to float
+; CHECK-NEXT: [[S:%.*]] = fsub float 0x36A0000000000000, [[F]]
+; CHECK-NEXT: [[CMP:%.*]] = fcmp olt float [[S]], 0x36A0000000000000
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %f = sitofp i32 %x to float
+ %s = fsub float 0x36A0000000000000, %f
+ %cmp = fcmp olt float %s, 0x36A0000000000000
+ ret i1 %cmp
+}
+
+define i1 @same_const_sub_no_fold_wrong_mantissa_width(i32 %x) {
+; CHECK-LABEL: @same_const_sub_no_fold_wrong_mantissa_width(
+; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to float
+; CHECK-NEXT: [[S:%.*]] = fsub float 0x4180000000000000, [[F]]
+; CHECK-NEXT: [[CMP:%.*]] = fcmp oeq float [[S]], 0x4180000000000000
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %f = sitofp i32 %x to float
+ %s = fsub float 3.3554432e+07, %f
+ %cmp = fcmp oeq float %s, 3.3554432e+07
+ ret i1 %cmp
+}
+
+define i1 @same_const_sub_sitofp_x86_fp80_eq(i32 %x) {
+; CHECK-LABEL: @same_const_sub_sitofp_x86_fp80_eq(
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i32 [[X:%.*]], 0
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %f = sitofp i32 %x to x86_fp80
+ %s = fsub x86_fp80 0xK3FFF8000000000000000, %f
+ %cmp = fcmp oeq x86_fp80 %s, 0xK3FFF8000000000000000
+ ret i1 %cmp
+}
+
+define i1 @same_const_sub_no_fold_x86_fp80_large_c(i32 %x) {
+; CHECK-LABEL: @same_const_sub_no_fold_x86_fp80_large_c(
+; CHECK-NOT: icmp
+; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to x86_fp80
+; CHECK-NEXT: [[S:%.*]] = fsub x86_fp80 0xK403F8000000000000000, [[F]]
+; CHECK-NEXT: [[CMP:%.*]] = fcmp oeq x86_fp80 [[S]], 0xK403F8000000000000000
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %f = sitofp i32 %x to x86_fp80
+ ; 2^64, so ilogb(C) == 64, which should fail `ilogb(C) < MantissaWidth`
+ %s = fsub x86_fp80 0xK403F8000000000000000, %f
+ %cmp = fcmp oeq x86_fp80 %s, 0xK403F8000000000000000
+ ret i1 %cmp
+}
+
+define i1 @same_const_sub_no_fold_ppcfp128(i32 %x) {
+; CHECK-LABEL: @same_const_sub_no_fold_ppcfp128(
+; CHECK-NOT: icmp
+; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to ppc_fp128
+; CHECK-NEXT: [[S:%.*]] = fsub ppc_fp128 0xM3FF00000000000000000000000000000, [[F]]
+; CHECK-NEXT: [[CMP:%.*]] = fcmp oeq ppc_fp128 [[S]], 0xM3FF00000000000000000000000000000
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %f = sitofp i32 %x to ppc_fp128
+ %s = fsub ppc_fp128 0xM3FF00000000000000000000000000000, %f
+ %cmp = fcmp oeq ppc_fp128 %s, 0xM3FF00000000000000000000000000000
+ ret i1 %cmp
+}
+
define i1 @fcmp_oge_fsub_const(float %x, float %y) {
; CHECK-LABEL: @fcmp_oge_fsub_const(
; CHECK-NEXT: [[FS:%.*]] = fsub float [[X:%.*]], [[Y:%.*]]
>From daf8b01ef3b703bad9d76307d83513be142d724b Mon Sep 17 00:00:00 2001
From: Shreeyash Pandey <shrpand at qti.qualcomm.com>
Date: Thu, 16 Apr 2026 03:28:54 -0700
Subject: [PATCH 7/7] use canBeCastedExactly and bring back ole/ule
Change-Id: I704152defcdb14ae8be242eb297ea6e85372e2f1
---
.../InstCombine/InstCombineCompares.cpp | 17 ++++++--
llvm/test/Transforms/InstCombine/fcmp.ll | 43 ++++++++++++++++---
2 files changed, 50 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
index 282f2812cb515..0b04fbd22b50c 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
@@ -8787,6 +8787,7 @@ static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I, InstCombinerImpl &IC) {
FCmpInst::Predicate Pred = I.getPredicate();
bool IsStrictLt = Pred == FCmpInst::FCMP_OLT || Pred == FCmpInst::FCMP_ULT;
+ bool IsLe = Pred == FCmpInst::FCMP_OLE || Pred == FCmpInst::FCMP_ULE;
bool IsStrictGt = Pred == FCmpInst::FCMP_OGT || Pred == FCmpInst::FCMP_UGT;
bool IsGe = Pred == FCmpInst::FCMP_OGE || Pred == FCmpInst::FCMP_UGE;
if (!IsStrictLt && !IsStrictGt && !IsGe)
@@ -8804,6 +8805,8 @@ static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I, InstCombinerImpl &IC) {
return nullptr;
if (IsGe && Cmp == APFloat::cmpGreaterThan)
return nullptr;
+ if (IsLe && Cmp == APFloat::cmpGreaterThan)
+ return nullptr;
if (IsStrictGt && Cmp != APFloat::cmpLessThan)
return nullptr;
@@ -8817,12 +8820,18 @@ static Instruction *foldFCmpFAbsFSubIntToFP(FCmpInst &I, InstCombinerImpl &IC) {
if (A->getType() != B->getType())
return nullptr;
- auto *FPCast = cast<CastInst>(cast<Instruction>(FAbsArg)->getOperand(0));
- if (!IC.isKnownExactCastIntToFP(*FPCast))
+ auto *FPCastOp0 = cast<CastInst>(cast<Instruction>(FAbsArg)->getOperand(0));
+ auto *FPCastOp1 = cast<CastInst>(cast<Instruction>(FAbsArg)->getOperand(1));
+ bool is_signed = FPCastOp0->getOpcode() == CastInst::SIToFP &&
+ FPCastOp1->getOpcode() == CastInst::SIToFP;
+ Type *FPTy = FPCastOp0->getType();
+ if (!IC.canBeCastedExactlyIntToFP(FPCastOp0->getOperand(0), FPTy, is_signed,
+ &I) ||
+ !IC.canBeCastedExactlyIntToFP(FPCastOp1->getOperand(0), FPTy, is_signed,
+ &I))
return nullptr;
-
ICmpInst::Predicate ResultPred =
- (IsStrictLt) ? ICmpInst::ICMP_EQ : ICmpInst::ICMP_NE;
+ IsStrictLt ? ICmpInst::ICMP_EQ : ICmpInst::ICMP_NE;
return new ICmpInst(ResultPred, A, B);
}
diff --git a/llvm/test/Transforms/InstCombine/fcmp.ll b/llvm/test/Transforms/InstCombine/fcmp.ll
index 4708876489ea4..a70d14b7fc627 100644
--- a/llvm/test/Transforms/InstCombine/fcmp.ll
+++ b/llvm/test/Transforms/InstCombine/fcmp.ll
@@ -1850,7 +1850,6 @@ define i1 @same_const_sub_uitofp_olt(i32 %x) {
define i1 @same_const_sub_no_fold_large_c(i32 %x) {
; CHECK-LABEL: @same_const_sub_no_fold_large_c(
-; CHECK-NOT: icmp
; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to float
; CHECK-NEXT: [[S:%.*]] = fsub float 0x417FFFFFE0000000, [[F]]
; CHECK-NEXT: [[CMP:%.*]] = fcmp oeq float [[S]], 0x417FFFFFE0000000
@@ -1870,7 +1869,7 @@ define <2 x i1> @same_const_sub_sitofp_vec_eq(<2 x i32> %x) {
%f = sitofp <2 x i32> %x to <2 x float>
%s = fsub <2 x float> <float 1.000000e+00, float 1.000000e+00>, %f
%cmp = fcmp oeq <2 x float> %s,
- <float 1.000000e+00, float 1.000000e+00>
+ <float 1.000000e+00, float 1.000000e+00>
ret <2 x i1> %cmp
}
@@ -1882,7 +1881,7 @@ define <2 x i1> @same_const_sub_uitofp_vec_olt(<2 x i32> %x) {
%f = uitofp <2 x i32> %x to <2 x float>
%s = fsub <2 x float> <float 2.000000e+00, float 2.000000e+00>, %f
%cmp = fcmp olt <2 x float> %s,
- <float 2.000000e+00, float 2.000000e+00>
+ <float 2.000000e+00, float 2.000000e+00>
ret <2 x i1> %cmp
}
@@ -1925,7 +1924,6 @@ define i1 @same_const_sub_sitofp_x86_fp80_eq(i32 %x) {
define i1 @same_const_sub_no_fold_x86_fp80_large_c(i32 %x) {
; CHECK-LABEL: @same_const_sub_no_fold_x86_fp80_large_c(
-; CHECK-NOT: icmp
; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to x86_fp80
; CHECK-NEXT: [[S:%.*]] = fsub x86_fp80 0xK403F8000000000000000, [[F]]
; CHECK-NEXT: [[CMP:%.*]] = fcmp oeq x86_fp80 [[S]], 0xK403F8000000000000000
@@ -1940,7 +1938,6 @@ define i1 @same_const_sub_no_fold_x86_fp80_large_c(i32 %x) {
define i1 @same_const_sub_no_fold_ppcfp128(i32 %x) {
; CHECK-LABEL: @same_const_sub_no_fold_ppcfp128(
-; CHECK-NOT: icmp
; CHECK-NEXT: [[F:%.*]] = sitofp i32 [[X:%.*]] to ppc_fp128
; CHECK-NEXT: [[S:%.*]] = fsub ppc_fp128 0xM3FF00000000000000000000000000000, [[F]]
; CHECK-NEXT: [[CMP:%.*]] = fcmp oeq ppc_fp128 [[S]], 0xM3FF00000000000000000000000000000
@@ -2597,7 +2594,7 @@ define i1 @fabs_uitofp_sub_olt_one(i16 %x, i16 %y) {
ret i1 %cmp
}
-; fabs(uitofp(a) - uitofp(b)) u< 1.0 --> a == b
+; fabs(uitofp(a) - uitofp(b)) < 1.0 --> a == b
define i1 @fabs_uitofp_sub_ult_one(i16 %x, i16 %y) {
; CHECK-LABEL: @fabs_uitofp_sub_ult_one(
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i16 [[X:%.*]], [[Y:%.*]]
@@ -2624,6 +2621,40 @@ define i1 @fabs_sitofp_sub_olt_one(i16 %x, i16 %y) {
ret i1 %cmp
}
+define i1 @fabs_uitofp_sub_ule_one(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_uitofp_sub_ule_one(
+; CHECK-NEXT: [[FX:%.*]] = uitofp i16 [[X:%.*]] to float
+; CHECK-NEXT: [[FY:%.*]] = uitofp i16 [[Y:%.*]] to float
+; CHECK-NEXT: [[SUB:%.*]] = fsub float [[FX]], [[FY]]
+; CHECK-NEXT: [[ABS:%.*]] = call float @llvm.fabs.f32(float [[SUB]])
+; CHECK-NEXT: [[CMP:%.*]] = fcmp ule float [[ABS]], 1.000000e+00
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = uitofp i16 %x to float
+ %fy = uitofp i16 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp ule float %abs, 1.0
+ ret i1 %cmp
+}
+
+define i1 @fabs_sitofp_sub_ole_one(i16 %x, i16 %y) {
+; CHECK-LABEL: @fabs_sitofp_sub_ole_one(
+; CHECK-NEXT: [[FX:%.*]] = sitofp i16 [[X:%.*]] to float
+; CHECK-NEXT: [[FY:%.*]] = sitofp i16 [[Y:%.*]] to float
+; CHECK-NEXT: [[SUB:%.*]] = fsub float [[FX]], [[FY]]
+; CHECK-NEXT: [[ABS:%.*]] = call float @llvm.fabs.f32(float [[SUB]])
+; CHECK-NEXT: [[CMP:%.*]] = fcmp ole float [[ABS]], 1.000000e+00
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %fx = sitofp i16 %x to float
+ %fy = sitofp i16 %y to float
+ %sub = fsub float %fx, %fy
+ %abs = call float @llvm.fabs.f32(float %sub)
+ %cmp = fcmp ole float %abs, 1.0
+ ret i1 %cmp
+}
+
define i1 @fabs_sitofp_sub_ogt_half(i16 %x, i16 %y) {
; CHECK-LABEL: @fabs_sitofp_sub_ogt_half(
; CHECK-NEXT: [[CMP:%.*]] = icmp ne i16 [[X:%.*]], [[Y:%.*]]
More information about the llvm-commits
mailing list