[llvm] [SelectionDAG] Fix fcmp fold for new min/max semantics (PR #223655)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 15 04:09:35 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Mitch Briles (MitchBriles)
<details>
<summary>Changes</summary>
Fixes #<!-- -->223089
**Changes**
- Updated the comments in `ISDOpcodes.h` with the new semantics for floating-point min/max operations.
- Fixed the following fold to use the correct min/max opcodes and correctly handle NaNs. This fold can now use `FMINIMUMNUM` and `FMAXIMUMNUM`.
```
(A < B) | (C < B) -> min(A, C) < B
(A < B) & (C < B) -> max(A, C) < B
```
**Note about X86**
X86 marks `ISD::FMAXIMUMNUM` and `ISD::FMINIMUMNUM` as custom, but the lowering is quite expensive for this fold, so I used `isProfitableToCombineMinNumMaxNum` to prevent regressions.
cc @<!-- -->arsenm @<!-- -->topperc
---
Full diff: https://github.com/llvm/llvm-project/pull/223655.diff
6 Files Affected:
- (modified) llvm/include/llvm/CodeGen/ISDOpcodes.h (+34-28)
- (modified) llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp (+50-45)
- (modified) llvm/test/CodeGen/AArch64/combine_andor_with_cmps.ll (+42-4)
- (modified) llvm/test/CodeGen/AMDGPU/combine_andor_with_cmps_nnan.ll (+7-6)
- (modified) llvm/test/CodeGen/AMDGPU/or.r600.ll (+15-10)
- (added) llvm/test/CodeGen/RISCV/combine-andor-with-cmps.ll (+83)
``````````diff
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index 9c3757203299e..5e38c7ccfbb3d 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -1085,47 +1085,53 @@ enum NodeType {
LRINT,
LLRINT,
- /// FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values,
- /// following IEEE-754 definitions except for signed zero behavior.
+ /// FMINNUM/FMAXNUM - NaN-discarding minimum/maximum: if one operand is a
+ /// quiet NaN and the other is a number, returns the number.
///
- /// If one input is a signaling NaN, returns a quiet NaN. This matches
- /// IEEE-754 2008's minNum/maxNum behavior for signaling NaNs (which differs
- /// from 2019).
+ /// If an operand is a signaling NaN, this will non-deterministically either:
+ /// - Return a NaN.
+ /// - Or treat the signaling NaN as a quiet NaN.
///
/// These treat -0 as ordered less than +0, matching the behavior of IEEE-754
- /// 2019's minimumNumber/maximumNumber.
- ///
- /// Note that that arithmetic on an sNaN doesn't consistently produce a qNaN,
- /// so arithmetic feeding into a minnum/maxnum can produce inconsistent
- /// results. FMAXIMUN/FMINIMUM or FMAXIMUMNUM/FMINIMUMNUM may be better choice
- /// for non-distinction of sNaN/qNaN handling.
+ /// 2019's minimumNumber/maximumNumber. With the nsz flag, one +0.0 and one
+ /// -0.0 operand may non-deterministically return either operand; contrary to
+ /// normal nsz semantics, if both operands have the same sign, so must the
+ /// result. Note that not all backends respect this ordering yet.
FMINNUM,
FMAXNUM,
- /// FMINNUM_IEEE/FMAXNUM_IEEE - Perform floating-point minimumNumber or
- /// maximumNumber on two values, following IEEE-754 definitions. This differs
- /// from FMINNUM/FMAXNUM in the handling of signaling NaNs, and signed zero.
- ///
- /// If one input is a signaling NaN, returns a quiet NaN. This matches
- /// IEEE-754 2008's minnum/maxnum behavior for signaling NaNs (which differs
- /// from 2019).
- ///
- /// These treat -0 as ordered less than +0, matching the behavior of IEEE-754
- /// 2019's minimumNumber/maximumNumber.
+ /// FMINNUM_IEEE/FMAXNUM_IEEE - Same as FMINNUM/FMAXNUM, except that a
+ /// signaling NaN operand deterministically returns a quiet NaN, matching/for
+ /// IEEE-754 2008's minNum/maxNum. Signed zeros are ordered identically to
+ /// FMINNUM/FMAXNUM: -0 is less than +0, relaxed by the nsz flag.
///
- /// Deprecated, and will be removed soon, as FMINNUM/FMAXNUM have the same
- /// semantics now.
+ /// Deprecated, and will be removed soon: this is a legal implementation of
+ /// FMINNUM/FMAXNUM, so targets should select those instead.
FMINNUM_IEEE,
FMAXNUM_IEEE,
- /// FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0
- /// as less than 0.0. While FMINNUM_IEEE/FMAXNUM_IEEE follow IEEE 754-2008
- /// semantics, FMINIMUM/FMAXIMUM follow IEEE 754-2019 semantics.
+ /// FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum: if either operand is
+ /// a NaN, returns a NaN. Follows C23's fminimum/fmaximum and IEEE-754 2019's
+ /// minimum/maximum, except that a signaling NaN operand is not guaranteed to
+ /// be quieted.
+ ///
+ /// These treat -0 as ordered less than +0. With the nsz flag, one +0.0 and
+ /// one -0.0 operand may non-deterministically return either operand;
+ /// contrary to normal nsz semantics, if both operands have the same sign, so
+ /// must the result.
FMINIMUM,
FMAXIMUM,
- /// FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with
- /// FMINNUM_IEEE and FMAXNUM_IEEE besides if either operand is sNaN.
+ /// FMINIMUMNUM/FMAXIMUMNUM - NaN-discarding minimum/maximum: if one operand
+ /// is a NaN and the other is a number, returns the number. Follows C23's
+ /// fminimum_num/fmaximum_num and IEEE-754 2019's minimumNumber/maximumNumber,
+ /// except that a signaling NaN operand is not guaranteed to be quieted.
+ /// Same as FMINNUM/FMAXNUM, but treats signaling NaNs as quiet NaNs.
+ ///
+ /// These treat -0 as ordered less than +0. With the nsz flag, one +0.0 and
+ /// one -0.0 operand may non-deterministically return either operand;
+ /// contrary to normal nsz semantics, if both operands have the same sign, so
+ /// must the result.
FMINIMUMNUM,
FMAXIMUMNUM,
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index a829d7a34d5a6..a84127e453b11 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -6935,61 +6935,57 @@ static unsigned getMinMaxOpcodeForClamp(bool IsMin, SDValue Operand1,
return ISD::DELETED_NODE;
}
-// FIXME: use FMINIMUMNUM if possible, such as for RISC-V.
static unsigned getMinMaxOpcodeForCompareFold(
SDValue Operand1, SDValue Operand2, bool SetCCNoNaNs, ISD::CondCode CC,
unsigned OrAndOpcode, SelectionDAG &DAG, bool isFMAXNUMFMINNUM_IEEE,
- bool isFMAXNUMFMINNUM) {
+ bool isFMAXNUMFMINNUM, bool isFMAXIMUMNUMFMINIMUMNUM) {
// The optimization cannot be applied for all the predicates because
- // of the way FMINNUM/FMAXNUM and FMINNUM_IEEE/FMAXNUM_IEEE handle
- // NaNs. For FMINNUM_IEEE/FMAXNUM_IEEE, the optimization cannot be
- // applied at all if one of the operands is a signaling NaN.
+ // of the way the min/max opcodes handle NaNs.
- // It is safe to use FMINNUM_IEEE/FMAXNUM_IEEE if all the operands
- // are non NaN values.
+ // It is safe to use FMINIMUMNUM/FMAXIMUMNUM or FMINNUM_IEEE/FMAXNUM_IEEE if
+ // all the operands are non NaN values.
if (((CC == ISD::SETLT || CC == ISD::SETLE) && (OrAndOpcode == ISD::OR)) ||
((CC == ISD::SETGT || CC == ISD::SETGE) && (OrAndOpcode == ISD::AND))) {
- return (SetCCNoNaNs || arebothOperandsNotNan(Operand1, Operand2, DAG)) &&
- isFMAXNUMFMINNUM_IEEE
- ? ISD::FMINNUM_IEEE
- : ISD::DELETED_NODE;
+ if (!SetCCNoNaNs && !arebothOperandsNotNan(Operand1, Operand2, DAG))
+ return ISD::DELETED_NODE;
+ return isFMAXIMUMNUMFMINIMUMNUM ? ISD::FMINIMUMNUM
+ : isFMAXNUMFMINNUM_IEEE ? ISD::FMINNUM_IEEE
+ : ISD::DELETED_NODE;
}
if (((CC == ISD::SETGT || CC == ISD::SETGE) && (OrAndOpcode == ISD::OR)) ||
((CC == ISD::SETLT || CC == ISD::SETLE) && (OrAndOpcode == ISD::AND))) {
- return (SetCCNoNaNs || arebothOperandsNotNan(Operand1, Operand2, DAG)) &&
- isFMAXNUMFMINNUM_IEEE
- ? ISD::FMAXNUM_IEEE
- : ISD::DELETED_NODE;
- }
-
- // Both FMINNUM/FMAXNUM and FMINNUM_IEEE/FMAXNUM_IEEE handle quiet
- // NaNs in the same way. But, FMINNUM/FMAXNUM and FMINNUM_IEEE/
- // FMAXNUM_IEEE handle signaling NaNs differently. If we cannot prove
- // that there are not any sNaNs, then the optimization is not valid
- // for FMINNUM_IEEE/FMAXNUM_IEEE. In the presence of sNaNs, we apply
- // the optimization using FMINNUM/FMAXNUM for the following cases. If
- // we can prove that we do not have any sNaNs, then we can do the
- // optimization using FMINNUM_IEEE/FMAXNUM_IEEE for the following
- // cases.
- if (((CC == ISD::SETOLT || CC == ISD::SETOLE) && (OrAndOpcode == ISD::OR)) ||
- ((CC == ISD::SETUGT || CC == ISD::SETUGE) && (OrAndOpcode == ISD::AND))) {
- return isFMAXNUMFMINNUM ? ISD::FMINNUM
- : arebothOperandsNotSNan(Operand1, Operand2, DAG) &&
- isFMAXNUMFMINNUM_IEEE
- ? ISD::FMINNUM_IEEE
- : ISD::DELETED_NODE;
- }
-
- if (((CC == ISD::SETOGT || CC == ISD::SETOGE) && (OrAndOpcode == ISD::OR)) ||
- ((CC == ISD::SETULT || CC == ISD::SETULE) && (OrAndOpcode == ISD::AND))) {
- return isFMAXNUMFMINNUM ? ISD::FMAXNUM
- : arebothOperandsNotSNan(Operand1, Operand2, DAG) &&
- isFMAXNUMFMINNUM_IEEE
- ? ISD::FMAXNUM_IEEE
- : ISD::DELETED_NODE;
+ if (!SetCCNoNaNs && !arebothOperandsNotNan(Operand1, Operand2, DAG))
+ return ISD::DELETED_NODE;
+ return isFMAXIMUMNUMFMINIMUMNUM ? ISD::FMAXIMUMNUM
+ : isFMAXNUMFMINNUM_IEEE ? ISD::FMAXNUM_IEEE
+ : ISD::DELETED_NODE;
}
+ bool IsMin;
+ if (((CC == ISD::SETOLT || CC == ISD::SETOLE) && (OrAndOpcode == ISD::OR)) ||
+ ((CC == ISD::SETUGT || CC == ISD::SETUGE) && (OrAndOpcode == ISD::AND)))
+ IsMin = true;
+ else if (((CC == ISD::SETOGT || CC == ISD::SETOGE) &&
+ (OrAndOpcode == ISD::OR)) ||
+ ((CC == ISD::SETULT || CC == ISD::SETULE) &&
+ (OrAndOpcode == ISD::AND)))
+ IsMin = false;
+ else
+ return ISD::DELETED_NODE;
+
+ // For the above predicates, the optimization is valid if a NaN operand is
+ // discarded. FMINIMUMNUM/FMAXIMUMNUM always do so. FMINNUM/FMAXNUM and
+ // FMINNUM_IEEE/FMAXNUM_IEEE only do so for quiet NaNs, as a signaling NaN
+ // operand may instead produce a NaN.
+ if (isFMAXIMUMNUMFMINIMUMNUM)
+ return IsMin ? ISD::FMINIMUMNUM : ISD::FMAXIMUMNUM;
+ if (!arebothOperandsNotSNan(Operand1, Operand2, DAG))
+ return ISD::DELETED_NODE;
+ if (isFMAXNUMFMINNUM)
+ return IsMin ? ISD::FMINNUM : ISD::FMAXNUM;
+ if (isFMAXNUMFMINNUM_IEEE)
+ return IsMin ? ISD::FMINNUM_IEEE : ISD::FMAXNUM_IEEE;
return ISD::DELETED_NODE;
}
@@ -7040,12 +7036,20 @@ static SDValue foldAndOrOfSETCC(SDNode *LogicOp, SelectionDAG &DAG) {
TLI.isOperationLegal(ISD::FMINNUM_IEEE, OpVT);
bool isFMAXNUMFMINNUM = TLI.isOperationLegalOrCustom(ISD::FMAXNUM, OpVT) &&
TLI.isOperationLegalOrCustom(ISD::FMINNUM, OpVT);
+ // A custom FMINIMUMNUM/FMAXIMUMNUM may be more expensive than the compares,
+ // so only use one if the target considers forming it profitable.
+ bool isFMAXIMUMNUMFMINIMUMNUM =
+ (TLI.isOperationLegal(ISD::FMAXIMUMNUM, OpVT) &&
+ TLI.isOperationLegal(ISD::FMINIMUMNUM, OpVT)) ||
+ (TLI.isOperationLegalOrCustom(ISD::FMAXIMUMNUM, OpVT) &&
+ TLI.isOperationLegalOrCustom(ISD::FMINIMUMNUM, OpVT) &&
+ TLI.isProfitableToCombineMinNumMaxNum(OpVT));
if (((OpVT.isInteger() && TLI.isOperationLegal(ISD::UMAX, OpVT) &&
TLI.isOperationLegal(ISD::SMAX, OpVT) &&
TLI.isOperationLegal(ISD::UMIN, OpVT) &&
TLI.isOperationLegal(ISD::SMIN, OpVT)) ||
- (OpVT.isFloatingPoint() &&
- (isFMAXNUMFMINNUM_IEEE || isFMAXNUMFMINNUM))) &&
+ (OpVT.isFloatingPoint() && (isFMAXNUMFMINNUM_IEEE || isFMAXNUMFMINNUM ||
+ isFMAXIMUMNUMFMINIMUMNUM))) &&
!ISD::isIntEqualitySetCC(CCL) && !ISD::isFPEqualitySetCC(CCL) &&
CCL != ISD::SETFALSE && CCL != ISD::SETO && CCL != ISD::SETUO &&
CCL != ISD::SETTRUE &&
@@ -7102,7 +7106,8 @@ static SDValue foldAndOrOfSETCC(SDNode *LogicOp, SelectionDAG &DAG) {
NewOpcode = getMinMaxOpcodeForCompareFold(
Operand1, Operand2,
LHSSetCCFlags.hasNoNaNs() && RHSSetCCFlags.hasNoNaNs(), CC,
- LogicOp->getOpcode(), DAG, isFMAXNUMFMINNUM_IEEE, isFMAXNUMFMINNUM);
+ LogicOp->getOpcode(), DAG, isFMAXNUMFMINNUM_IEEE, isFMAXNUMFMINNUM,
+ isFMAXIMUMNUMFMINIMUMNUM);
if (NewOpcode != ISD::DELETED_NODE) {
// Propagate fast-math flags from setcc.
diff --git a/llvm/test/CodeGen/AArch64/combine_andor_with_cmps.ll b/llvm/test/CodeGen/AArch64/combine_andor_with_cmps.ll
index 89cb25f3d9d75..a1eaac062e6b4 100644
--- a/llvm/test/CodeGen/AArch64/combine_andor_with_cmps.ll
+++ b/llvm/test/CodeGen/AArch64/combine_andor_with_cmps.ll
@@ -5,12 +5,14 @@
; CMP(A,C)||CMP(B,C) => CMP(MIN/MAX(A,B), C)
; CMP(A,C)&&CMP(B,C) => CMP(MIN/MAX(A,B), C)
+; Not folded: A or B may be a signaling NaN, for which MIN/MAX may return NaN.
define i1 @test1(float %arg1, float %arg2, float %arg3) #0 {
; CHECK-LABEL: test1:
; CHECK: // %bb.0:
-; CHECK-NEXT: fminnm s0, s0, s1
; CHECK-NEXT: fcmp s0, s2
-; CHECK-NEXT: cset w0, mi
+; CHECK-NEXT: cset w8, mi
+; CHECK-NEXT: fcmp s1, s2
+; CHECK-NEXT: csinc w0, w8, wzr, pl
; CHECK-NEXT: ret
%cmp1 = fcmp olt float %arg1, %arg3
%cmp2 = fcmp olt float %arg2, %arg3
@@ -21,9 +23,10 @@ define i1 @test1(float %arg1, float %arg2, float %arg3) #0 {
define i1 @test2(double %arg1, double %arg2, double %arg3) #0 {
; CHECK-LABEL: test2:
; CHECK: // %bb.0:
-; CHECK-NEXT: fmaxnm d0, d0, d1
; CHECK-NEXT: fcmp d0, d2
-; CHECK-NEXT: cset w0, gt
+; CHECK-NEXT: cset w8, gt
+; CHECK-NEXT: fcmp d1, d2
+; CHECK-NEXT: csinc w0, w8, wzr, le
; CHECK-NEXT: ret
%cmp1 = fcmp ogt double %arg1, %arg3
%cmp2 = fcmp ogt double %arg2, %arg3
@@ -69,3 +72,38 @@ define i1 @test4(float %arg1, float %arg2, float %arg3) {
ret i1 %or1
}
+; A and B cannot be signaling NaNs, so the folds happen.
+define i1 @test5(i32 %arg1, i32 %arg2, float %arg3) {
+; CHECK-LABEL: test5:
+; CHECK: // %bb.0:
+; CHECK-NEXT: scvtf s1, w0
+; CHECK-NEXT: scvtf s2, w1
+; CHECK-NEXT: fminnm s1, s1, s2
+; CHECK-NEXT: fcmp s1, s0
+; CHECK-NEXT: cset w0, mi
+; CHECK-NEXT: ret
+ %conv1 = sitofp i32 %arg1 to float
+ %conv2 = sitofp i32 %arg2 to float
+ %cmp1 = fcmp olt float %conv1, %arg3
+ %cmp2 = fcmp olt float %conv2, %arg3
+ %or1 = or i1 %cmp1, %cmp2
+ ret i1 %or1
+}
+
+define i1 @test6(i32 %arg1, i32 %arg2, double %arg3) {
+; CHECK-LABEL: test6:
+; CHECK: // %bb.0:
+; CHECK-NEXT: scvtf d1, w0
+; CHECK-NEXT: scvtf d2, w1
+; CHECK-NEXT: fmaxnm d1, d1, d2
+; CHECK-NEXT: fcmp d1, d0
+; CHECK-NEXT: cset w0, lt
+; CHECK-NEXT: ret
+ %conv1 = sitofp i32 %arg1 to double
+ %conv2 = sitofp i32 %arg2 to double
+ %cmp1 = fcmp ult double %conv1, %arg3
+ %cmp2 = fcmp ult double %conv2, %arg3
+ %and1 = and i1 %cmp1, %cmp2
+ ret i1 %and1
+}
+
diff --git a/llvm/test/CodeGen/AMDGPU/combine_andor_with_cmps_nnan.ll b/llvm/test/CodeGen/AMDGPU/combine_andor_with_cmps_nnan.ll
index 550cfec413d79..87ac3b4c73d6d 100644
--- a/llvm/test/CodeGen/AMDGPU/combine_andor_with_cmps_nnan.ll
+++ b/llvm/test/CodeGen/AMDGPU/combine_andor_with_cmps_nnan.ll
@@ -954,15 +954,16 @@ define i1 @test117(float %arg1, float %arg2, float %arg3, float %arg4, float %ar
; GCN-LABEL: test117:
; GCN: ; %bb.0:
; GCN-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GCN-NEXT: v_min3_f32 v4, v4, v5, v6
; GCN-NEXT: v_min_f32_e32 v0, v0, v1
-; GCN-NEXT: v_min3_f32 v1, v8, v9, v10
-; GCN-NEXT: v_min_f32_e32 v2, v2, v3
-; GCN-NEXT: v_min_f32_e32 v3, v4, v7
+; GCN-NEXT: v_min3_f32 v1, v4, v5, v6
+; GCN-NEXT: v_max_f32_e32 v4, v7, v7
+; GCN-NEXT: v_min3_f32 v5, v8, v9, v10
+; GCN-NEXT: v_max_f32_e32 v6, v11, v11
+; GCN-NEXT: v_dual_min_f32 v2, v2, v3 :: v_dual_min_f32 v1, v1, v4
; GCN-NEXT: v_cmp_lt_f32_e32 vcc_lo, v0, v12
-; GCN-NEXT: v_min_f32_e32 v0, v1, v11
+; GCN-NEXT: v_min_f32_e32 v0, v5, v6
; GCN-NEXT: v_cmp_lt_f32_e64 s0, v2, v13
-; GCN-NEXT: v_cmp_lt_f32_e64 s1, v3, v13
+; GCN-NEXT: v_cmp_lt_f32_e64 s1, v1, v13
; GCN-NEXT: v_cmp_lt_f32_e64 s2, v0, v12
; GCN-NEXT: s_or_b32 s0, vcc_lo, s0
; GCN-NEXT: s_or_b32 s0, s0, s1
diff --git a/llvm/test/CodeGen/AMDGPU/or.r600.ll b/llvm/test/CodeGen/AMDGPU/or.r600.ll
index ed9d0085fd82a..2b06ffaaf387c 100644
--- a/llvm/test/CodeGen/AMDGPU/or.r600.ll
+++ b/llvm/test/CodeGen/AMDGPU/or.r600.ll
@@ -456,21 +456,26 @@ define amdgpu_kernel void @trunc_i64_or_to_i32(ptr addrspace(1) %out, [8 x i32],
define amdgpu_kernel void @or_i1(ptr addrspace(1) %out, ptr addrspace(1) %in0, ptr addrspace(1) %in1) {
; EG-LABEL: or_i1:
; EG: ; %bb.0:
-; EG-NEXT: ALU 1, @10, KC0[CB0:0-32], KC1[]
-; EG-NEXT: TEX 1 @6
-; EG-NEXT: ALU 4, @12, KC0[CB0:0-32], KC1[]
+; EG-NEXT: ALU 0, @12, KC0[CB0:0-32], KC1[]
+; EG-NEXT: TEX 0 @8
+; EG-NEXT: ALU 0, @13, KC0[CB0:0-32], KC1[]
+; EG-NEXT: TEX 0 @10
+; EG-NEXT: ALU 5, @14, KC0[CB0:0-32], KC1[]
; EG-NEXT: MEM_RAT_CACHELESS STORE_RAW T0.X, T1.X, 1
; EG-NEXT: CF_END
; EG-NEXT: PAD
-; EG-NEXT: Fetch clause starting at 6:
-; EG-NEXT: VTX_READ_32 T1.X, T1.X, 0, #1
+; EG-NEXT: Fetch clause starting at 8:
; EG-NEXT: VTX_READ_32 T0.X, T0.X, 0, #1
-; EG-NEXT: ALU clause starting at 10:
-; EG-NEXT: MOV T0.X, KC0[2].Z,
-; EG-NEXT: MOV * T1.X, KC0[2].W,
+; EG-NEXT: Fetch clause starting at 10:
+; EG-NEXT: VTX_READ_32 T1.X, T1.X, 0, #1
; EG-NEXT: ALU clause starting at 12:
-; EG-NEXT: MAX_DX10 * T0.W, T0.X, T1.X,
-; EG-NEXT: SETGE_DX10 * T0.W, PV.W, 0.0,
+; EG-NEXT: MOV * T0.X, KC0[2].W,
+; EG-NEXT: ALU clause starting at 13:
+; EG-NEXT: MOV * T1.X, KC0[2].Z,
+; EG-NEXT: ALU clause starting at 14:
+; EG-NEXT: SETGE_DX10 T0.W, T0.X, 0.0,
+; EG-NEXT: SETGE_DX10 * T1.W, T1.X, 0.0,
+; EG-NEXT: OR_INT * T0.W, PS, PV.W,
; EG-NEXT: AND_INT T0.X, PV.W, 1,
; EG-NEXT: LSHR * T1.X, KC0[2].Y, literal.x,
; EG-NEXT: 2(2.802597e-45), 0(0.000000e+00)
diff --git a/llvm/test/CodeGen/RISCV/combine-andor-with-cmps.ll b/llvm/test/CodeGen/RISCV/combine-andor-with-cmps.ll
new file mode 100644
index 0000000000000..99a3788b6fa8f
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/combine-andor-with-cmps.ll
@@ -0,0 +1,83 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=riscv64 -mattr=+d -verify-machineinstrs < %s | FileCheck %s
+
+; The tests check the following optimization of DAGCombiner:
+; CMP(A,C)||CMP(B,C) => CMP(MIN/MAX(A,B), C)
+; CMP(A,C)&&CMP(B,C) => CMP(MIN/MAX(A,B), C)
+; FMINIMUMNUM/FMAXIMUMNUM discard signaling NaNs, so A and B may be sNaNs.
+
+define i1 @olt_or(float %arg1, float %arg2, float %arg3) {
+; CHECK-LABEL: olt_or:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmin.s fa5, fa0, fa1
+; CHECK-NEXT: flt.s a0, fa5, fa2
+; CHECK-NEXT: ret
+ %cmp1 = fcmp olt float %arg1, %arg3
+ %cmp2 = fcmp olt float %arg2, %arg3
+ %or1 = or i1 %cmp1, %cmp2
+ ret i1 %or1
+}
+
+define i1 @ogt_or(double %arg1, double %arg2, double %arg3) {
+; CHECK-LABEL: ogt_or:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmax.d fa5, fa0, fa1
+; CHECK-NEXT: flt.d a0, fa2, fa5
+; CHECK-NEXT: ret
+ %cmp1 = fcmp ogt double %arg1, %arg3
+ %cmp2 = fcmp ogt double %arg2, %arg3
+ %or1 = or i1 %cmp1, %cmp2
+ ret i1 %or1
+}
+
+define i1 @ugt_and(float %arg1, float %arg2, float %arg3) {
+; CHECK-LABEL: ugt_and:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmin.s fa5, fa0, fa1
+; CHECK-NEXT: fle.s a0, fa5, fa2
+; CHECK-NEXT: xori a0, a0, 1
+; CHECK-NEXT: ret
+ %cmp1 = fcmp ugt float %arg1, %arg3
+ %cmp2 = fcmp ugt float %arg2, %arg3
+ %and1 = and i1 %cmp1, %cmp2
+ ret i1 %and1
+}
+
+define i1 @ult_and(double %arg1, double %arg2, double %arg3) {
+; CHECK-LABEL: ult_and:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmax.d fa5, fa0, fa1
+; CHECK-NEXT: fle.d a0, fa2, fa5
+; CHECK-NEXT: xori a0, a0, 1
+; CHECK-NEXT: ret
+ %cmp1 = fcmp ult double %arg1, %arg3
+ %cmp2 = fcmp ult double %arg2, %arg3
+ %and1 = and i1 %cmp1, %cmp2
+ ret i1 %and1
+}
+
+define i1 @nnan_olt_and(float %arg1, float %arg2, float %arg3) {
+; CHECK-LABEL: nnan_olt_and:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmax.s fa5, fa0, fa1
+; CHECK-NEXT: flt.s a0, fa5, fa2
+; CHECK-NEXT: ret
+ %cmp1 = fcmp nnan olt float %arg1, %arg3
+ %cmp2 = fcmp nnan olt float %arg2, %arg3
+ %and1 = and i1 %cmp1, %cmp2
+ ret i1 %and1
+}
+
+; Negative test: a NaN operand makes its compare false, but MAX discards it.
+define i1 @olt_and(float %arg1, float %arg2, float %arg3) {
+; CHECK-LABEL: olt_and:
+; CHECK: # %bb.0:
+; CHECK-NEXT: flt.s a0, fa1, fa2
+; CHECK-NEXT: flt.s a1, fa0, fa2
+; CHECK-NEXT: and a0, a1, a0
+; CHECK-NEXT: ret
+ %cmp1 = fcmp olt float %arg1, %arg3
+ %cmp2 = fcmp olt float %arg2, %arg3
+ %and1 = and i1 %cmp1, %cmp2
+ ret i1 %and1
+}
``````````
</details>
https://github.com/llvm/llvm-project/pull/223655
More information about the llvm-commits
mailing list