[llvm] [SelectionDAG] Fix fcmp fold for new min/max semantics (PR #223655)

via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 15 04:09:35 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-amdgpu

Author: Mitch Briles (MitchBriles)

<details>
<summary>Changes</summary>

Fixes #<!-- -->223089

**Changes**

- Updated the comments in `ISDOpcodes.h` with the new semantics for floating-point min/max operations.
- Fixed the following fold to use the correct min/max opcodes and correctly handle NaNs. This fold can now use `FMINIMUMNUM` and `FMAXIMUMNUM`.
```
(A < B) | (C < B) -> min(A, C) < B
(A < B) & (C < B) -> max(A, C) < B
```

**Note about X86**

X86 marks `ISD::FMAXIMUMNUM` and `ISD::FMINIMUMNUM` as custom, but the lowering is quite expensive for this fold, so I used `isProfitableToCombineMinNumMaxNum` to prevent regressions.

cc @<!-- -->arsenm @<!-- -->topperc 

---
Full diff: https://github.com/llvm/llvm-project/pull/223655.diff


6 Files Affected:

- (modified) llvm/include/llvm/CodeGen/ISDOpcodes.h (+34-28) 
- (modified) llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp (+50-45) 
- (modified) llvm/test/CodeGen/AArch64/combine_andor_with_cmps.ll (+42-4) 
- (modified) llvm/test/CodeGen/AMDGPU/combine_andor_with_cmps_nnan.ll (+7-6) 
- (modified) llvm/test/CodeGen/AMDGPU/or.r600.ll (+15-10) 
- (added) llvm/test/CodeGen/RISCV/combine-andor-with-cmps.ll (+83) 


``````````diff
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index 9c3757203299e..5e38c7ccfbb3d 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -1085,47 +1085,53 @@ enum NodeType {
   LRINT,
   LLRINT,
 
-  /// FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values,
-  /// following IEEE-754 definitions except for signed zero behavior.
+  /// FMINNUM/FMAXNUM - NaN-discarding minimum/maximum: if one operand is a
+  /// quiet NaN and the other is a number, returns the number.
   ///
-  /// If one input is a signaling NaN, returns a quiet NaN. This matches
-  /// IEEE-754 2008's minNum/maxNum behavior for signaling NaNs (which differs
-  /// from 2019).
+  /// If an operand is a signaling NaN, this will non-deterministically either:
+  /// - Return a NaN.
+  /// - Or treat the signaling NaN as a quiet NaN.
   ///
   /// These treat -0 as ordered less than +0, matching the behavior of IEEE-754
-  /// 2019's minimumNumber/maximumNumber.
-  ///
-  /// Note that that arithmetic on an sNaN doesn't consistently produce a qNaN,
-  /// so arithmetic feeding into a minnum/maxnum can produce inconsistent
-  /// results. FMAXIMUN/FMINIMUM or FMAXIMUMNUM/FMINIMUMNUM may be better choice
-  /// for non-distinction of sNaN/qNaN handling.
+  /// 2019's minimumNumber/maximumNumber. With the nsz flag, one +0.0 and one
+  /// -0.0 operand may non-deterministically return either operand; contrary to
+  /// normal nsz semantics, if both operands have the same sign, so must the
+  /// result. Note that not all backends respect this ordering yet.
   FMINNUM,
   FMAXNUM,
 
-  /// FMINNUM_IEEE/FMAXNUM_IEEE - Perform floating-point minimumNumber or
-  /// maximumNumber on two values, following IEEE-754 definitions. This differs
-  /// from FMINNUM/FMAXNUM in the handling of signaling NaNs, and signed zero.
-  ///
-  /// If one input is a signaling NaN, returns a quiet NaN. This matches
-  /// IEEE-754 2008's minnum/maxnum behavior for signaling NaNs (which differs
-  /// from 2019).
-  ///
-  /// These treat -0 as ordered less than +0, matching the behavior of IEEE-754
-  /// 2019's minimumNumber/maximumNumber.
+  /// FMINNUM_IEEE/FMAXNUM_IEEE - Same as FMINNUM/FMAXNUM, except that a
+  /// signaling NaN operand deterministically returns a quiet NaN, matching/for
+  /// IEEE-754 2008's minNum/maxNum. Signed zeros are ordered identically to
+  /// FMINNUM/FMAXNUM: -0 is less than +0, relaxed by the nsz flag.
   ///
-  /// Deprecated, and will be removed soon, as FMINNUM/FMAXNUM have the same
-  /// semantics now.
+  /// Deprecated, and will be removed soon: this is a legal implementation of
+  /// FMINNUM/FMAXNUM, so targets should select those instead.
   FMINNUM_IEEE,
   FMAXNUM_IEEE,
 
-  /// FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0
-  /// as less than 0.0. While FMINNUM_IEEE/FMAXNUM_IEEE follow IEEE 754-2008
-  /// semantics, FMINIMUM/FMAXIMUM follow IEEE 754-2019 semantics.
+  /// FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum: if either operand is
+  /// a NaN, returns a NaN. Follows C23's fminimum/fmaximum and IEEE-754 2019's
+  /// minimum/maximum, except that a signaling NaN operand is not guaranteed to
+  /// be quieted.
+  ///
+  /// These treat -0 as ordered less than +0. With the nsz flag, one +0.0 and
+  /// one -0.0 operand may non-deterministically return either operand;
+  /// contrary to normal nsz semantics, if both operands have the same sign, so
+  /// must the result.
   FMINIMUM,
   FMAXIMUM,
 
-  /// FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with
-  /// FMINNUM_IEEE and FMAXNUM_IEEE besides if either operand is sNaN.
+  /// FMINIMUMNUM/FMAXIMUMNUM - NaN-discarding minimum/maximum: if one operand
+  /// is a NaN and the other is a number, returns the number. Follows C23's
+  /// fminimum_num/fmaximum_num and IEEE-754 2019's minimumNumber/maximumNumber,
+  /// except that a signaling NaN operand is not guaranteed to be quieted.
+  /// Same as FMINNUM/FMAXNUM, but treats signaling NaNs as quiet NaNs.
+  ///
+  /// These treat -0 as ordered less than +0. With the nsz flag, one +0.0 and
+  /// one -0.0 operand may non-deterministically return either operand;
+  /// contrary to normal nsz semantics, if both operands have the same sign, so
+  /// must the result.
   FMINIMUMNUM,
   FMAXIMUMNUM,
 
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index a829d7a34d5a6..a84127e453b11 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -6935,61 +6935,57 @@ static unsigned getMinMaxOpcodeForClamp(bool IsMin, SDValue Operand1,
   return ISD::DELETED_NODE;
 }
 
-// FIXME: use FMINIMUMNUM if possible, such as for RISC-V.
 static unsigned getMinMaxOpcodeForCompareFold(
     SDValue Operand1, SDValue Operand2, bool SetCCNoNaNs, ISD::CondCode CC,
     unsigned OrAndOpcode, SelectionDAG &DAG, bool isFMAXNUMFMINNUM_IEEE,
-    bool isFMAXNUMFMINNUM) {
+    bool isFMAXNUMFMINNUM, bool isFMAXIMUMNUMFMINIMUMNUM) {
   // The optimization cannot be applied for all the predicates because
-  // of the way FMINNUM/FMAXNUM and FMINNUM_IEEE/FMAXNUM_IEEE handle
-  // NaNs. For FMINNUM_IEEE/FMAXNUM_IEEE, the optimization cannot be
-  // applied at all if one of the operands is a signaling NaN.
+  // of the way the min/max opcodes handle NaNs.
 
-  // It is safe to use FMINNUM_IEEE/FMAXNUM_IEEE if all the operands
-  // are non NaN values.
+  // It is safe to use FMINIMUMNUM/FMAXIMUMNUM or FMINNUM_IEEE/FMAXNUM_IEEE if
+  // all the operands are non NaN values.
   if (((CC == ISD::SETLT || CC == ISD::SETLE) && (OrAndOpcode == ISD::OR)) ||
       ((CC == ISD::SETGT || CC == ISD::SETGE) && (OrAndOpcode == ISD::AND))) {
-    return (SetCCNoNaNs || arebothOperandsNotNan(Operand1, Operand2, DAG)) &&
-                   isFMAXNUMFMINNUM_IEEE
-               ? ISD::FMINNUM_IEEE
-               : ISD::DELETED_NODE;
+    if (!SetCCNoNaNs && !arebothOperandsNotNan(Operand1, Operand2, DAG))
+      return ISD::DELETED_NODE;
+    return isFMAXIMUMNUMFMINIMUMNUM ? ISD::FMINIMUMNUM
+           : isFMAXNUMFMINNUM_IEEE  ? ISD::FMINNUM_IEEE
+                                    : ISD::DELETED_NODE;
   }
 
   if (((CC == ISD::SETGT || CC == ISD::SETGE) && (OrAndOpcode == ISD::OR)) ||
       ((CC == ISD::SETLT || CC == ISD::SETLE) && (OrAndOpcode == ISD::AND))) {
-    return (SetCCNoNaNs || arebothOperandsNotNan(Operand1, Operand2, DAG)) &&
-                   isFMAXNUMFMINNUM_IEEE
-               ? ISD::FMAXNUM_IEEE
-               : ISD::DELETED_NODE;
-  }
-
-  // Both FMINNUM/FMAXNUM and FMINNUM_IEEE/FMAXNUM_IEEE handle quiet
-  // NaNs in the same way. But, FMINNUM/FMAXNUM and FMINNUM_IEEE/
-  // FMAXNUM_IEEE handle signaling NaNs differently. If we cannot prove
-  // that there are not any sNaNs, then the optimization is not valid
-  // for FMINNUM_IEEE/FMAXNUM_IEEE. In the presence of sNaNs, we apply
-  // the optimization using FMINNUM/FMAXNUM for the following cases. If
-  // we can prove that we do not have any sNaNs, then we can do the
-  // optimization using FMINNUM_IEEE/FMAXNUM_IEEE for the following
-  // cases.
-  if (((CC == ISD::SETOLT || CC == ISD::SETOLE) && (OrAndOpcode == ISD::OR)) ||
-      ((CC == ISD::SETUGT || CC == ISD::SETUGE) && (OrAndOpcode == ISD::AND))) {
-    return isFMAXNUMFMINNUM ? ISD::FMINNUM
-           : arebothOperandsNotSNan(Operand1, Operand2, DAG) &&
-                   isFMAXNUMFMINNUM_IEEE
-               ? ISD::FMINNUM_IEEE
-               : ISD::DELETED_NODE;
-  }
-
-  if (((CC == ISD::SETOGT || CC == ISD::SETOGE) && (OrAndOpcode == ISD::OR)) ||
-      ((CC == ISD::SETULT || CC == ISD::SETULE) && (OrAndOpcode == ISD::AND))) {
-    return isFMAXNUMFMINNUM ? ISD::FMAXNUM
-           : arebothOperandsNotSNan(Operand1, Operand2, DAG) &&
-                   isFMAXNUMFMINNUM_IEEE
-               ? ISD::FMAXNUM_IEEE
-               : ISD::DELETED_NODE;
+    if (!SetCCNoNaNs && !arebothOperandsNotNan(Operand1, Operand2, DAG))
+      return ISD::DELETED_NODE;
+    return isFMAXIMUMNUMFMINIMUMNUM ? ISD::FMAXIMUMNUM
+           : isFMAXNUMFMINNUM_IEEE  ? ISD::FMAXNUM_IEEE
+                                    : ISD::DELETED_NODE;
   }
 
+  bool IsMin;
+  if (((CC == ISD::SETOLT || CC == ISD::SETOLE) && (OrAndOpcode == ISD::OR)) ||
+      ((CC == ISD::SETUGT || CC == ISD::SETUGE) && (OrAndOpcode == ISD::AND)))
+    IsMin = true;
+  else if (((CC == ISD::SETOGT || CC == ISD::SETOGE) &&
+            (OrAndOpcode == ISD::OR)) ||
+           ((CC == ISD::SETULT || CC == ISD::SETULE) &&
+            (OrAndOpcode == ISD::AND)))
+    IsMin = false;
+  else
+    return ISD::DELETED_NODE;
+
+  // For the above predicates, the optimization is valid if a NaN operand is
+  // discarded. FMINIMUMNUM/FMAXIMUMNUM always do so. FMINNUM/FMAXNUM and
+  // FMINNUM_IEEE/FMAXNUM_IEEE only do so for quiet NaNs, as a signaling NaN
+  // operand may instead produce a NaN.
+  if (isFMAXIMUMNUMFMINIMUMNUM)
+    return IsMin ? ISD::FMINIMUMNUM : ISD::FMAXIMUMNUM;
+  if (!arebothOperandsNotSNan(Operand1, Operand2, DAG))
+    return ISD::DELETED_NODE;
+  if (isFMAXNUMFMINNUM)
+    return IsMin ? ISD::FMINNUM : ISD::FMAXNUM;
+  if (isFMAXNUMFMINNUM_IEEE)
+    return IsMin ? ISD::FMINNUM_IEEE : ISD::FMAXNUM_IEEE;
   return ISD::DELETED_NODE;
 }
 
@@ -7040,12 +7036,20 @@ static SDValue foldAndOrOfSETCC(SDNode *LogicOp, SelectionDAG &DAG) {
                                TLI.isOperationLegal(ISD::FMINNUM_IEEE, OpVT);
   bool isFMAXNUMFMINNUM = TLI.isOperationLegalOrCustom(ISD::FMAXNUM, OpVT) &&
                           TLI.isOperationLegalOrCustom(ISD::FMINNUM, OpVT);
+  // A custom FMINIMUMNUM/FMAXIMUMNUM may be more expensive than the compares,
+  // so only use one if the target considers forming it profitable.
+  bool isFMAXIMUMNUMFMINIMUMNUM =
+      (TLI.isOperationLegal(ISD::FMAXIMUMNUM, OpVT) &&
+       TLI.isOperationLegal(ISD::FMINIMUMNUM, OpVT)) ||
+      (TLI.isOperationLegalOrCustom(ISD::FMAXIMUMNUM, OpVT) &&
+       TLI.isOperationLegalOrCustom(ISD::FMINIMUMNUM, OpVT) &&
+       TLI.isProfitableToCombineMinNumMaxNum(OpVT));
   if (((OpVT.isInteger() && TLI.isOperationLegal(ISD::UMAX, OpVT) &&
         TLI.isOperationLegal(ISD::SMAX, OpVT) &&
         TLI.isOperationLegal(ISD::UMIN, OpVT) &&
         TLI.isOperationLegal(ISD::SMIN, OpVT)) ||
-       (OpVT.isFloatingPoint() &&
-        (isFMAXNUMFMINNUM_IEEE || isFMAXNUMFMINNUM))) &&
+       (OpVT.isFloatingPoint() && (isFMAXNUMFMINNUM_IEEE || isFMAXNUMFMINNUM ||
+                                   isFMAXIMUMNUMFMINIMUMNUM))) &&
       !ISD::isIntEqualitySetCC(CCL) && !ISD::isFPEqualitySetCC(CCL) &&
       CCL != ISD::SETFALSE && CCL != ISD::SETO && CCL != ISD::SETUO &&
       CCL != ISD::SETTRUE &&
@@ -7102,7 +7106,8 @@ static SDValue foldAndOrOfSETCC(SDNode *LogicOp, SelectionDAG &DAG) {
         NewOpcode = getMinMaxOpcodeForCompareFold(
             Operand1, Operand2,
             LHSSetCCFlags.hasNoNaNs() && RHSSetCCFlags.hasNoNaNs(), CC,
-            LogicOp->getOpcode(), DAG, isFMAXNUMFMINNUM_IEEE, isFMAXNUMFMINNUM);
+            LogicOp->getOpcode(), DAG, isFMAXNUMFMINNUM_IEEE, isFMAXNUMFMINNUM,
+            isFMAXIMUMNUMFMINIMUMNUM);
 
       if (NewOpcode != ISD::DELETED_NODE) {
         // Propagate fast-math flags from setcc.
diff --git a/llvm/test/CodeGen/AArch64/combine_andor_with_cmps.ll b/llvm/test/CodeGen/AArch64/combine_andor_with_cmps.ll
index 89cb25f3d9d75..a1eaac062e6b4 100644
--- a/llvm/test/CodeGen/AArch64/combine_andor_with_cmps.ll
+++ b/llvm/test/CodeGen/AArch64/combine_andor_with_cmps.ll
@@ -5,12 +5,14 @@
 ; CMP(A,C)||CMP(B,C) => CMP(MIN/MAX(A,B), C)
 ; CMP(A,C)&&CMP(B,C) => CMP(MIN/MAX(A,B), C)
 
+; Not folded: A or B may be a signaling NaN, for which MIN/MAX may return NaN.
 define i1 @test1(float %arg1, float %arg2, float %arg3) #0 {
 ; CHECK-LABEL: test1:
 ; CHECK:       // %bb.0:
-; CHECK-NEXT:    fminnm s0, s0, s1
 ; CHECK-NEXT:    fcmp s0, s2
-; CHECK-NEXT:    cset w0, mi
+; CHECK-NEXT:    cset w8, mi
+; CHECK-NEXT:    fcmp s1, s2
+; CHECK-NEXT:    csinc w0, w8, wzr, pl
 ; CHECK-NEXT:    ret
   %cmp1 = fcmp olt float %arg1, %arg3
   %cmp2 = fcmp olt float %arg2, %arg3
@@ -21,9 +23,10 @@ define i1 @test1(float %arg1, float %arg2, float %arg3) #0 {
 define i1 @test2(double %arg1, double %arg2, double %arg3) #0 {
 ; CHECK-LABEL: test2:
 ; CHECK:       // %bb.0:
-; CHECK-NEXT:    fmaxnm d0, d0, d1
 ; CHECK-NEXT:    fcmp d0, d2
-; CHECK-NEXT:    cset w0, gt
+; CHECK-NEXT:    cset w8, gt
+; CHECK-NEXT:    fcmp d1, d2
+; CHECK-NEXT:    csinc w0, w8, wzr, le
 ; CHECK-NEXT:    ret
   %cmp1 = fcmp ogt double %arg1, %arg3
   %cmp2 = fcmp ogt double %arg2, %arg3
@@ -69,3 +72,38 @@ define i1 @test4(float %arg1, float %arg2, float %arg3) {
   ret i1 %or1
 }
 
+; A and B cannot be signaling NaNs, so the folds happen.
+define i1 @test5(i32 %arg1, i32 %arg2, float %arg3) {
+; CHECK-LABEL: test5:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    scvtf s1, w0
+; CHECK-NEXT:    scvtf s2, w1
+; CHECK-NEXT:    fminnm s1, s1, s2
+; CHECK-NEXT:    fcmp s1, s0
+; CHECK-NEXT:    cset w0, mi
+; CHECK-NEXT:    ret
+  %conv1 = sitofp i32 %arg1 to float
+  %conv2 = sitofp i32 %arg2 to float
+  %cmp1 = fcmp olt float %conv1, %arg3
+  %cmp2 = fcmp olt float %conv2, %arg3
+  %or1  = or i1 %cmp1, %cmp2
+  ret i1 %or1
+}
+
+define i1 @test6(i32 %arg1, i32 %arg2, double %arg3) {
+; CHECK-LABEL: test6:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    scvtf d1, w0
+; CHECK-NEXT:    scvtf d2, w1
+; CHECK-NEXT:    fmaxnm d1, d1, d2
+; CHECK-NEXT:    fcmp d1, d0
+; CHECK-NEXT:    cset w0, lt
+; CHECK-NEXT:    ret
+  %conv1 = sitofp i32 %arg1 to double
+  %conv2 = sitofp i32 %arg2 to double
+  %cmp1 = fcmp ult double %conv1, %arg3
+  %cmp2 = fcmp ult double %conv2, %arg3
+  %and1 = and i1 %cmp1, %cmp2
+  ret i1 %and1
+}
+
diff --git a/llvm/test/CodeGen/AMDGPU/combine_andor_with_cmps_nnan.ll b/llvm/test/CodeGen/AMDGPU/combine_andor_with_cmps_nnan.ll
index 550cfec413d79..87ac3b4c73d6d 100644
--- a/llvm/test/CodeGen/AMDGPU/combine_andor_with_cmps_nnan.ll
+++ b/llvm/test/CodeGen/AMDGPU/combine_andor_with_cmps_nnan.ll
@@ -954,15 +954,16 @@ define i1 @test117(float %arg1, float %arg2, float %arg3, float %arg4, float %ar
 ; GCN-LABEL: test117:
 ; GCN:       ; %bb.0:
 ; GCN-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GCN-NEXT:    v_min3_f32 v4, v4, v5, v6
 ; GCN-NEXT:    v_min_f32_e32 v0, v0, v1
-; GCN-NEXT:    v_min3_f32 v1, v8, v9, v10
-; GCN-NEXT:    v_min_f32_e32 v2, v2, v3
-; GCN-NEXT:    v_min_f32_e32 v3, v4, v7
+; GCN-NEXT:    v_min3_f32 v1, v4, v5, v6
+; GCN-NEXT:    v_max_f32_e32 v4, v7, v7
+; GCN-NEXT:    v_min3_f32 v5, v8, v9, v10
+; GCN-NEXT:    v_max_f32_e32 v6, v11, v11
+; GCN-NEXT:    v_dual_min_f32 v2, v2, v3 :: v_dual_min_f32 v1, v1, v4
 ; GCN-NEXT:    v_cmp_lt_f32_e32 vcc_lo, v0, v12
-; GCN-NEXT:    v_min_f32_e32 v0, v1, v11
+; GCN-NEXT:    v_min_f32_e32 v0, v5, v6
 ; GCN-NEXT:    v_cmp_lt_f32_e64 s0, v2, v13
-; GCN-NEXT:    v_cmp_lt_f32_e64 s1, v3, v13
+; GCN-NEXT:    v_cmp_lt_f32_e64 s1, v1, v13
 ; GCN-NEXT:    v_cmp_lt_f32_e64 s2, v0, v12
 ; GCN-NEXT:    s_or_b32 s0, vcc_lo, s0
 ; GCN-NEXT:    s_or_b32 s0, s0, s1
diff --git a/llvm/test/CodeGen/AMDGPU/or.r600.ll b/llvm/test/CodeGen/AMDGPU/or.r600.ll
index ed9d0085fd82a..2b06ffaaf387c 100644
--- a/llvm/test/CodeGen/AMDGPU/or.r600.ll
+++ b/llvm/test/CodeGen/AMDGPU/or.r600.ll
@@ -456,21 +456,26 @@ define amdgpu_kernel void @trunc_i64_or_to_i32(ptr addrspace(1) %out, [8 x i32],
 define amdgpu_kernel void @or_i1(ptr addrspace(1) %out, ptr addrspace(1) %in0, ptr addrspace(1) %in1) {
 ; EG-LABEL: or_i1:
 ; EG:       ; %bb.0:
-; EG-NEXT:    ALU 1, @10, KC0[CB0:0-32], KC1[]
-; EG-NEXT:    TEX 1 @6
-; EG-NEXT:    ALU 4, @12, KC0[CB0:0-32], KC1[]
+; EG-NEXT:    ALU 0, @12, KC0[CB0:0-32], KC1[]
+; EG-NEXT:    TEX 0 @8
+; EG-NEXT:    ALU 0, @13, KC0[CB0:0-32], KC1[]
+; EG-NEXT:    TEX 0 @10
+; EG-NEXT:    ALU 5, @14, KC0[CB0:0-32], KC1[]
 ; EG-NEXT:    MEM_RAT_CACHELESS STORE_RAW T0.X, T1.X, 1
 ; EG-NEXT:    CF_END
 ; EG-NEXT:    PAD
-; EG-NEXT:    Fetch clause starting at 6:
-; EG-NEXT:     VTX_READ_32 T1.X, T1.X, 0, #1
+; EG-NEXT:    Fetch clause starting at 8:
 ; EG-NEXT:     VTX_READ_32 T0.X, T0.X, 0, #1
-; EG-NEXT:    ALU clause starting at 10:
-; EG-NEXT:     MOV T0.X, KC0[2].Z,
-; EG-NEXT:     MOV * T1.X, KC0[2].W,
+; EG-NEXT:    Fetch clause starting at 10:
+; EG-NEXT:     VTX_READ_32 T1.X, T1.X, 0, #1
 ; EG-NEXT:    ALU clause starting at 12:
-; EG-NEXT:     MAX_DX10 * T0.W, T0.X, T1.X,
-; EG-NEXT:     SETGE_DX10 * T0.W, PV.W, 0.0,
+; EG-NEXT:     MOV * T0.X, KC0[2].W,
+; EG-NEXT:    ALU clause starting at 13:
+; EG-NEXT:     MOV * T1.X, KC0[2].Z,
+; EG-NEXT:    ALU clause starting at 14:
+; EG-NEXT:     SETGE_DX10 T0.W, T0.X, 0.0,
+; EG-NEXT:     SETGE_DX10 * T1.W, T1.X, 0.0,
+; EG-NEXT:     OR_INT * T0.W, PS, PV.W,
 ; EG-NEXT:     AND_INT T0.X, PV.W, 1,
 ; EG-NEXT:     LSHR * T1.X, KC0[2].Y, literal.x,
 ; EG-NEXT:    2(2.802597e-45), 0(0.000000e+00)
diff --git a/llvm/test/CodeGen/RISCV/combine-andor-with-cmps.ll b/llvm/test/CodeGen/RISCV/combine-andor-with-cmps.ll
new file mode 100644
index 0000000000000..99a3788b6fa8f
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/combine-andor-with-cmps.ll
@@ -0,0 +1,83 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=riscv64 -mattr=+d -verify-machineinstrs < %s | FileCheck %s
+
+; The tests check the following optimization of DAGCombiner:
+; CMP(A,C)||CMP(B,C) => CMP(MIN/MAX(A,B), C)
+; CMP(A,C)&&CMP(B,C) => CMP(MIN/MAX(A,B), C)
+; FMINIMUMNUM/FMAXIMUMNUM discard signaling NaNs, so A and B may be sNaNs.
+
+define i1 @olt_or(float %arg1, float %arg2, float %arg3) {
+; CHECK-LABEL: olt_or:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    fmin.s fa5, fa0, fa1
+; CHECK-NEXT:    flt.s a0, fa5, fa2
+; CHECK-NEXT:    ret
+  %cmp1 = fcmp olt float %arg1, %arg3
+  %cmp2 = fcmp olt float %arg2, %arg3
+  %or1  = or i1 %cmp1, %cmp2
+  ret i1 %or1
+}
+
+define i1 @ogt_or(double %arg1, double %arg2, double %arg3) {
+; CHECK-LABEL: ogt_or:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    fmax.d fa5, fa0, fa1
+; CHECK-NEXT:    flt.d a0, fa2, fa5
+; CHECK-NEXT:    ret
+  %cmp1 = fcmp ogt double %arg1, %arg3
+  %cmp2 = fcmp ogt double %arg2, %arg3
+  %or1  = or i1 %cmp1, %cmp2
+  ret i1 %or1
+}
+
+define i1 @ugt_and(float %arg1, float %arg2, float %arg3) {
+; CHECK-LABEL: ugt_and:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    fmin.s fa5, fa0, fa1
+; CHECK-NEXT:    fle.s a0, fa5, fa2
+; CHECK-NEXT:    xori a0, a0, 1
+; CHECK-NEXT:    ret
+  %cmp1 = fcmp ugt float %arg1, %arg3
+  %cmp2 = fcmp ugt float %arg2, %arg3
+  %and1 = and i1 %cmp1, %cmp2
+  ret i1 %and1
+}
+
+define i1 @ult_and(double %arg1, double %arg2, double %arg3) {
+; CHECK-LABEL: ult_and:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    fmax.d fa5, fa0, fa1
+; CHECK-NEXT:    fle.d a0, fa2, fa5
+; CHECK-NEXT:    xori a0, a0, 1
+; CHECK-NEXT:    ret
+  %cmp1 = fcmp ult double %arg1, %arg3
+  %cmp2 = fcmp ult double %arg2, %arg3
+  %and1 = and i1 %cmp1, %cmp2
+  ret i1 %and1
+}
+
+define i1 @nnan_olt_and(float %arg1, float %arg2, float %arg3) {
+; CHECK-LABEL: nnan_olt_and:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    fmax.s fa5, fa0, fa1
+; CHECK-NEXT:    flt.s a0, fa5, fa2
+; CHECK-NEXT:    ret
+  %cmp1 = fcmp nnan olt float %arg1, %arg3
+  %cmp2 = fcmp nnan olt float %arg2, %arg3
+  %and1 = and i1 %cmp1, %cmp2
+  ret i1 %and1
+}
+
+; Negative test: a NaN operand makes its compare false, but MAX discards it.
+define i1 @olt_and(float %arg1, float %arg2, float %arg3) {
+; CHECK-LABEL: olt_and:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    flt.s a0, fa1, fa2
+; CHECK-NEXT:    flt.s a1, fa0, fa2
+; CHECK-NEXT:    and a0, a1, a0
+; CHECK-NEXT:    ret
+  %cmp1 = fcmp olt float %arg1, %arg3
+  %cmp2 = fcmp olt float %arg2, %arg3
+  %and1 = and i1 %cmp1, %cmp2
+  ret i1 %and1
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/223655


More information about the llvm-commits mailing list