[llvm] [X86] Fold FP UNORD/ORD compare against known-non-NaN to self-compare (PR #206943)

Simon Pilgrim via llvm-commits llvm-commits at lists.llvm.org
Wed Jul 1 07:17:05 PDT 2026


https://github.com/RKSimon updated https://github.com/llvm/llvm-project/pull/206943

>From 522c6791315af857609ee10595677fa1c005f203 Mon Sep 17 00:00:00 2001
From: Harishankar <harishankarpp7 at gmail.com>
Date: Wed, 1 Jul 2026 16:12:20 +0530
Subject: [PATCH] [X86] Fold FP UNORD/ORD compare against known-non-NaN to
 self-compare

Replace a known-non-NaN operand of an UNORD/ORD FP compare with the
other operand, so we emit `cmpp a, a` instead of materializing the
constant with `vxorpd`. Matches GCC.

Added combine-fcmp-uno-self.ll which covers the folding and a negative case.

Fixes #202756
---
 llvm/lib/Target/X86/X86ISelLowering.cpp       |  6 ++
 llvm/test/CodeGen/X86/bitcast-vector-bool.ll  | 26 +++---
 .../test/CodeGen/X86/combine-fcmp-uno-self.ll | 83 +++++++++++++++++++
 3 files changed, 100 insertions(+), 15 deletions(-)
 create mode 100644 llvm/test/CodeGen/X86/combine-fcmp-uno-self.ll

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index cf4e0c081a1b8..b8fb7eed00915 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -24952,6 +24952,12 @@ static SDValue LowerVSETCC(SDValue Op, const X86Subtarget &Subtarget,
     SDValue Cmp;
     bool IsAlwaysSignaling;
     unsigned SSECC = translateX86FSETCC(Cond, Op0, Op1, IsAlwaysSignaling);
+    if ((Cond == ISD::SETO || Cond == ISD::SETUO) && Op0 != Op1) {
+      if (DAG.isKnownNeverNaN(Op1))
+        Op1 = Op0;
+      else if (DAG.isKnownNeverNaN(Op0))
+        Op0 = Op1;
+    }
     if (!Subtarget.hasAVX()) {
       // TODO: We could use following steps to handle a quiet compare with
       // signaling encodings.
diff --git a/llvm/test/CodeGen/X86/bitcast-vector-bool.ll b/llvm/test/CodeGen/X86/bitcast-vector-bool.ll
index f0688482fbeb8..f680520c05cb5 100644
--- a/llvm/test/CodeGen/X86/bitcast-vector-bool.ll
+++ b/llvm/test/CodeGen/X86/bitcast-vector-bool.ll
@@ -1267,12 +1267,11 @@ define i1 @trunc_v128i8_cmp(<128 x i8> %a0) nounwind {
 define [2 x i8] @PR58546(<16 x float> %a0) {
 ; SSE-LABEL: PR58546:
 ; SSE:       # %bb.0:
-; SSE-NEXT:    xorps %xmm4, %xmm4
-; SSE-NEXT:    cmpunordps %xmm4, %xmm3
-; SSE-NEXT:    cmpunordps %xmm4, %xmm2
+; SSE-NEXT:    cmpunordps %xmm3, %xmm3
+; SSE-NEXT:    cmpunordps %xmm2, %xmm2
 ; SSE-NEXT:    packssdw %xmm3, %xmm2
-; SSE-NEXT:    cmpunordps %xmm4, %xmm1
-; SSE-NEXT:    cmpunordps %xmm4, %xmm0
+; SSE-NEXT:    cmpunordps %xmm1, %xmm1
+; SSE-NEXT:    cmpunordps %xmm0, %xmm0
 ; SSE-NEXT:    packssdw %xmm1, %xmm0
 ; SSE-NEXT:    packsswb %xmm2, %xmm0
 ; SSE-NEXT:    pmovmskb %xmm0, %eax
@@ -1284,11 +1283,10 @@ define [2 x i8] @PR58546(<16 x float> %a0) {
 ;
 ; AVX1-LABEL: PR58546:
 ; AVX1:       # %bb.0:
-; AVX1-NEXT:    vxorps %xmm2, %xmm2, %xmm2
-; AVX1-NEXT:    vcmpunordps %ymm2, %ymm1, %ymm1
-; AVX1-NEXT:    vextractf128 $1, %ymm1, %xmm3
-; AVX1-NEXT:    vpackssdw %xmm3, %xmm1, %xmm1
-; AVX1-NEXT:    vcmpunordps %ymm2, %ymm0, %ymm0
+; AVX1-NEXT:    vcmpunordps %ymm1, %ymm1, %ymm1
+; AVX1-NEXT:    vextractf128 $1, %ymm1, %xmm2
+; AVX1-NEXT:    vpackssdw %xmm2, %xmm1, %xmm1
+; AVX1-NEXT:    vcmpunordps %ymm0, %ymm0, %ymm0
 ; AVX1-NEXT:    vextractf128 $1, %ymm0, %xmm2
 ; AVX1-NEXT:    vpackssdw %xmm2, %xmm0, %xmm0
 ; AVX1-NEXT:    vpacksswb %xmm1, %xmm0, %xmm0
@@ -1302,9 +1300,8 @@ define [2 x i8] @PR58546(<16 x float> %a0) {
 ;
 ; AVX2-LABEL: PR58546:
 ; AVX2:       # %bb.0:
-; AVX2-NEXT:    vxorps %xmm2, %xmm2, %xmm2
-; AVX2-NEXT:    vcmpunordps %ymm2, %ymm1, %ymm1
-; AVX2-NEXT:    vcmpunordps %ymm2, %ymm0, %ymm0
+; AVX2-NEXT:    vcmpunordps %ymm1, %ymm1, %ymm1
+; AVX2-NEXT:    vcmpunordps %ymm0, %ymm0, %ymm0
 ; AVX2-NEXT:    vpackssdw %ymm1, %ymm0, %ymm0
 ; AVX2-NEXT:    vextracti128 $1, %ymm0, %xmm1
 ; AVX2-NEXT:    vpacksswb %xmm1, %xmm0, %xmm0
@@ -1319,8 +1316,7 @@ define [2 x i8] @PR58546(<16 x float> %a0) {
 ;
 ; AVX512-LABEL: PR58546:
 ; AVX512:       # %bb.0:
-; AVX512-NEXT:    vxorps %xmm1, %xmm1, %xmm1
-; AVX512-NEXT:    vcmpunordps %zmm1, %zmm0, %k0
+; AVX512-NEXT:    vcmpunordps %zmm0, %zmm0, %k0
 ; AVX512-NEXT:    kshiftrw $8, %k0, %k1
 ; AVX512-NEXT:    kmovd %k0, %eax
 ; AVX512-NEXT:    kmovd %k1, %edx
diff --git a/llvm/test/CodeGen/X86/combine-fcmp-uno-self.ll b/llvm/test/CodeGen/X86/combine-fcmp-uno-self.ll
new file mode 100644
index 0000000000000..dab04e522668d
--- /dev/null
+++ b/llvm/test/CodeGen/X86/combine-fcmp-uno-self.ll
@@ -0,0 +1,83 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+sse2 | FileCheck %s --check-prefixes=SSE
+; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+avx  | FileCheck %s --check-prefixes=AVX
+; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+avx2 | FileCheck %s --check-prefixes=AVX
+
+; Regression test for https://github.com/llvm/llvm-project/issues/202756
+;
+; When lowering an FP UNORD/ORD compare against a known-non-NaN constant
+; (e.g. zeroinitializer, which arises after middle-end canonicalization of
+; `fcmp uno x, x` into `fcmp uno x, 0`), the constant operand should be
+; folded to the other operand, so we emit `cmpp a, a` instead of
+; materializing the zero and comparing against it.
+
+define <4 x double> @uno_self_v4f64(<4 x double> %a) {
+; SSE-LABEL: uno_self_v4f64:
+; SSE:       # %bb.0:
+; SSE-NEXT:    cmpunordpd %xmm0, %xmm0
+; SSE-NEXT:    cmpunordpd %xmm1, %xmm1
+; SSE-NEXT:    retq
+;
+; AVX-LABEL: uno_self_v4f64:
+; AVX:       # %bb.0:
+; AVX-NEXT:    vcmpunordpd %ymm0, %ymm0, %ymm0
+; AVX-NEXT:    retq
+  %cmp = fcmp uno <4 x double> %a, zeroinitializer
+  %sext = sext <4 x i1> %cmp to <4 x i64>
+  %res = bitcast <4 x i64> %sext to <4 x double>
+  ret <4 x double> %res
+}
+
+define <4 x double> @ord_self_v4f64(<4 x double> %a) {
+; SSE-LABEL: ord_self_v4f64:
+; SSE:       # %bb.0:
+; SSE-NEXT:    cmpordpd %xmm0, %xmm0
+; SSE-NEXT:    cmpordpd %xmm1, %xmm1
+; SSE-NEXT:    retq
+;
+; AVX-LABEL: ord_self_v4f64:
+; AVX:       # %bb.0:
+; AVX-NEXT:    vcmpordpd %ymm0, %ymm0, %ymm0
+; AVX-NEXT:    retq
+  %cmp = fcmp ord <4 x double> %a, zeroinitializer
+  %sext = sext <4 x i1> %cmp to <4 x i64>
+  %res = bitcast <4 x i64> %sext to <4 x double>
+  ret <4 x double> %res
+}
+
+define <8 x float> @uno_self_v8f32(<8 x float> %a) {
+; SSE-LABEL: uno_self_v8f32:
+; SSE:       # %bb.0:
+; SSE-NEXT:    cmpunordps %xmm0, %xmm0
+; SSE-NEXT:    cmpunordps %xmm1, %xmm1
+; SSE-NEXT:    retq
+;
+; AVX-LABEL: uno_self_v8f32:
+; AVX:       # %bb.0:
+; AVX-NEXT:    vcmpunordps %ymm0, %ymm0, %ymm0
+; AVX-NEXT:    retq
+  %cmp = fcmp uno <8 x float> %a, zeroinitializer
+  %sext = sext <8 x i1> %cmp to <8 x i32>
+  %res = bitcast <8 x i32> %sext to <8 x float>
+  ret <8 x float> %res
+}
+
+; Negative test: neither operand is a known-non-NaN constant, so the fold
+; must NOT fire. Both %a and %b must still appear in the compare.
+
+define <4 x double> @uno_two_operands_v4f64(<4 x double> %a, <4 x double> %b) {
+; SSE-LABEL: uno_two_operands_v4f64:
+; SSE:       # %bb.0:
+; SSE-NEXT:    cmpunordpd %xmm2, %xmm0
+; SSE-NEXT:    cmpunordpd %xmm3, %xmm1
+; SSE-NEXT:    retq
+;
+; AVX-LABEL: uno_two_operands_v4f64:
+; AVX:       # %bb.0:
+; AVX-NEXT:    vcmpunordpd %ymm1, %ymm0, %ymm0
+; AVX-NEXT:    retq
+  %cmp = fcmp uno <4 x double> %a, %b
+  %sext = sext <4 x i1> %cmp to <4 x i64>
+  %res = bitcast <4 x i64> %sext to <4 x double>
+  ret <4 x double> %res
+}



More information about the llvm-commits mailing list