[llvm] [X86] Fold FP UNORD/ORD compare against known-non-NaN to self-compare (PR #206943)
Simon Pilgrim via llvm-commits
llvm-commits at lists.llvm.org
Wed Jul 1 07:17:05 PDT 2026
https://github.com/RKSimon updated https://github.com/llvm/llvm-project/pull/206943
>From 522c6791315af857609ee10595677fa1c005f203 Mon Sep 17 00:00:00 2001
From: Harishankar <harishankarpp7 at gmail.com>
Date: Wed, 1 Jul 2026 16:12:20 +0530
Subject: [PATCH] [X86] Fold FP UNORD/ORD compare against known-non-NaN to
self-compare
Replace a known-non-NaN operand of an UNORD/ORD FP compare with the
other operand, so we emit `cmpp a, a` instead of materializing the
constant with `vxorpd`. Matches GCC.
Added combine-fcmp-uno-self.ll which covers the folding and a negative case.
Fixes #202756
---
llvm/lib/Target/X86/X86ISelLowering.cpp | 6 ++
llvm/test/CodeGen/X86/bitcast-vector-bool.ll | 26 +++---
.../test/CodeGen/X86/combine-fcmp-uno-self.ll | 83 +++++++++++++++++++
3 files changed, 100 insertions(+), 15 deletions(-)
create mode 100644 llvm/test/CodeGen/X86/combine-fcmp-uno-self.ll
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index cf4e0c081a1b8..b8fb7eed00915 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -24952,6 +24952,12 @@ static SDValue LowerVSETCC(SDValue Op, const X86Subtarget &Subtarget,
SDValue Cmp;
bool IsAlwaysSignaling;
unsigned SSECC = translateX86FSETCC(Cond, Op0, Op1, IsAlwaysSignaling);
+ if ((Cond == ISD::SETO || Cond == ISD::SETUO) && Op0 != Op1) {
+ if (DAG.isKnownNeverNaN(Op1))
+ Op1 = Op0;
+ else if (DAG.isKnownNeverNaN(Op0))
+ Op0 = Op1;
+ }
if (!Subtarget.hasAVX()) {
// TODO: We could use following steps to handle a quiet compare with
// signaling encodings.
diff --git a/llvm/test/CodeGen/X86/bitcast-vector-bool.ll b/llvm/test/CodeGen/X86/bitcast-vector-bool.ll
index f0688482fbeb8..f680520c05cb5 100644
--- a/llvm/test/CodeGen/X86/bitcast-vector-bool.ll
+++ b/llvm/test/CodeGen/X86/bitcast-vector-bool.ll
@@ -1267,12 +1267,11 @@ define i1 @trunc_v128i8_cmp(<128 x i8> %a0) nounwind {
define [2 x i8] @PR58546(<16 x float> %a0) {
; SSE-LABEL: PR58546:
; SSE: # %bb.0:
-; SSE-NEXT: xorps %xmm4, %xmm4
-; SSE-NEXT: cmpunordps %xmm4, %xmm3
-; SSE-NEXT: cmpunordps %xmm4, %xmm2
+; SSE-NEXT: cmpunordps %xmm3, %xmm3
+; SSE-NEXT: cmpunordps %xmm2, %xmm2
; SSE-NEXT: packssdw %xmm3, %xmm2
-; SSE-NEXT: cmpunordps %xmm4, %xmm1
-; SSE-NEXT: cmpunordps %xmm4, %xmm0
+; SSE-NEXT: cmpunordps %xmm1, %xmm1
+; SSE-NEXT: cmpunordps %xmm0, %xmm0
; SSE-NEXT: packssdw %xmm1, %xmm0
; SSE-NEXT: packsswb %xmm2, %xmm0
; SSE-NEXT: pmovmskb %xmm0, %eax
@@ -1284,11 +1283,10 @@ define [2 x i8] @PR58546(<16 x float> %a0) {
;
; AVX1-LABEL: PR58546:
; AVX1: # %bb.0:
-; AVX1-NEXT: vxorps %xmm2, %xmm2, %xmm2
-; AVX1-NEXT: vcmpunordps %ymm2, %ymm1, %ymm1
-; AVX1-NEXT: vextractf128 $1, %ymm1, %xmm3
-; AVX1-NEXT: vpackssdw %xmm3, %xmm1, %xmm1
-; AVX1-NEXT: vcmpunordps %ymm2, %ymm0, %ymm0
+; AVX1-NEXT: vcmpunordps %ymm1, %ymm1, %ymm1
+; AVX1-NEXT: vextractf128 $1, %ymm1, %xmm2
+; AVX1-NEXT: vpackssdw %xmm2, %xmm1, %xmm1
+; AVX1-NEXT: vcmpunordps %ymm0, %ymm0, %ymm0
; AVX1-NEXT: vextractf128 $1, %ymm0, %xmm2
; AVX1-NEXT: vpackssdw %xmm2, %xmm0, %xmm0
; AVX1-NEXT: vpacksswb %xmm1, %xmm0, %xmm0
@@ -1302,9 +1300,8 @@ define [2 x i8] @PR58546(<16 x float> %a0) {
;
; AVX2-LABEL: PR58546:
; AVX2: # %bb.0:
-; AVX2-NEXT: vxorps %xmm2, %xmm2, %xmm2
-; AVX2-NEXT: vcmpunordps %ymm2, %ymm1, %ymm1
-; AVX2-NEXT: vcmpunordps %ymm2, %ymm0, %ymm0
+; AVX2-NEXT: vcmpunordps %ymm1, %ymm1, %ymm1
+; AVX2-NEXT: vcmpunordps %ymm0, %ymm0, %ymm0
; AVX2-NEXT: vpackssdw %ymm1, %ymm0, %ymm0
; AVX2-NEXT: vextracti128 $1, %ymm0, %xmm1
; AVX2-NEXT: vpacksswb %xmm1, %xmm0, %xmm0
@@ -1319,8 +1316,7 @@ define [2 x i8] @PR58546(<16 x float> %a0) {
;
; AVX512-LABEL: PR58546:
; AVX512: # %bb.0:
-; AVX512-NEXT: vxorps %xmm1, %xmm1, %xmm1
-; AVX512-NEXT: vcmpunordps %zmm1, %zmm0, %k0
+; AVX512-NEXT: vcmpunordps %zmm0, %zmm0, %k0
; AVX512-NEXT: kshiftrw $8, %k0, %k1
; AVX512-NEXT: kmovd %k0, %eax
; AVX512-NEXT: kmovd %k1, %edx
diff --git a/llvm/test/CodeGen/X86/combine-fcmp-uno-self.ll b/llvm/test/CodeGen/X86/combine-fcmp-uno-self.ll
new file mode 100644
index 0000000000000..dab04e522668d
--- /dev/null
+++ b/llvm/test/CodeGen/X86/combine-fcmp-uno-self.ll
@@ -0,0 +1,83 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+sse2 | FileCheck %s --check-prefixes=SSE
+; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+avx | FileCheck %s --check-prefixes=AVX
+; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+avx2 | FileCheck %s --check-prefixes=AVX
+
+; Regression test for https://github.com/llvm/llvm-project/issues/202756
+;
+; When lowering an FP UNORD/ORD compare against a known-non-NaN constant
+; (e.g. zeroinitializer, which arises after middle-end canonicalization of
+; `fcmp uno x, x` into `fcmp uno x, 0`), the constant operand should be
+; folded to the other operand, so we emit `cmpp a, a` instead of
+; materializing the zero and comparing against it.
+
+define <4 x double> @uno_self_v4f64(<4 x double> %a) {
+; SSE-LABEL: uno_self_v4f64:
+; SSE: # %bb.0:
+; SSE-NEXT: cmpunordpd %xmm0, %xmm0
+; SSE-NEXT: cmpunordpd %xmm1, %xmm1
+; SSE-NEXT: retq
+;
+; AVX-LABEL: uno_self_v4f64:
+; AVX: # %bb.0:
+; AVX-NEXT: vcmpunordpd %ymm0, %ymm0, %ymm0
+; AVX-NEXT: retq
+ %cmp = fcmp uno <4 x double> %a, zeroinitializer
+ %sext = sext <4 x i1> %cmp to <4 x i64>
+ %res = bitcast <4 x i64> %sext to <4 x double>
+ ret <4 x double> %res
+}
+
+define <4 x double> @ord_self_v4f64(<4 x double> %a) {
+; SSE-LABEL: ord_self_v4f64:
+; SSE: # %bb.0:
+; SSE-NEXT: cmpordpd %xmm0, %xmm0
+; SSE-NEXT: cmpordpd %xmm1, %xmm1
+; SSE-NEXT: retq
+;
+; AVX-LABEL: ord_self_v4f64:
+; AVX: # %bb.0:
+; AVX-NEXT: vcmpordpd %ymm0, %ymm0, %ymm0
+; AVX-NEXT: retq
+ %cmp = fcmp ord <4 x double> %a, zeroinitializer
+ %sext = sext <4 x i1> %cmp to <4 x i64>
+ %res = bitcast <4 x i64> %sext to <4 x double>
+ ret <4 x double> %res
+}
+
+define <8 x float> @uno_self_v8f32(<8 x float> %a) {
+; SSE-LABEL: uno_self_v8f32:
+; SSE: # %bb.0:
+; SSE-NEXT: cmpunordps %xmm0, %xmm0
+; SSE-NEXT: cmpunordps %xmm1, %xmm1
+; SSE-NEXT: retq
+;
+; AVX-LABEL: uno_self_v8f32:
+; AVX: # %bb.0:
+; AVX-NEXT: vcmpunordps %ymm0, %ymm0, %ymm0
+; AVX-NEXT: retq
+ %cmp = fcmp uno <8 x float> %a, zeroinitializer
+ %sext = sext <8 x i1> %cmp to <8 x i32>
+ %res = bitcast <8 x i32> %sext to <8 x float>
+ ret <8 x float> %res
+}
+
+; Negative test: neither operand is a known-non-NaN constant, so the fold
+; must NOT fire. Both %a and %b must still appear in the compare.
+
+define <4 x double> @uno_two_operands_v4f64(<4 x double> %a, <4 x double> %b) {
+; SSE-LABEL: uno_two_operands_v4f64:
+; SSE: # %bb.0:
+; SSE-NEXT: cmpunordpd %xmm2, %xmm0
+; SSE-NEXT: cmpunordpd %xmm3, %xmm1
+; SSE-NEXT: retq
+;
+; AVX-LABEL: uno_two_operands_v4f64:
+; AVX: # %bb.0:
+; AVX-NEXT: vcmpunordpd %ymm1, %ymm0, %ymm0
+; AVX-NEXT: retq
+ %cmp = fcmp uno <4 x double> %a, %b
+ %sext = sext <4 x i1> %cmp to <4 x i64>
+ %res = bitcast <4 x i64> %sext to <4 x double>
+ ret <4 x double> %res
+}
More information about the llvm-commits
mailing list