[llvm] [X86][DAG] remove LowerFCanonicalize (PR #188127)
Gergo Stomfai via llvm-commits
llvm-commits at lists.llvm.org
Thu Mar 26 03:16:09 PDT 2026
https://github.com/stomfaig updated https://github.com/llvm/llvm-project/pull/188127
>From 1c44ff00b3b6debc99a2d24b0cbc3f9365093426 Mon Sep 17 00:00:00 2001
From: stomfaig <stomfaig at gmail.com>
Date: Mon, 23 Mar 2026 20:56:12 +0000
Subject: [PATCH 1/4] remove LowerFCanonicalize
---
.../SelectionDAG/LegalizeVectorOps.cpp | 20 ++++++++
llvm/lib/Target/X86/X86ISelLowering.cpp | 49 ++++++-------------
.../test/CodeGen/SystemZ/canonicalize-vars.ll | 48 ++++--------------
3 files changed, 44 insertions(+), 73 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
index 2409a1f31e26e..59bf364504a05 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
@@ -1073,6 +1073,26 @@ void VectorLegalizer::Expand(SDNode *Node, SmallVectorImpl<SDValue> &Results) {
return;
}
break;
+ case ISD::FCANONICALIZE: {
+ // If the scalar element type has a
+ // Legal/Custom FCANONICALIZE, don't
+ // mess with the vector, fall back.
+ EVT VT = Node->getValueType(0);
+ EVT EltVT = VT.getVectorElementType();
+ if (TLI.getOperationAction(ISD::FCANONICALIZE, EltVT.getSimpleVT()) !=
+ TargetLowering::Expand)
+ break;
+ // Otherwise multiply the whole vector
+ SDLoc DL(Node);
+ SDNodeFlags Flags = Node->getFlags();
+ Flags.setNoFPExcept(true);
+ SDValue One = DAG.getConstantFP(1.0, DL, VT);
+ SDValue Mul =
+ DAG.getNode(ISD::STRICT_FMUL, DL, {VT, MVT::Other},
+ {DAG.getEntryNode(), Node->getOperand(0), One}, Flags);
+ Results.push_back(Mul);
+ return;
+ }
case ISD::FSUB:
ExpandFSUB(Node, Results);
return;
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index ced39e6fa4118..e1038dc509fe7 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -315,8 +315,8 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::FP_TO_UINT_SAT, VT, Custom);
setOperationAction(ISD::FP_TO_SINT_SAT, VT, Custom);
}
- setOperationAction(ISD::FCANONICALIZE, MVT::f32, Custom);
- setOperationAction(ISD::FCANONICALIZE, MVT::f64, Custom);
+ setOperationAction(ISD::FCANONICALIZE, MVT::f32, Expand);
+ setOperationAction(ISD::FCANONICALIZE, MVT::f64, Expand);
if (Subtarget.is64Bit()) {
setOperationAction(ISD::FP_TO_UINT_SAT, MVT::i64, Custom);
setOperationAction(ISD::FP_TO_SINT_SAT, MVT::i64, Custom);
@@ -346,8 +346,8 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
if (!Subtarget.hasSSE2()) {
setOperationAction(ISD::BITCAST , MVT::f32 , Expand);
setOperationAction(ISD::BITCAST , MVT::i32 , Expand);
- setOperationAction(ISD::FCANONICALIZE, MVT::f32, Custom);
- setOperationAction(ISD::FCANONICALIZE, MVT::f64, Custom);
+ setOperationAction(ISD::FCANONICALIZE, MVT::f32, Expand);
+ setOperationAction(ISD::FCANONICALIZE, MVT::f64, Expand);
if (Subtarget.is64Bit()) {
setOperationAction(ISD::BITCAST , MVT::f64 , Expand);
// Without SSE, i64->f64 goes through memory.
@@ -716,7 +716,7 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::STRICT_FROUNDEVEN, MVT::f16, Promote);
setOperationAction(ISD::STRICT_FTRUNC, MVT::f16, Promote);
setOperationAction(ISD::STRICT_FP_ROUND, MVT::f16, Custom);
- setOperationAction(ISD::FCANONICALIZE, MVT::f16, Custom);
+ setOperationAction(ISD::FCANONICALIZE, MVT::f16, Expand);
setOperationAction(ISD::STRICT_FP_EXTEND, MVT::f32, Custom);
setOperationAction(ISD::STRICT_FP_EXTEND, MVT::f64, Custom);
@@ -879,7 +879,7 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::STRICT_FMUL , MVT::f80, Legal);
setOperationAction(ISD::STRICT_FDIV , MVT::f80, Legal);
setOperationAction(ISD::STRICT_FSQRT , MVT::f80, Legal);
- setOperationAction(ISD::FCANONICALIZE , MVT::f80, Custom);
+ setOperationAction(ISD::FCANONICALIZE, MVT::f80, Expand);
if (isTypeLegal(MVT::f16)) {
setOperationAction(ISD::FP_EXTEND, MVT::f80, Custom);
setOperationAction(ISD::STRICT_FP_EXTEND, MVT::f80, Custom);
@@ -942,7 +942,7 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
if (isTypeLegal(MVT::f80)) {
setOperationAction(ISD::FP_ROUND, MVT::f80, Custom);
setOperationAction(ISD::STRICT_FP_ROUND, MVT::f80, Custom);
- setOperationAction(ISD::FCANONICALIZE, MVT::f80, Custom);
+ setOperationAction(ISD::FCANONICALIZE, MVT::f80, Expand);
}
setOperationAction(ISD::SETCC, MVT::f128, Custom);
@@ -1078,11 +1078,11 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::VSELECT, MVT::v4f32, Custom);
setOperationAction(ISD::EXTRACT_VECTOR_ELT, MVT::v4f32, Custom);
setOperationAction(ISD::SELECT, MVT::v4f32, Custom);
- setOperationAction(ISD::FCANONICALIZE, MVT::v4f32, Custom);
+ setOperationAction(ISD::FCANONICALIZE, MVT::v4f32, Expand);
setOperationAction(ISD::LOAD, MVT::v2f32, Custom);
setOperationAction(ISD::STORE, MVT::v2f32, Custom);
- setOperationAction(ISD::FCANONICALIZE, MVT::v2f32, Custom);
+ setOperationAction(ISD::FCANONICALIZE, MVT::v2f32, Expand);
setOperationAction(ISD::STRICT_FADD, MVT::v4f32, Legal);
setOperationAction(ISD::STRICT_FSUB, MVT::v4f32, Legal);
@@ -1145,7 +1145,7 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::UMULO, MVT::v2i32, Custom);
setOperationAction(ISD::FNEG, MVT::v2f64, Custom);
- setOperationAction(ISD::FCANONICALIZE, MVT::v2f64, Custom);
+ setOperationAction(ISD::FCANONICALIZE, MVT::v2f64, Expand);
setOperationAction(ISD::FABS, MVT::v2f64, Custom);
setOperationAction(ISD::FCOPYSIGN, MVT::v2f64, Custom);
@@ -1504,7 +1504,7 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::FMINIMUM, VT, Custom);
setOperationAction(ISD::FMAXIMUMNUM, VT, Custom);
setOperationAction(ISD::FMINIMUMNUM, VT, Custom);
- setOperationAction(ISD::FCANONICALIZE, VT, Custom);
+ setOperationAction(ISD::FCANONICALIZE, VT, Expand);
}
setOperationAction(ISD::LRINT, MVT::v8f32, Custom);
@@ -1791,9 +1791,9 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::FP_TO_UINT, MVT::v2i1, Custom);
setOperationAction(ISD::STRICT_FP_TO_SINT, MVT::v2i1, Custom);
setOperationAction(ISD::STRICT_FP_TO_UINT, MVT::v2i1, Custom);
- setOperationAction(ISD::FCANONICALIZE, MVT::v8f16, Custom);
- setOperationAction(ISD::FCANONICALIZE, MVT::v16f16, Custom);
- setOperationAction(ISD::FCANONICALIZE, MVT::v32f16, Custom);
+ setOperationAction(ISD::FCANONICALIZE, MVT::v8f16, Expand);
+ setOperationAction(ISD::FCANONICALIZE, MVT::v16f16, Expand);
+ setOperationAction(ISD::FCANONICALIZE, MVT::v32f16, Expand);
// There is no byte sized k-register load or store without AVX512DQ.
if (!Subtarget.hasDQI()) {
@@ -1875,7 +1875,7 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::FMA, VT, Legal);
setOperationAction(ISD::STRICT_FMA, VT, Legal);
setOperationAction(ISD::FCOPYSIGN, VT, Custom);
- setOperationAction(ISD::FCANONICALIZE, VT, Custom);
+ setOperationAction(ISD::FCANONICALIZE, VT, Expand);
}
setOperationAction(ISD::LRINT, MVT::v16f32,
Subtarget.hasDQI() ? Legal : Custom);
@@ -34066,24 +34066,6 @@ static SDValue LowerPREFETCH(SDValue Op, const X86Subtarget &Subtarget,
return Op;
}
-static SDValue LowerFCanonicalize(SDValue Op, SelectionDAG &DAG) {
- SDNode *N = Op.getNode();
- SDValue Operand = N->getOperand(0);
- EVT VT = Operand.getValueType();
- SDLoc dl(N);
-
- SDValue One = DAG.getConstantFP(1.0, dl, VT);
-
- // TODO: Fix Crash for bf16 when generating strict_fmul as it
- // leads to a error : SoftPromoteHalfResult #0: t11: bf16,ch = strict_fmul t0,
- // ConstantFP:bf16<APFloat(16256)>, t5 LLVM ERROR: Do not know how to soft
- // promote this operator's result!
- SDValue Chain = DAG.getEntryNode();
- SDValue StrictFmul = DAG.getNode(ISD::STRICT_FMUL, dl, {VT, MVT::Other},
- {Chain, Operand, One});
- return StrictFmul;
-}
-
static StringRef getInstrStrFromOpNo(const SmallVectorImpl<StringRef> &AsmStrs,
unsigned OpNo) {
const APInt Operand(32, OpNo);
@@ -34225,7 +34207,6 @@ SDValue X86TargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const {
case ISD::SRL_PARTS: return LowerShiftParts(Op, DAG);
case ISD::FSHL:
case ISD::FSHR: return LowerFunnelShift(Op, Subtarget, DAG);
- case ISD::FCANONICALIZE: return LowerFCanonicalize(Op, DAG);
case ISD::STRICT_SINT_TO_FP:
case ISD::SINT_TO_FP: return LowerSINT_TO_FP(Op, DAG);
case ISD::STRICT_UINT_TO_FP:
diff --git a/llvm/test/CodeGen/SystemZ/canonicalize-vars.ll b/llvm/test/CodeGen/SystemZ/canonicalize-vars.ll
index d0f3414e89497..e6659d385ae5f 100644
--- a/llvm/test/CodeGen/SystemZ/canonicalize-vars.ll
+++ b/llvm/test/CodeGen/SystemZ/canonicalize-vars.ll
@@ -205,17 +205,8 @@ define <8 x half> @canonicalize_v8f16(<8 x half> %a) nounwind {
define <4 x float> @canonicalize_v4f32(<4 x float> %a) {
; Z16-LABEL: canonicalize_v4f32:
; Z16: # %bb.0:
-; Z16-NEXT: vrepf %v0, %v24, 3
-; Z16-NEXT: vgmf %v1, 2, 8
-; Z16-NEXT: vrepf %v2, %v24, 2
-; Z16-NEXT: meebr %f0, %f1
-; Z16-NEXT: meebr %f2, %f1
-; Z16-NEXT: vrepf %v3, %v24, 1
-; Z16-NEXT: vmrhf %v0, %v2, %v0
-; Z16-NEXT: wfmsb %f2, %v24, %f1
-; Z16-NEXT: wfmsb %f1, %f3, %f1
-; Z16-NEXT: vmrhf %v1, %v2, %v1
-; Z16-NEXT: vmrhg %v24, %v1, %v0
+; Z16-NEXT: vgmf %v0, 2, 8
+; Z16-NEXT: vfmsb %v24, %v24, %v0
; Z16-NEXT: br %r14
%canonicalized = call <4 x float> @llvm.canonicalize.v4f32(<4 x float> %a)
ret <4 x float> %canonicalized
@@ -225,14 +216,8 @@ define <4 x double> @canonicalize_v4f64(<4 x double> %a) {
; Z16-LABEL: canonicalize_v4f64:
; Z16: # %bb.0:
; Z16-NEXT: vgmg %v0, 2, 11
-; Z16-NEXT: vrepg %v2, %v24, 1
-; Z16-NEXT: wfmdb %f1, %v24, %f0
-; Z16-NEXT: mdbr %f2, %f0
-; Z16-NEXT: vmrhg %v24, %v1, %v2
-; Z16-NEXT: vrepg %v2, %v26, 1
-; Z16-NEXT: wfmdb %f1, %v26, %f0
-; Z16-NEXT: wfmdb %f0, %f2, %f0
-; Z16-NEXT: vmrhg %v26, %v1, %v0
+; Z16-NEXT: vfmdb %v24, %v24, %v0
+; Z16-NEXT: vfmdb %v26, %v26, %v0
; Z16-NEXT: br %r14
%canonicalized = call <4 x double> @llvm.canonicalize.v4f64(<4 x double> %a)
ret <4 x double> %canonicalized
@@ -358,17 +343,8 @@ define void @canonicalize_ptr_v4f32(ptr %out) {
; Z16-LABEL: canonicalize_ptr_v4f32:
; Z16: # %bb.0:
; Z16-NEXT: vl %v0, 0(%r2), 3
-; Z16-NEXT: vrepf %v1, %v0, 3
-; Z16-NEXT: vgmf %v2, 2, 8
-; Z16-NEXT: vrepf %v3, %v0, 2
-; Z16-NEXT: meebr %f1, %f2
-; Z16-NEXT: meebr %f3, %f2
-; Z16-NEXT: vmrhf %v1, %v3, %v1
-; Z16-NEXT: wfmsb %f3, %f0, %f2
-; Z16-NEXT: vrepf %v0, %v0, 1
-; Z16-NEXT: meebr %f0, %f2
-; Z16-NEXT: vmrhf %v0, %v3, %v0
-; Z16-NEXT: vmrhg %v0, %v0, %v1
+; Z16-NEXT: vgmf %v1, 2, 8
+; Z16-NEXT: vfmsb %v0, %v0, %v1
; Z16-NEXT: vst %v0, 0(%r2), 3
; Z16-NEXT: br %r14
%val = load <4 x float>, ptr %out
@@ -380,17 +356,11 @@ define void @canonicalize_ptr_v4f32(ptr %out) {
define void @canonicalize_ptr_v4f64(ptr %out) {
; Z16-LABEL: canonicalize_ptr_v4f64:
; Z16: # %bb.0:
+; Z16-NEXT: vl %v0, 0(%r2), 4
; Z16-NEXT: vl %v1, 16(%r2), 4
; Z16-NEXT: vgmg %v2, 2, 11
-; Z16-NEXT: wfmdb %f3, %f1, %f2
-; Z16-NEXT: vrepg %v1, %v1, 1
-; Z16-NEXT: mdbr %f1, %f2
-; Z16-NEXT: vl %v0, 0(%r2), 4
-; Z16-NEXT: vmrhg %v1, %v3, %v1
-; Z16-NEXT: wfmdb %f3, %f0, %f2
-; Z16-NEXT: vrepg %v0, %v0, 1
-; Z16-NEXT: mdbr %f0, %f2
-; Z16-NEXT: vmrhg %v0, %v3, %v0
+; Z16-NEXT: vfmdb %v1, %v1, %v2
+; Z16-NEXT: vfmdb %v0, %v0, %v2
; Z16-NEXT: vst %v0, 0(%r2), 4
; Z16-NEXT: vst %v1, 16(%r2), 4
; Z16-NEXT: br %r14
>From 78534f243101d7e49da89608153dc3f3cad7179e Mon Sep 17 00:00:00 2001
From: stomfaig <stomfaig at gmail.com>
Date: Wed, 25 Mar 2026 23:35:18 +0000
Subject: [PATCH 2/4] remove redundant action annotations
---
llvm/lib/Target/X86/X86ISelLowering.cpp | 7 +------
1 file changed, 1 insertion(+), 6 deletions(-)
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index e1038dc509fe7..6ba46a8bcecc2 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -315,8 +315,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::FP_TO_UINT_SAT, VT, Custom);
setOperationAction(ISD::FP_TO_SINT_SAT, VT, Custom);
}
- setOperationAction(ISD::FCANONICALIZE, MVT::f32, Expand);
- setOperationAction(ISD::FCANONICALIZE, MVT::f64, Expand);
if (Subtarget.is64Bit()) {
setOperationAction(ISD::FP_TO_UINT_SAT, MVT::i64, Custom);
setOperationAction(ISD::FP_TO_SINT_SAT, MVT::i64, Custom);
@@ -345,9 +343,7 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
// TODO: when we have SSE, these could be more efficient, by using movd/movq.
if (!Subtarget.hasSSE2()) {
setOperationAction(ISD::BITCAST , MVT::f32 , Expand);
- setOperationAction(ISD::BITCAST , MVT::i32 , Expand);
- setOperationAction(ISD::FCANONICALIZE, MVT::f32, Expand);
- setOperationAction(ISD::FCANONICALIZE, MVT::f64, Expand);
+ setOperationAction(ISD::BITCAST, MVT::i32, Expand);
if (Subtarget.is64Bit()) {
setOperationAction(ISD::BITCAST , MVT::f64 , Expand);
// Without SSE, i64->f64 goes through memory.
@@ -716,7 +712,6 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
setOperationAction(ISD::STRICT_FROUNDEVEN, MVT::f16, Promote);
setOperationAction(ISD::STRICT_FTRUNC, MVT::f16, Promote);
setOperationAction(ISD::STRICT_FP_ROUND, MVT::f16, Custom);
- setOperationAction(ISD::FCANONICALIZE, MVT::f16, Expand);
setOperationAction(ISD::STRICT_FP_EXTEND, MVT::f32, Custom);
setOperationAction(ISD::STRICT_FP_EXTEND, MVT::f64, Custom);
>From 7b357831fdc400d7c15fd5ffbfa4298dc1430e83 Mon Sep 17 00:00:00 2001
From: stomfaig <stomfaig at gmail.com>
Date: Thu, 26 Mar 2026 10:15:01 +0000
Subject: [PATCH 3/4] undo formatting change
---
llvm/lib/Target/X86/X86ISelLowering.cpp | 2 +-
llvm/test.ll | 4 ++++
2 files changed, 5 insertions(+), 1 deletion(-)
create mode 100644 llvm/test.ll
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 6ba46a8bcecc2..b7dd12cad53ca 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -343,7 +343,7 @@ X86TargetLowering::X86TargetLowering(const X86TargetMachine &TM,
// TODO: when we have SSE, these could be more efficient, by using movd/movq.
if (!Subtarget.hasSSE2()) {
setOperationAction(ISD::BITCAST , MVT::f32 , Expand);
- setOperationAction(ISD::BITCAST, MVT::i32, Expand);
+ setOperationAction(ISD::BITCAST , MVT::i32 , Expand);
if (Subtarget.is64Bit()) {
setOperationAction(ISD::BITCAST , MVT::f64 , Expand);
// Without SSE, i64->f64 goes through memory.
diff --git a/llvm/test.ll b/llvm/test.ll
new file mode 100644
index 0000000000000..a1f34f2c4b9b1
--- /dev/null
+++ b/llvm/test.ll
@@ -0,0 +1,4 @@
+define <4 x float> @canon_fp32_varargsv4f32(<4 x float> %a) {
+ %canonicalized = call <4 x float> @llvm.canonicalize.v4f32(<4 x float> %a)
+ ret <4 x float> %canonicalized
+}
\ No newline at end of file
>From 637cecf3bae0cbae9620e21acc07fb1c53656e70 Mon Sep 17 00:00:00 2001
From: stomfaig <stomfaig at gmail.com>
Date: Thu, 26 Mar 2026 10:15:56 +0000
Subject: [PATCH 4/4] remove random test file
---
llvm/test.ll | 4 ----
1 file changed, 4 deletions(-)
delete mode 100644 llvm/test.ll
diff --git a/llvm/test.ll b/llvm/test.ll
deleted file mode 100644
index a1f34f2c4b9b1..0000000000000
--- a/llvm/test.ll
+++ /dev/null
@@ -1,4 +0,0 @@
-define <4 x float> @canon_fp32_varargsv4f32(<4 x float> %a) {
- %canonicalized = call <4 x float> @llvm.canonicalize.v4f32(<4 x float> %a)
- ret <4 x float> %canonicalized
-}
\ No newline at end of file
More information about the llvm-commits
mailing list