[llvm] [NVPTX] Decide .ftz on the whole denormal mode, not just the output half (PR #222645)
Ashutosh Kumar Singh via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 10 07:03:12 PDT 2026
https://github.com/Ashutosh0x updated https://github.com/llvm/llvm-project/pull/222645
>From 87169cd5d28df4c7ff6860e80dfc69ce06b38060 Mon Sep 17 00:00:00 2001
From: Ashutosh0x <161562995+Ashutosh0x at users.noreply.github.com>
Date: Thu, 10 Sep 2026 18:34:33 +0530
Subject: [PATCH] [NVPTX] Decide .ftz on the whole denormal mode, not just the
output half
PTX .ftz flushes subnormal inputs *and* results to sign-preserving zero, so it
faithfully implements exactly one denormal mode: preservesign on both halves.
useF32FTZ() decided on the output mode alone, which is wrong in both
directions -- flushing an output is permitted but never mandated, while an ieee
input mode forbids flushing the inputs. So preservesign|ieee emitted
add.rn.ftz.f32 and flushed inputs the mode declares ieee. Compare the whole
DenormalMode instead, the way AMDGPU's atomicIgnoresDenormalModeOrFPModeIsFTZ
does.
NVPTX also had no ISD::FCANONICALIZE lowering, so llvm.canonicalize -- the
remedy LangRef names for a mandated input flush -- fell to the generic
multiply-by-1.0 expansion, whose .ftz likewise comes from useF32FTZ(). Under
ieee|preservesign that produced a plain mul.rn.f32, making the canonicalize a
no-op. Lower FCANONICALIZE for f32 keyed on the input mode alone.
Only the IR attribute can reach a mixed mode: -denormal-fp-math[-f32]= is a
single DenormalModeKind and always sets Input == Output, so no in-tree test
changes.
Partially addresses #222484.
Assisted-by: Claude Code (Anthropic)
---
llvm/lib/Target/NVPTX/NVPTXISelLowering.cpp | 49 ++++++++++-
llvm/lib/Target/NVPTX/NVPTXISelLowering.h | 1 +
.../NVPTX/denormal-fpenv-mixed-modes.ll | 83 +++++++++++++++++++
3 files changed, 132 insertions(+), 1 deletion(-)
create mode 100644 llvm/test/CodeGen/NVPTX/denormal-fpenv-mixed-modes.ll
diff --git a/llvm/lib/Target/NVPTX/NVPTXISelLowering.cpp b/llvm/lib/Target/NVPTX/NVPTXISelLowering.cpp
index a21af472576cd..4e788854373a4 100644
--- a/llvm/lib/Target/NVPTX/NVPTXISelLowering.cpp
+++ b/llvm/lib/Target/NVPTX/NVPTXISelLowering.cpp
@@ -156,8 +156,22 @@ bool NVPTXTargetLowering::usePrecSqrtF32(const SDNode *N) const {
return true;
}
+// PTX .ftz flushes subnormal *inputs and results* to sign-preserving zero, so
+// it faithfully implements exactly one denormal mode: preservesign on both
+// halves. Deciding on the output mode alone is wrong in both directions:
+// flushing an output is permitted but never mandated, while an ieee input mode
+// forbids flushing the inputs. So preservesign|ieee must not take .ftz even
+// though the output permission is there for the taking.
bool NVPTXTargetLowering::useF32FTZ(const MachineFunction &MF) const {
- return MF.getDenormalMode(APFloat::IEEEsingle()).Output ==
+ return MF.getDenormalMode(APFloat::IEEEsingle()) ==
+ DenormalMode::getPreserveSign();
+}
+
+/// True when the f32 input denormal mode mandates that subnormal inputs be
+/// treated as zero. Unlike useF32FTZ() this deliberately ignores the output
+/// mode: input flushing is mandated on its own.
+static bool flushF32Inputs(const MachineFunction &MF) {
+ return MF.getDenormalMode(APFloat::IEEEsingle()).Input ==
DenormalMode::PreserveSign;
}
@@ -1029,6 +1043,10 @@ NVPTXTargetLowering::NVPTXTargetLowering(const NVPTXTargetMachine &TM,
setOperationAction(ISD::FCOPYSIGN, MVT::f32, Custom);
setOperationAction(ISD::FCOPYSIGN, MVT::f64, Custom);
+ // Only f32 needs custom handling; see LowerFCANONICALIZE. Everything else
+ // keeps the generic multiply-by-1.0 expansion.
+ setOperationAction(ISD::FCANONICALIZE, MVT::f32, Custom);
+
// These map to corresponding instructions for f32/f64. f16 must be
// promoted to f32. v2f16 is expanded to f16, which is then promoted
// to f32.
@@ -2255,6 +2273,33 @@ SDValue NVPTXTargetLowering::LowerFCOPYSIGN(SDValue Op,
return DAG.getNode(NVPTXISD::FCOPYSIGN, DL, VT, In1, In2);
}
+// NVPTX has no canonicalize instruction, so llvm.canonicalize falls to the
+// generic multiply-by-1.0 expansion. That expansion only flushes subnormals if
+// the multiply it produces carries .ftz, and whether it does is decided by
+// useF32FTZ() -- which is a property of the whole denormal mode. When the input
+// mode mandates flushing but the output mode does not permit it, the generic
+// expansion picks a plain mul.rn.f32 and the canonicalize silently becomes a
+// no-op, which is precisely the case LangRef points at llvm.canonicalize to
+// handle. Emit an explicitly-.ftz multiply instead.
+SDValue NVPTXTargetLowering::LowerFCANONICALIZE(SDValue Op,
+ SelectionDAG &DAG) const {
+ assert(Op.getValueType() == MVT::f32 && "only f32 is marked Custom");
+ const MachineFunction &MF = DAG.getMachineFunction();
+
+ // Where useF32FTZ() already agrees with the input mode, the generic
+ // expansion selects the right multiply on its own. Note that a positivezero
+ // input mode is left to it as well: .ftz preserves the sign, so it cannot
+ // implement that mode either way.
+ if (!flushF32Inputs(MF) || useF32FTZ(MF))
+ return SDValue();
+
+ SDLoc DL(Op);
+ return DAG.getNode(
+ ISD::INTRINSIC_WO_CHAIN, DL, MVT::f32,
+ DAG.getConstant(Intrinsic::nvvm_mul_rn_ftz_f, DL, MVT::i32),
+ Op.getOperand(0), DAG.getConstantFP(1.0, DL, MVT::f32));
+}
+
SDValue NVPTXTargetLowering::LowerFROUND(SDValue Op, SelectionDAG &DAG) const {
EVT VT = Op.getValueType();
@@ -3494,6 +3539,8 @@ NVPTXTargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const {
return LowerFROUND(Op, DAG);
case ISD::FCOPYSIGN:
return LowerFCOPYSIGN(Op, DAG);
+ case ISD::FCANONICALIZE:
+ return LowerFCANONICALIZE(Op, DAG);
case ISD::SINT_TO_FP:
case ISD::UINT_TO_FP:
return LowerINT_TO_FP(Op, DAG);
diff --git a/llvm/lib/Target/NVPTX/NVPTXISelLowering.h b/llvm/lib/Target/NVPTX/NVPTXISelLowering.h
index 4942408f6449b..9a00e97581588 100644
--- a/llvm/lib/Target/NVPTX/NVPTXISelLowering.h
+++ b/llvm/lib/Target/NVPTX/NVPTXISelLowering.h
@@ -198,6 +198,7 @@ class NVPTXTargetLowering : public TargetLowering {
SDValue LowerVECTOR_SHUFFLE(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerFCOPYSIGN(SDValue Op, SelectionDAG &DAG) const;
+ SDValue LowerFCANONICALIZE(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerFROUND(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerFROUND32(SDValue Op, SelectionDAG &DAG) const;
diff --git a/llvm/test/CodeGen/NVPTX/denormal-fpenv-mixed-modes.ll b/llvm/test/CodeGen/NVPTX/denormal-fpenv-mixed-modes.ll
new file mode 100644
index 0000000000000..b6951ae44d5b0
--- /dev/null
+++ b/llvm/test/CodeGen/NVPTX/denormal-fpenv-mixed-modes.ll
@@ -0,0 +1,83 @@
+; RUN: llc < %s -mcpu=sm_90 -mattr=+ptx78 -verify-machineinstrs | FileCheck %s
+; RUN: %if ptxas %{ llc < %s -mcpu=sm_90 -mattr=+ptx78 | %ptxas-verify -arch=sm_90 %}
+
+; PTX .ftz flushes subnormal inputs *and* results, so it implements only the
+; preservesign|preservesign mode. Flushing an output is permitted but never
+; mandated, while flushing an ieee input is forbidden; conversely a preservesign
+; input mode mandates a flush that a plain op does not perform.
+
+target triple = "nvptx64-nvidia-cuda"
+
+declare float @llvm.canonicalize.f32(float)
+
+; The output permission alone must not buy .ftz: the inputs are ieee.
+define float @fadd_preservesign_ieee(float %x, float %y) #0 {
+; CHECK-LABEL: fadd_preservesign_ieee(
+; CHECK: add.rn.f32
+; CHECK-NOT: add.rn.ftz.f32
+ %r = fadd float %x, %y
+ ret float %r
+}
+
+; Flushing the result is not permitted here, so .ftz is out.
+define float @fadd_ieee_preservesign(float %x, float %y) #1 {
+; CHECK-LABEL: fadd_ieee_preservesign(
+; CHECK: add.rn.f32
+; CHECK-NOT: add.rn.ftz.f32
+ %r = fadd float %x, %y
+ ret float %r
+}
+
+; The one mode .ftz actually implements. This is what a command line produces.
+define float @fadd_preservesign_both(float %x, float %y) #2 {
+; CHECK-LABEL: fadd_preservesign_both(
+; CHECK: add.rn.ftz.f32
+ %r = fadd float %x, %y
+ ret float %r
+}
+
+define float @fadd_ieee_both(float %x, float %y) #3 {
+; CHECK-LABEL: fadd_ieee_both(
+; CHECK: add.rn.f32
+; CHECK-NOT: add.rn.ftz.f32
+ %r = fadd float %x, %y
+ ret float %r
+}
+
+; A mandated input flush that no plain op performs: canonicalize has to carry
+; .ftz even though the output mode forbids flushing results.
+define float @canonicalize_ieee_preservesign(float %x) #1 {
+; CHECK-LABEL: canonicalize_ieee_preservesign(
+; CHECK: mul.rn.ftz.f32
+ %r = call float @llvm.canonicalize.f32(float %x)
+ ret float %r
+}
+
+; Inputs are ieee, so canonicalize must not flush them.
+define float @canonicalize_preservesign_ieee(float %x) #0 {
+; CHECK-LABEL: canonicalize_preservesign_ieee(
+; CHECK: mul.rn.f32
+; CHECK-NOT: mul.rn.ftz.f32
+ %r = call float @llvm.canonicalize.f32(float %x)
+ ret float %r
+}
+
+define float @canonicalize_preservesign_both(float %x) #2 {
+; CHECK-LABEL: canonicalize_preservesign_both(
+; CHECK: mul.rn.ftz.f32
+ %r = call float @llvm.canonicalize.f32(float %x)
+ ret float %r
+}
+
+define float @canonicalize_ieee_both(float %x) #3 {
+; CHECK-LABEL: canonicalize_ieee_both(
+; CHECK: mul.rn.f32
+; CHECK-NOT: mul.rn.ftz.f32
+ %r = call float @llvm.canonicalize.f32(float %x)
+ ret float %r
+}
+
+attributes #0 = { denormal_fpenv(float: preservesign|ieee) }
+attributes #1 = { denormal_fpenv(float: ieee|preservesign) }
+attributes #2 = { denormal_fpenv(float: preservesign|preservesign) }
+attributes #3 = { denormal_fpenv(float: ieee|ieee) }
More information about the llvm-commits
mailing list