[llvm] [LLVM][CodeGen][SVE] Add lowering for bfloat strict-fp cast operators. (PR #223709)
Paul Walker via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 15 07:05:08 PDT 2026
https://github.com/paulwalker-arm created https://github.com/llvm/llvm-project/pull/223709
The majority of the changes are just a case of ensuring the chain is routed correctly and the matching STRICT passthrough node is used.
NOTE: At present full strict-fp support has a minimum requirement of +sve2+bf16, otherwise we lack the necessary cast instructions. Of these +bf16 is fundamental whereas +sve2 is only required for double->bfloat.
NOTE: The test layout is due to follow-on work where I figured it better to have a more complete structure now so the later PRs have less churn.
>From f06afe250ba363eccd4c94efaf41ea9fcff099a4 Mon Sep 17 00:00:00 2001
From: Paul Walker <paul.walker at arm.com>
Date: Tue, 8 Sep 2026 14:12:39 +0100
Subject: [PATCH] [LLVM][CodeGen][SVE] Add lowering for bfloat strict-fp cast
operators.
NOTE: At present full strict-fp support has a minimum requirement of
+sve2+bf16, otherwise we lack the necessary cast instructions. Of these
+bf16 is fundamental whereas +sve2 is only required for double->bfloat.
NOTE: The test layout is due to follow-on work where I figured it better
to have a more complete structure now so the later PRs have less churn.
---
.../SelectionDAG/LegalizeVectorOps.cpp | 6 +
.../Target/AArch64/AArch64ISelLowering.cpp | 78 ++--
.../AArch64/sve-bf-constrained-intrinsics.ll | 409 ++++++++++++++++++
3 files changed, 462 insertions(+), 31 deletions(-)
create mode 100644 llvm/test/CodeGen/AArch64/sve-bf-constrained-intrinsics.ll
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
index 5e3f252fdd3d4..b41f010d4a2aa 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
@@ -2252,6 +2252,12 @@ bool VectorLegalizer::tryExpandVecMathCall(
void VectorLegalizer::UnrollStrictFPOp(SDNode *Node,
SmallVectorImpl<SDValue> &Results) {
EVT VT = Node->getValueType(0);
+
+ // Cannot unroll a scalable vector. Delay error reporting until the final
+ // operation legalisation phase to maximise the chances of removing the node.
+ if (VT.isScalableVector())
+ return;
+
EVT EltVT = VT.getVectorElementType();
unsigned NumElems = VT.getVectorNumElements();
unsigned NumOpers = Node->getNumOperands();
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 2a50c0c474ab9..a4c834aa3bd7c 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -1972,8 +1972,8 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::FABS, VT, Custom);
setOperationAction(ISD::FCOPYSIGN, VT, Custom);
setOperationAction(ISD::FNEG, VT, Custom);
- setOperationAction(ISD::FP_EXTEND, VT, Custom);
- setOperationAction(ISD::FP_ROUND, VT, Custom);
+ setOperationAction({ISD::FP_EXTEND, ISD::STRICT_FP_EXTEND}, VT, Custom);
+ setOperationAction({ISD::FP_ROUND, ISD::STRICT_FP_ROUND}, VT, Custom);
setOperationAction(ISD::MLOAD, VT, Custom);
setOperationAction(ISD::INSERT_SUBVECTOR, VT, Custom);
setOperationAction(ISD::SELECT, VT, Custom);
@@ -4962,18 +4962,23 @@ SDValue AArch64TargetLowering::LowerFP_EXTEND(SDValue Op,
bool IsStrict = Op->isStrictFPOpcode();
if (VT.isScalableVector()) {
- SDValue SrcVal = Op.getOperand(0);
+ SDValue SrcVal = Op.getOperand(IsStrict ? 1 : 0);
if (VT == MVT::nxv2f64 && SrcVal.getValueType() == MVT::nxv2bf16) {
- // TODO: Missing support for bfloat strict-fp operations.
- if (IsStrict)
- return SDValue();
-
- // Break conversion in two with the first part converting to f32 and the
- // second using native f32->VT instructions.
SDLoc DL(Op);
- return DAG.getNode(ISD::FP_EXTEND, DL, VT,
- DAG.getNode(ISD::FP_EXTEND, DL, MVT::nxv2f32, SrcVal));
+
+ // Split cast into two phases, using float as the intermediary.
+ SDVTList CvtF32VTs = IsStrict ? DAG.getVTList(MVT::nxv2f32, MVT::Other)
+ : DAG.getVTList(MVT::nxv2f32);
+ SmallVector<SDValue> CvtF32Ops(Op->ops());
+ SDValue CvtF32 = DAG.getNode(Op.getOpcode(), DL, CvtF32VTs, CvtF32Ops);
+
+ // Extend from float to double.
+ SmallVector<SDValue, 2> CvtF64Ops;
+ if (IsStrict)
+ CvtF64Ops.push_back(CvtF32.getValue(1)); // Chain
+ CvtF64Ops.push_back(CvtF32);
+ return DAG.getNode(Op.getOpcode(), DL, Op->getVTList(), CvtF64Ops);
}
return LowerToPredicatedOp(Op, DAG,
@@ -5023,15 +5028,11 @@ SDValue AArch64TargetLowering::LowerFP_ROUND(SDValue Op,
if (SrcVT == MVT::nxv8f32)
return Op;
+ unsigned MergePasthruOpc = IsStrict
+ ? AArch64ISD::STRICT_FP_ROUND_MERGE_PASSTHRU
+ : AArch64ISD::FP_ROUND_MERGE_PASSTHRU;
if (VT.getScalarType() != MVT::bf16)
- return LowerToPredicatedOp(
- Op, DAG,
- IsStrict ? AArch64ISD::STRICT_FP_ROUND_MERGE_PASSTHRU
- : AArch64ISD::FP_ROUND_MERGE_PASSTHRU);
-
- // TODO: Missing support for bfloat strict-fp operations.
- if (IsStrict)
- return SDValue();
+ return LowerToPredicatedOp(Op, DAG, MergePasthruOpc);
SDLoc DL(Op);
constexpr EVT I32 = MVT::nxv4i32;
@@ -5042,8 +5043,7 @@ SDValue AArch64TargetLowering::LowerFP_ROUND(SDValue Op,
if (SrcVT == MVT::nxv2f32 || SrcVT == MVT::nxv4f32) {
if (Subtarget->hasBF16())
- return LowerToPredicatedOp(Op, DAG,
- AArch64ISD::FP_ROUND_MERGE_PASSTHRU);
+ return LowerToPredicatedOp(Op, DAG, MergePasthruOpc);
Narrow = getSVESafeBitCast(I32, SrcVal, DAG);
@@ -5052,20 +5052,36 @@ SDValue AArch64TargetLowering::LowerFP_ROUND(SDValue Op,
NaN = DAG.getNode(ISD::OR, DL, I32, Narrow, ImmV(0x400000));
} else if (SrcVT == MVT::nxv2f64 &&
(Subtarget->hasSVE2() || Subtarget->isStreamingSVEAvailable())) {
- // Round to float without introducing rounding errors and try again.
- SDValue Pg = getPredicateForVector(DAG, DL, MVT::nxv2f32);
- Narrow = DAG.getNode(AArch64ISD::FCVTX_MERGE_PASSTHRU, DL, MVT::nxv2f32,
- Pg, SrcVal, DAG.getPOISON(MVT::nxv2f32));
-
- SmallVector<SDValue, 3> NewOps;
+ // Split cast into two phases, using float as the intermediary.
+ SDVTList CvtF32VTs = IsStrict ? DAG.getVTList(MVT::nxv2f32, MVT::Other)
+ : DAG.getVTList(MVT::nxv2f32);
+
+ // Round to float without introducing rounding errors.
+ unsigned CvtF32Opc = IsStrict ? AArch64ISD::STRICT_FCVTX_MERGE_PASSTHRU
+ : AArch64ISD::FCVTX_MERGE_PASSTHRU;
+ SmallVector<SDValue, 3> CvtF32Ops;
if (IsStrict)
- NewOps.push_back(Op.getOperand(0));
- NewOps.push_back(Narrow);
- NewOps.push_back(Op.getOperand(IsStrict ? 2 : 1));
- return DAG.getNode(Op.getOpcode(), DL, VT, NewOps, Op->getFlags());
+ CvtF32Ops.push_back(Op.getOperand(0)); // Chain
+ CvtF32Ops.push_back(getPredicateForVector(DAG, DL, MVT::nxv2f32));
+ CvtF32Ops.push_back(SrcVal);
+ CvtF32Ops.push_back(DAG.getPOISON(MVT::nxv2f32));
+ SDValue CvtF32 = DAG.getNode(CvtF32Opc, DL, CvtF32VTs, CvtF32Ops);
+
+ // Round from float to bfloat.
+ SmallVector<SDValue, 3> CvtBfOps;
+ if (IsStrict)
+ CvtBfOps.push_back(CvtF32.getValue(1)); // Chain
+ CvtBfOps.push_back(CvtF32);
+ CvtBfOps.push_back(Op.getOperand(IsStrict ? 2 : 1)); // Trunc
+ return DAG.getNode(Op.getOpcode(), DL, Op->getVTList(), CvtBfOps,
+ Op->getFlags());
} else
return SDValue();
+ // TODO: Missing support for bfloat strict-fp operations.
+ if (IsStrict)
+ return SDValue();
+
if (!Trunc) {
SDValue Lsb = DAG.getNode(ISD::SRL, DL, I32, Narrow, ImmV(16));
Lsb = DAG.getNode(ISD::AND, DL, I32, Lsb, ImmV(1));
diff --git a/llvm/test/CodeGen/AArch64/sve-bf-constrained-intrinsics.ll b/llvm/test/CodeGen/AArch64/sve-bf-constrained-intrinsics.ll
new file mode 100644
index 0000000000000..d894a90da599b
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-bf-constrained-intrinsics.ll
@@ -0,0 +1,409 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mattr=+sve,+bf16 < %s -o - | FileCheck %s -check-prefixes=CHECK,BF16,NONSTREAMING
+; RUN: llc -mattr=+sme --force-streaming < %s -o - | FileCheck %s -check-prefixes=CHECK,BF16,STREAMING
+
+; RUN: llc -mattr=+sve,+bf16,+sve-b16b16 < %s -o - | FileCheck %s --check-prefixes=CHECK,SVE-B16B16,SVE-B16B16_NONSTREAMING
+; RUN: llc -mattr=+sme2,+sve-b16b16 --force-streaming < %s -o - | FileCheck %s --check-prefixes=CHECK,SVE-B16B16,SVE-B16B16_STREAMING
+
+; NOTE: Full strict-fp support requires SVE variants of FCVTX and BFCVT.
+
+target triple = "aarch64-unknown-linux-gnu"
+
+;
+; constrained.ceil (TODO)
+;
+
+;
+; constrained.fadd (TODO)
+;
+
+;
+; constrained.fcmp (TODO)
+;
+
+;
+; constrained.fcmps (TODO)
+;
+
+;
+; constrained.fdiv (TODO)
+;
+
+;
+; constrained.floor (TODO)
+;
+
+;
+; constrained.fma (TODO)
+;
+
+;
+; constrained.fmul (TODO)
+;
+
+;
+; constrained.fmuladd (TODO)
+;
+
+;
+; constrained.fpext
+;
+
+define <vscale x 2 x float> @fpext_nxv2bf16_to_nxv2f32(<vscale x 2 x bfloat> %a) {
+; CHECK-LABEL: fpext_nxv2bf16_to_nxv2f32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: lsl z0.s, z0.s, #16
+; CHECK-NEXT: ret
+ %r = call <vscale x 2 x float> @llvm.experimental.constrained.fpext(<vscale x 2 x bfloat> %a, metadata !"fpexcept.strict")
+ ret <vscale x 2 x float> %r
+}
+
+define <vscale x 2 x double> @fpext_nxv2bf16_to_nxv2f64(<vscale x 2 x bfloat> %a) {
+; CHECK-LABEL: fpext_nxv2bf16_to_nxv2f64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: lsl z0.s, z0.s, #16
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: fcvt z0.d, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 2 x double> @llvm.experimental.constrained.fpext(<vscale x 2 x bfloat> %a, metadata !"fpexcept.strict")
+ ret <vscale x 2 x double> %r
+}
+
+;
+; constrained.fptosi
+;
+
+define <vscale x 2 x i64> @fptosi_nxv2bf16_to_nxv2i64(<vscale x 2 x bfloat> %a) {
+; CHECK-LABEL: fptosi_nxv2bf16_to_nxv2i64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: lsl z0.s, z0.s, #16
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: fcvtzs z0.d, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 2 x i64> @llvm.experimental.constrained.fptosi(<vscale x 2 x bfloat> %a, metadata !"fpexcept.strict")
+ ret <vscale x 2 x i64> %r
+}
+
+define <vscale x 4 x i32> @fptosi_nxv4bf16_to_nxv4i32(<vscale x 4 x bfloat> %a) {
+; CHECK-LABEL: fptosi_nxv4bf16_to_nxv4i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: lsl z0.s, z0.s, #16
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: fcvtzs z0.s, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 4 x i32> @llvm.experimental.constrained.fptosi(<vscale x 4 x bfloat> %a, metadata !"fpexcept.strict")
+ ret <vscale x 4 x i32> %r
+}
+
+define <vscale x 8 x i16> @fptosi_nxv8bf16_to_nxv8i16(<vscale x 8 x bfloat> %a) {
+; NONSTREAMING-LABEL: fptosi_nxv8bf16_to_nxv8i16:
+; NONSTREAMING: // %bb.0:
+; NONSTREAMING-NEXT: movi v1.2d, #0000000000000000
+; NONSTREAMING-NEXT: ptrue p0.s
+; NONSTREAMING-NEXT: zip1 z2.h, z1.h, z0.h
+; NONSTREAMING-NEXT: zip2 z0.h, z1.h, z0.h
+; NONSTREAMING-NEXT: fcvtzs z0.s, p0/m, z0.s
+; NONSTREAMING-NEXT: fcvtzs z2.s, p0/m, z2.s
+; NONSTREAMING-NEXT: uzp1 z0.h, z2.h, z0.h
+; NONSTREAMING-NEXT: ret
+;
+; STREAMING-LABEL: fptosi_nxv8bf16_to_nxv8i16:
+; STREAMING: // %bb.0:
+; STREAMING-NEXT: mov z1.h, #0 // =0x0
+; STREAMING-NEXT: ptrue p0.s
+; STREAMING-NEXT: zip1 z2.h, z1.h, z0.h
+; STREAMING-NEXT: zip2 z0.h, z1.h, z0.h
+; STREAMING-NEXT: fcvtzs z0.s, p0/m, z0.s
+; STREAMING-NEXT: fcvtzs z2.s, p0/m, z2.s
+; STREAMING-NEXT: uzp1 z0.h, z2.h, z0.h
+; STREAMING-NEXT: ret
+;
+; SVE-B16B16_NONSTREAMING-LABEL: fptosi_nxv8bf16_to_nxv8i16:
+; SVE-B16B16_NONSTREAMING: // %bb.0:
+; SVE-B16B16_NONSTREAMING-NEXT: movi v1.2d, #0000000000000000
+; SVE-B16B16_NONSTREAMING-NEXT: ptrue p0.s
+; SVE-B16B16_NONSTREAMING-NEXT: zip1 z2.h, z1.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT: zip2 z0.h, z1.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT: fcvtzs z0.s, p0/m, z0.s
+; SVE-B16B16_NONSTREAMING-NEXT: fcvtzs z2.s, p0/m, z2.s
+; SVE-B16B16_NONSTREAMING-NEXT: uzp1 z0.h, z2.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT: ret
+;
+; SVE-B16B16_STREAMING-LABEL: fptosi_nxv8bf16_to_nxv8i16:
+; SVE-B16B16_STREAMING: // %bb.0:
+; SVE-B16B16_STREAMING-NEXT: mov z1.h, #0 // =0x0
+; SVE-B16B16_STREAMING-NEXT: ptrue p0.s
+; SVE-B16B16_STREAMING-NEXT: zip1 z2.h, z1.h, z0.h
+; SVE-B16B16_STREAMING-NEXT: zip2 z0.h, z1.h, z0.h
+; SVE-B16B16_STREAMING-NEXT: fcvtzs z0.s, p0/m, z0.s
+; SVE-B16B16_STREAMING-NEXT: fcvtzs z2.s, p0/m, z2.s
+; SVE-B16B16_STREAMING-NEXT: uzp1 z0.h, z2.h, z0.h
+; SVE-B16B16_STREAMING-NEXT: ret
+ %r = call <vscale x 8 x i16> @llvm.experimental.constrained.fptosi(<vscale x 8 x bfloat> %a, metadata !"fpexcept.strict")
+ ret <vscale x 8 x i16> %r
+}
+
+;
+; constrained.fptoui
+;
+
+define <vscale x 2 x i64> @fptoui_nxv2bf16_to_nxv2i64(<vscale x 2 x bfloat> %a) {
+; CHECK-LABEL: fptoui_nxv2bf16_to_nxv2i64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: lsl z0.s, z0.s, #16
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: fcvtzu z0.d, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 2 x i64> @llvm.experimental.constrained.fptoui(<vscale x 2 x bfloat> %a, metadata !"fpexcept.strict")
+ ret <vscale x 2 x i64> %r
+}
+
+define <vscale x 4 x i32> @fptoui_nxv4bf16_to_nxv4i32(<vscale x 4 x bfloat> %a) {
+; CHECK-LABEL: fptoui_nxv4bf16_to_nxv4i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: lsl z0.s, z0.s, #16
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: fcvtzu z0.s, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 4 x i32> @llvm.experimental.constrained.fptoui(<vscale x 4 x bfloat> %a, metadata !"fpexcept.strict")
+ ret <vscale x 4 x i32> %r
+}
+
+define <vscale x 8 x i16> @fptoui_nxv8bf16_to_nxv8i16(<vscale x 8 x bfloat> %a) {
+; NONSTREAMING-LABEL: fptoui_nxv8bf16_to_nxv8i16:
+; NONSTREAMING: // %bb.0:
+; NONSTREAMING-NEXT: movi v1.2d, #0000000000000000
+; NONSTREAMING-NEXT: ptrue p0.s
+; NONSTREAMING-NEXT: zip1 z2.h, z1.h, z0.h
+; NONSTREAMING-NEXT: zip2 z0.h, z1.h, z0.h
+; NONSTREAMING-NEXT: fcvtzs z0.s, p0/m, z0.s
+; NONSTREAMING-NEXT: fcvtzs z2.s, p0/m, z2.s
+; NONSTREAMING-NEXT: uzp1 z0.h, z2.h, z0.h
+; NONSTREAMING-NEXT: ret
+;
+; STREAMING-LABEL: fptoui_nxv8bf16_to_nxv8i16:
+; STREAMING: // %bb.0:
+; STREAMING-NEXT: mov z1.h, #0 // =0x0
+; STREAMING-NEXT: ptrue p0.s
+; STREAMING-NEXT: zip1 z2.h, z1.h, z0.h
+; STREAMING-NEXT: zip2 z0.h, z1.h, z0.h
+; STREAMING-NEXT: fcvtzs z0.s, p0/m, z0.s
+; STREAMING-NEXT: fcvtzs z2.s, p0/m, z2.s
+; STREAMING-NEXT: uzp1 z0.h, z2.h, z0.h
+; STREAMING-NEXT: ret
+;
+; SVE-B16B16_NONSTREAMING-LABEL: fptoui_nxv8bf16_to_nxv8i16:
+; SVE-B16B16_NONSTREAMING: // %bb.0:
+; SVE-B16B16_NONSTREAMING-NEXT: movi v1.2d, #0000000000000000
+; SVE-B16B16_NONSTREAMING-NEXT: ptrue p0.s
+; SVE-B16B16_NONSTREAMING-NEXT: zip1 z2.h, z1.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT: zip2 z0.h, z1.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT: fcvtzs z0.s, p0/m, z0.s
+; SVE-B16B16_NONSTREAMING-NEXT: fcvtzs z2.s, p0/m, z2.s
+; SVE-B16B16_NONSTREAMING-NEXT: uzp1 z0.h, z2.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT: ret
+;
+; SVE-B16B16_STREAMING-LABEL: fptoui_nxv8bf16_to_nxv8i16:
+; SVE-B16B16_STREAMING: // %bb.0:
+; SVE-B16B16_STREAMING-NEXT: mov z1.h, #0 // =0x0
+; SVE-B16B16_STREAMING-NEXT: ptrue p0.s
+; SVE-B16B16_STREAMING-NEXT: zip1 z2.h, z1.h, z0.h
+; SVE-B16B16_STREAMING-NEXT: zip2 z0.h, z1.h, z0.h
+; SVE-B16B16_STREAMING-NEXT: fcvtzs z0.s, p0/m, z0.s
+; SVE-B16B16_STREAMING-NEXT: fcvtzs z2.s, p0/m, z2.s
+; SVE-B16B16_STREAMING-NEXT: uzp1 z0.h, z2.h, z0.h
+; SVE-B16B16_STREAMING-NEXT: ret
+ %r = call <vscale x 8 x i16> @llvm.experimental.constrained.fptoui(<vscale x 8 x bfloat> %a, metadata !"fpexcept.strict")
+ ret <vscale x 8 x i16> %r
+}
+
+;
+; constrained.fptrunc
+;
+
+define <vscale x 2 x bfloat> @fptrunc_nv2f32_to_nxv2bf16(<vscale x 2 x float> %a) {
+; CHECK-LABEL: fptrunc_nv2f32_to_nxv2bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 2 x bfloat> @llvm.experimental.constrained.fptrunc(<vscale x 2 x float> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+ ret <vscale x 2 x bfloat> %r
+}
+
+define <vscale x 4 x bfloat> @fptrunc_nv4f32_to_nxv4bf16(<vscale x 4 x float> %a) {
+; CHECK-LABEL: fptrunc_nv4f32_to_nxv4bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 4 x bfloat> @llvm.experimental.constrained.fptrunc(<vscale x 4 x float> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+ ret <vscale x 4 x bfloat> %r
+}
+
+; TODO: Is there are viable lowering when FCVTX is not available?
+define <vscale x 2 x bfloat> @fptrunc_nv2f64_to_nxv2bf16(<vscale x 2 x double> %a) "target-features" = "+sve2" {
+; CHECK-LABEL: fptrunc_nv2f64_to_nxv2bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: fcvtx z0.s, p0/m, z0.d
+; CHECK-NEXT: bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 2 x bfloat> @llvm.experimental.constrained.fptrunc(<vscale x 2 x double> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+ ret <vscale x 2 x bfloat> %r
+}
+
+;
+; constrained.frem (TODO)
+;
+
+;
+; constrained.fsub (TODO)
+;
+
+;
+; constrained.ldexp (TODO)
+;
+
+;
+; constrained.llrint (TODO)
+;
+
+;
+; constrained.llround (TODO)
+;
+
+;
+; constrained.lrint (TODO)
+;
+
+;
+; constrained.lround (TODO)
+;
+
+;
+; constrained.maximum (TODO)
+;
+
+;
+; constrained.maxnum (TODO)
+;
+
+;
+; constrained.minimum (TODO)
+;
+
+;
+; constrained.minnum (TODO)
+;
+
+;
+; constrained.nearbyint (TODO)
+;
+
+;
+; constrained.rint (TODO)
+;
+
+;
+; constrained.round (TODO)
+;
+
+;
+; constrained.roundeven (TODO)
+;
+
+;
+; constrained.sqrt (TODO)
+;
+
+;
+; constrained_sitofp
+;
+
+define <vscale x 8 x bfloat> @sitofp_nxv8i16_to_nxv8bf16(<vscale x 8 x i16> %a) {
+; CHECK-LABEL: sitofp_nxv8i16_to_nxv8bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: sunpklo z1.s, z0.h
+; CHECK-NEXT: sunpkhi z0.s, z0.h
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: scvtf z1.s, p0/m, z1.s
+; CHECK-NEXT: scvtf z0.s, p0/m, z0.s
+; CHECK-NEXT: bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT: bfcvt z1.h, p0/m, z1.s
+; CHECK-NEXT: uzp1 z0.h, z1.h, z0.h
+; CHECK-NEXT: ret
+ %r = call <vscale x 8 x bfloat> @llvm.experimental.constrained.sitofp(<vscale x 8 x i16> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+ ret <vscale x 8 x bfloat> %r
+}
+
+define <vscale x 4 x bfloat> @sitofp_nxv4i32_to_nxv4bf16(<vscale x 4 x i32> %a) {
+; CHECK-LABEL: sitofp_nxv4i32_to_nxv4bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: scvtf z0.s, p0/m, z0.s
+; CHECK-NEXT: bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 4 x bfloat> @llvm.experimental.constrained.sitofp(<vscale x 4 x i32> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+ ret <vscale x 4 x bfloat> %r
+}
+
+define <vscale x 2 x bfloat> @sitofp_nxv2i64_to_nxv2bf16(<vscale x 2 x i64> %a) {
+; CHECK-LABEL: sitofp_nxv2i64_to_nxv2bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: scvtf z0.s, p0/m, z0.d
+; CHECK-NEXT: bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 2 x bfloat> @llvm.experimental.constrained.sitofp(<vscale x 2 x i64> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+ ret <vscale x 2 x bfloat> %r
+}
+
+;
+; constrained.trunc (TODO)
+;
+
+;
+; constrained_uitofp
+;
+
+define <vscale x 8 x bfloat> @uitofp_nxv8i16_to_nxv8bf16(<vscale x 8 x i16> %a) {
+; CHECK-LABEL: uitofp_nxv8i16_to_nxv8bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: uunpklo z1.s, z0.h
+; CHECK-NEXT: uunpkhi z0.s, z0.h
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: ucvtf z1.s, p0/m, z1.s
+; CHECK-NEXT: ucvtf z0.s, p0/m, z0.s
+; CHECK-NEXT: bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT: bfcvt z1.h, p0/m, z1.s
+; CHECK-NEXT: uzp1 z0.h, z1.h, z0.h
+; CHECK-NEXT: ret
+ %r = call <vscale x 8 x bfloat> @llvm.experimental.constrained.uitofp(<vscale x 8 x i16> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+ ret <vscale x 8 x bfloat> %r
+}
+
+define <vscale x 4 x bfloat> @uitofp_nxv4i32_to_nxv4bf16(<vscale x 4 x i32> %a) {
+; CHECK-LABEL: uitofp_nxv4i32_to_nxv4bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: ucvtf z0.s, p0/m, z0.s
+; CHECK-NEXT: bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 4 x bfloat> @llvm.experimental.constrained.uitofp(<vscale x 4 x i32> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+ ret <vscale x 4 x bfloat> %r
+}
+
+define <vscale x 2 x bfloat> @uitofp_nxv2i64_to_nxv2bf16(<vscale x 2 x i64> %a) {
+; CHECK-LABEL: uitofp_nxv2i64_to_nxv2bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: ucvtf z0.s, p0/m, z0.d
+; CHECK-NEXT: bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT: ret
+ %r = call <vscale x 2 x bfloat> @llvm.experimental.constrained.uitofp(<vscale x 2 x i64> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+ ret <vscale x 2 x bfloat> %r
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; BF16: {{.*}}
+; SVE-B16B16: {{.*}}
More information about the llvm-commits
mailing list