[llvm] [LLVM][CodeGen][SVE] Add lowering for bfloat strict-fp cast operators. (PR #223709)

Paul Walker via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 15 07:05:08 PDT 2026


https://github.com/paulwalker-arm created https://github.com/llvm/llvm-project/pull/223709

The majority of the changes are just a case of ensuring the chain is routed correctly and the matching STRICT passthrough node is used.

NOTE: At present full strict-fp support has a minimum requirement of +sve2+bf16, otherwise we lack the necessary cast instructions. Of these +bf16 is fundamental whereas +sve2 is only required for double->bfloat.

NOTE: The test layout is due to follow-on work where I figured it better to have a more complete structure now so the later PRs have less churn.

>From f06afe250ba363eccd4c94efaf41ea9fcff099a4 Mon Sep 17 00:00:00 2001
From: Paul Walker <paul.walker at arm.com>
Date: Tue, 8 Sep 2026 14:12:39 +0100
Subject: [PATCH] [LLVM][CodeGen][SVE] Add lowering for bfloat strict-fp cast
 operators.

NOTE: At present full strict-fp support has a minimum requirement of
+sve2+bf16, otherwise we lack the necessary cast instructions. Of these
+bf16 is fundamental whereas +sve2 is only required for double->bfloat.

NOTE: The test layout is due to follow-on work where I figured it better
to have a more complete structure now so the later PRs have less churn.
---
 .../SelectionDAG/LegalizeVectorOps.cpp        |   6 +
 .../Target/AArch64/AArch64ISelLowering.cpp    |  78 ++--
 .../AArch64/sve-bf-constrained-intrinsics.ll  | 409 ++++++++++++++++++
 3 files changed, 462 insertions(+), 31 deletions(-)
 create mode 100644 llvm/test/CodeGen/AArch64/sve-bf-constrained-intrinsics.ll

diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
index 5e3f252fdd3d4..b41f010d4a2aa 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
@@ -2252,6 +2252,12 @@ bool VectorLegalizer::tryExpandVecMathCall(
 void VectorLegalizer::UnrollStrictFPOp(SDNode *Node,
                                        SmallVectorImpl<SDValue> &Results) {
   EVT VT = Node->getValueType(0);
+
+  // Cannot unroll a scalable vector. Delay error reporting until the final
+  // operation legalisation phase to maximise the chances of removing the node.
+  if (VT.isScalableVector())
+    return;
+
   EVT EltVT = VT.getVectorElementType();
   unsigned NumElems = VT.getVectorNumElements();
   unsigned NumOpers = Node->getNumOperands();
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 2a50c0c474ab9..a4c834aa3bd7c 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -1972,8 +1972,8 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
       setOperationAction(ISD::FABS, VT, Custom);
       setOperationAction(ISD::FCOPYSIGN, VT, Custom);
       setOperationAction(ISD::FNEG, VT, Custom);
-      setOperationAction(ISD::FP_EXTEND, VT, Custom);
-      setOperationAction(ISD::FP_ROUND, VT, Custom);
+      setOperationAction({ISD::FP_EXTEND, ISD::STRICT_FP_EXTEND}, VT, Custom);
+      setOperationAction({ISD::FP_ROUND, ISD::STRICT_FP_ROUND}, VT, Custom);
       setOperationAction(ISD::MLOAD, VT, Custom);
       setOperationAction(ISD::INSERT_SUBVECTOR, VT, Custom);
       setOperationAction(ISD::SELECT, VT, Custom);
@@ -4962,18 +4962,23 @@ SDValue AArch64TargetLowering::LowerFP_EXTEND(SDValue Op,
   bool IsStrict = Op->isStrictFPOpcode();
 
   if (VT.isScalableVector()) {
-    SDValue SrcVal = Op.getOperand(0);
+    SDValue SrcVal = Op.getOperand(IsStrict ? 1 : 0);
 
     if (VT == MVT::nxv2f64 && SrcVal.getValueType() == MVT::nxv2bf16) {
-      // TODO: Missing support for bfloat strict-fp operations.
-      if (IsStrict)
-        return SDValue();
-
-      // Break conversion in two with the first part converting to f32 and the
-      // second using native f32->VT instructions.
       SDLoc DL(Op);
-      return DAG.getNode(ISD::FP_EXTEND, DL, VT,
-                         DAG.getNode(ISD::FP_EXTEND, DL, MVT::nxv2f32, SrcVal));
+
+      // Split cast into two phases, using float as the intermediary.
+      SDVTList CvtF32VTs = IsStrict ? DAG.getVTList(MVT::nxv2f32, MVT::Other)
+                                    : DAG.getVTList(MVT::nxv2f32);
+      SmallVector<SDValue> CvtF32Ops(Op->ops());
+      SDValue CvtF32 = DAG.getNode(Op.getOpcode(), DL, CvtF32VTs, CvtF32Ops);
+
+      // Extend from float to double.
+      SmallVector<SDValue, 2> CvtF64Ops;
+      if (IsStrict)
+        CvtF64Ops.push_back(CvtF32.getValue(1)); // Chain
+      CvtF64Ops.push_back(CvtF32);
+      return DAG.getNode(Op.getOpcode(), DL, Op->getVTList(), CvtF64Ops);
     }
 
     return LowerToPredicatedOp(Op, DAG,
@@ -5023,15 +5028,11 @@ SDValue AArch64TargetLowering::LowerFP_ROUND(SDValue Op,
     if (SrcVT == MVT::nxv8f32)
       return Op;
 
+    unsigned MergePasthruOpc = IsStrict
+                                   ? AArch64ISD::STRICT_FP_ROUND_MERGE_PASSTHRU
+                                   : AArch64ISD::FP_ROUND_MERGE_PASSTHRU;
     if (VT.getScalarType() != MVT::bf16)
-      return LowerToPredicatedOp(
-          Op, DAG,
-          IsStrict ? AArch64ISD::STRICT_FP_ROUND_MERGE_PASSTHRU
-                   : AArch64ISD::FP_ROUND_MERGE_PASSTHRU);
-
-    // TODO: Missing support for bfloat strict-fp operations.
-    if (IsStrict)
-      return SDValue();
+      return LowerToPredicatedOp(Op, DAG, MergePasthruOpc);
 
     SDLoc DL(Op);
     constexpr EVT I32 = MVT::nxv4i32;
@@ -5042,8 +5043,7 @@ SDValue AArch64TargetLowering::LowerFP_ROUND(SDValue Op,
 
     if (SrcVT == MVT::nxv2f32 || SrcVT == MVT::nxv4f32) {
       if (Subtarget->hasBF16())
-        return LowerToPredicatedOp(Op, DAG,
-                                   AArch64ISD::FP_ROUND_MERGE_PASSTHRU);
+        return LowerToPredicatedOp(Op, DAG, MergePasthruOpc);
 
       Narrow = getSVESafeBitCast(I32, SrcVal, DAG);
 
@@ -5052,20 +5052,36 @@ SDValue AArch64TargetLowering::LowerFP_ROUND(SDValue Op,
         NaN = DAG.getNode(ISD::OR, DL, I32, Narrow, ImmV(0x400000));
     } else if (SrcVT == MVT::nxv2f64 &&
                (Subtarget->hasSVE2() || Subtarget->isStreamingSVEAvailable())) {
-      // Round to float without introducing rounding errors and try again.
-      SDValue Pg = getPredicateForVector(DAG, DL, MVT::nxv2f32);
-      Narrow = DAG.getNode(AArch64ISD::FCVTX_MERGE_PASSTHRU, DL, MVT::nxv2f32,
-                           Pg, SrcVal, DAG.getPOISON(MVT::nxv2f32));
-
-      SmallVector<SDValue, 3> NewOps;
+      // Split cast into two phases, using float as the intermediary.
+      SDVTList CvtF32VTs = IsStrict ? DAG.getVTList(MVT::nxv2f32, MVT::Other)
+                                    : DAG.getVTList(MVT::nxv2f32);
+
+      // Round to float without introducing rounding errors.
+      unsigned CvtF32Opc = IsStrict ? AArch64ISD::STRICT_FCVTX_MERGE_PASSTHRU
+                                    : AArch64ISD::FCVTX_MERGE_PASSTHRU;
+      SmallVector<SDValue, 3> CvtF32Ops;
       if (IsStrict)
-        NewOps.push_back(Op.getOperand(0));
-      NewOps.push_back(Narrow);
-      NewOps.push_back(Op.getOperand(IsStrict ? 2 : 1));
-      return DAG.getNode(Op.getOpcode(), DL, VT, NewOps, Op->getFlags());
+        CvtF32Ops.push_back(Op.getOperand(0)); // Chain
+      CvtF32Ops.push_back(getPredicateForVector(DAG, DL, MVT::nxv2f32));
+      CvtF32Ops.push_back(SrcVal);
+      CvtF32Ops.push_back(DAG.getPOISON(MVT::nxv2f32));
+      SDValue CvtF32 = DAG.getNode(CvtF32Opc, DL, CvtF32VTs, CvtF32Ops);
+
+      // Round from float to bfloat.
+      SmallVector<SDValue, 3> CvtBfOps;
+      if (IsStrict)
+        CvtBfOps.push_back(CvtF32.getValue(1)); // Chain
+      CvtBfOps.push_back(CvtF32);
+      CvtBfOps.push_back(Op.getOperand(IsStrict ? 2 : 1)); // Trunc
+      return DAG.getNode(Op.getOpcode(), DL, Op->getVTList(), CvtBfOps,
+                         Op->getFlags());
     } else
       return SDValue();
 
+    // TODO: Missing support for bfloat strict-fp operations.
+    if (IsStrict)
+      return SDValue();
+
     if (!Trunc) {
       SDValue Lsb = DAG.getNode(ISD::SRL, DL, I32, Narrow, ImmV(16));
       Lsb = DAG.getNode(ISD::AND, DL, I32, Lsb, ImmV(1));
diff --git a/llvm/test/CodeGen/AArch64/sve-bf-constrained-intrinsics.ll b/llvm/test/CodeGen/AArch64/sve-bf-constrained-intrinsics.ll
new file mode 100644
index 0000000000000..d894a90da599b
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-bf-constrained-intrinsics.ll
@@ -0,0 +1,409 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mattr=+sve,+bf16 < %s -o - | FileCheck %s -check-prefixes=CHECK,BF16,NONSTREAMING
+; RUN: llc -mattr=+sme --force-streaming < %s -o - | FileCheck %s -check-prefixes=CHECK,BF16,STREAMING
+
+; RUN: llc -mattr=+sve,+bf16,+sve-b16b16 < %s -o - | FileCheck %s --check-prefixes=CHECK,SVE-B16B16,SVE-B16B16_NONSTREAMING
+; RUN: llc -mattr=+sme2,+sve-b16b16 --force-streaming < %s -o - | FileCheck %s --check-prefixes=CHECK,SVE-B16B16,SVE-B16B16_STREAMING
+
+; NOTE: Full strict-fp support requires SVE variants of FCVTX and BFCVT.
+
+target triple = "aarch64-unknown-linux-gnu"
+
+;
+; constrained.ceil (TODO)
+;
+
+;
+; constrained.fadd (TODO)
+;
+
+;
+; constrained.fcmp (TODO)
+;
+
+;
+; constrained.fcmps (TODO)
+;
+
+;
+; constrained.fdiv (TODO)
+;
+
+;
+; constrained.floor (TODO)
+;
+
+;
+; constrained.fma (TODO)
+;
+
+;
+; constrained.fmul (TODO)
+;
+
+;
+; constrained.fmuladd (TODO)
+;
+
+;
+; constrained.fpext
+;
+
+define <vscale x 2 x float> @fpext_nxv2bf16_to_nxv2f32(<vscale x 2 x bfloat> %a) {
+; CHECK-LABEL: fpext_nxv2bf16_to_nxv2f32:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    lsl z0.s, z0.s, #16
+; CHECK-NEXT:    ret
+  %r = call <vscale x 2 x float> @llvm.experimental.constrained.fpext(<vscale x 2 x bfloat> %a, metadata !"fpexcept.strict")
+  ret <vscale x 2 x float> %r
+}
+
+define <vscale x 2 x double> @fpext_nxv2bf16_to_nxv2f64(<vscale x 2 x bfloat> %a) {
+; CHECK-LABEL: fpext_nxv2bf16_to_nxv2f64:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    lsl z0.s, z0.s, #16
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    fcvt z0.d, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 2 x double> @llvm.experimental.constrained.fpext(<vscale x 2 x bfloat> %a, metadata !"fpexcept.strict")
+  ret <vscale x 2 x double> %r
+}
+
+;
+; constrained.fptosi
+;
+
+define <vscale x 2 x i64> @fptosi_nxv2bf16_to_nxv2i64(<vscale x 2 x bfloat> %a) {
+; CHECK-LABEL: fptosi_nxv2bf16_to_nxv2i64:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    lsl z0.s, z0.s, #16
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    fcvtzs z0.d, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 2 x i64> @llvm.experimental.constrained.fptosi(<vscale x 2 x bfloat> %a, metadata !"fpexcept.strict")
+  ret <vscale x 2 x i64> %r
+}
+
+define <vscale x 4 x i32> @fptosi_nxv4bf16_to_nxv4i32(<vscale x 4 x bfloat> %a) {
+; CHECK-LABEL: fptosi_nxv4bf16_to_nxv4i32:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    lsl z0.s, z0.s, #16
+; CHECK-NEXT:    ptrue p0.s
+; CHECK-NEXT:    fcvtzs z0.s, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 4 x i32> @llvm.experimental.constrained.fptosi(<vscale x 4 x bfloat> %a, metadata !"fpexcept.strict")
+  ret <vscale x 4 x i32> %r
+}
+
+define <vscale x 8 x i16> @fptosi_nxv8bf16_to_nxv8i16(<vscale x 8 x bfloat> %a) {
+; NONSTREAMING-LABEL: fptosi_nxv8bf16_to_nxv8i16:
+; NONSTREAMING:       // %bb.0:
+; NONSTREAMING-NEXT:    movi v1.2d, #0000000000000000
+; NONSTREAMING-NEXT:    ptrue p0.s
+; NONSTREAMING-NEXT:    zip1 z2.h, z1.h, z0.h
+; NONSTREAMING-NEXT:    zip2 z0.h, z1.h, z0.h
+; NONSTREAMING-NEXT:    fcvtzs z0.s, p0/m, z0.s
+; NONSTREAMING-NEXT:    fcvtzs z2.s, p0/m, z2.s
+; NONSTREAMING-NEXT:    uzp1 z0.h, z2.h, z0.h
+; NONSTREAMING-NEXT:    ret
+;
+; STREAMING-LABEL: fptosi_nxv8bf16_to_nxv8i16:
+; STREAMING:       // %bb.0:
+; STREAMING-NEXT:    mov z1.h, #0 // =0x0
+; STREAMING-NEXT:    ptrue p0.s
+; STREAMING-NEXT:    zip1 z2.h, z1.h, z0.h
+; STREAMING-NEXT:    zip2 z0.h, z1.h, z0.h
+; STREAMING-NEXT:    fcvtzs z0.s, p0/m, z0.s
+; STREAMING-NEXT:    fcvtzs z2.s, p0/m, z2.s
+; STREAMING-NEXT:    uzp1 z0.h, z2.h, z0.h
+; STREAMING-NEXT:    ret
+;
+; SVE-B16B16_NONSTREAMING-LABEL: fptosi_nxv8bf16_to_nxv8i16:
+; SVE-B16B16_NONSTREAMING:       // %bb.0:
+; SVE-B16B16_NONSTREAMING-NEXT:    movi v1.2d, #0000000000000000
+; SVE-B16B16_NONSTREAMING-NEXT:    ptrue p0.s
+; SVE-B16B16_NONSTREAMING-NEXT:    zip1 z2.h, z1.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT:    zip2 z0.h, z1.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT:    fcvtzs z0.s, p0/m, z0.s
+; SVE-B16B16_NONSTREAMING-NEXT:    fcvtzs z2.s, p0/m, z2.s
+; SVE-B16B16_NONSTREAMING-NEXT:    uzp1 z0.h, z2.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT:    ret
+;
+; SVE-B16B16_STREAMING-LABEL: fptosi_nxv8bf16_to_nxv8i16:
+; SVE-B16B16_STREAMING:       // %bb.0:
+; SVE-B16B16_STREAMING-NEXT:    mov z1.h, #0 // =0x0
+; SVE-B16B16_STREAMING-NEXT:    ptrue p0.s
+; SVE-B16B16_STREAMING-NEXT:    zip1 z2.h, z1.h, z0.h
+; SVE-B16B16_STREAMING-NEXT:    zip2 z0.h, z1.h, z0.h
+; SVE-B16B16_STREAMING-NEXT:    fcvtzs z0.s, p0/m, z0.s
+; SVE-B16B16_STREAMING-NEXT:    fcvtzs z2.s, p0/m, z2.s
+; SVE-B16B16_STREAMING-NEXT:    uzp1 z0.h, z2.h, z0.h
+; SVE-B16B16_STREAMING-NEXT:    ret
+  %r = call <vscale x 8 x i16> @llvm.experimental.constrained.fptosi(<vscale x 8 x bfloat> %a, metadata !"fpexcept.strict")
+  ret <vscale x 8 x i16> %r
+}
+
+;
+; constrained.fptoui
+;
+
+define <vscale x 2 x i64> @fptoui_nxv2bf16_to_nxv2i64(<vscale x 2 x bfloat> %a) {
+; CHECK-LABEL: fptoui_nxv2bf16_to_nxv2i64:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    lsl z0.s, z0.s, #16
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    fcvtzu z0.d, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 2 x i64> @llvm.experimental.constrained.fptoui(<vscale x 2 x bfloat> %a, metadata !"fpexcept.strict")
+  ret <vscale x 2 x i64> %r
+}
+
+define <vscale x 4 x i32> @fptoui_nxv4bf16_to_nxv4i32(<vscale x 4 x bfloat> %a) {
+; CHECK-LABEL: fptoui_nxv4bf16_to_nxv4i32:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    lsl z0.s, z0.s, #16
+; CHECK-NEXT:    ptrue p0.s
+; CHECK-NEXT:    fcvtzu z0.s, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 4 x i32> @llvm.experimental.constrained.fptoui(<vscale x 4 x bfloat> %a, metadata !"fpexcept.strict")
+  ret <vscale x 4 x i32> %r
+}
+
+define <vscale x 8 x i16> @fptoui_nxv8bf16_to_nxv8i16(<vscale x 8 x bfloat> %a) {
+; NONSTREAMING-LABEL: fptoui_nxv8bf16_to_nxv8i16:
+; NONSTREAMING:       // %bb.0:
+; NONSTREAMING-NEXT:    movi v1.2d, #0000000000000000
+; NONSTREAMING-NEXT:    ptrue p0.s
+; NONSTREAMING-NEXT:    zip1 z2.h, z1.h, z0.h
+; NONSTREAMING-NEXT:    zip2 z0.h, z1.h, z0.h
+; NONSTREAMING-NEXT:    fcvtzs z0.s, p0/m, z0.s
+; NONSTREAMING-NEXT:    fcvtzs z2.s, p0/m, z2.s
+; NONSTREAMING-NEXT:    uzp1 z0.h, z2.h, z0.h
+; NONSTREAMING-NEXT:    ret
+;
+; STREAMING-LABEL: fptoui_nxv8bf16_to_nxv8i16:
+; STREAMING:       // %bb.0:
+; STREAMING-NEXT:    mov z1.h, #0 // =0x0
+; STREAMING-NEXT:    ptrue p0.s
+; STREAMING-NEXT:    zip1 z2.h, z1.h, z0.h
+; STREAMING-NEXT:    zip2 z0.h, z1.h, z0.h
+; STREAMING-NEXT:    fcvtzs z0.s, p0/m, z0.s
+; STREAMING-NEXT:    fcvtzs z2.s, p0/m, z2.s
+; STREAMING-NEXT:    uzp1 z0.h, z2.h, z0.h
+; STREAMING-NEXT:    ret
+;
+; SVE-B16B16_NONSTREAMING-LABEL: fptoui_nxv8bf16_to_nxv8i16:
+; SVE-B16B16_NONSTREAMING:       // %bb.0:
+; SVE-B16B16_NONSTREAMING-NEXT:    movi v1.2d, #0000000000000000
+; SVE-B16B16_NONSTREAMING-NEXT:    ptrue p0.s
+; SVE-B16B16_NONSTREAMING-NEXT:    zip1 z2.h, z1.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT:    zip2 z0.h, z1.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT:    fcvtzs z0.s, p0/m, z0.s
+; SVE-B16B16_NONSTREAMING-NEXT:    fcvtzs z2.s, p0/m, z2.s
+; SVE-B16B16_NONSTREAMING-NEXT:    uzp1 z0.h, z2.h, z0.h
+; SVE-B16B16_NONSTREAMING-NEXT:    ret
+;
+; SVE-B16B16_STREAMING-LABEL: fptoui_nxv8bf16_to_nxv8i16:
+; SVE-B16B16_STREAMING:       // %bb.0:
+; SVE-B16B16_STREAMING-NEXT:    mov z1.h, #0 // =0x0
+; SVE-B16B16_STREAMING-NEXT:    ptrue p0.s
+; SVE-B16B16_STREAMING-NEXT:    zip1 z2.h, z1.h, z0.h
+; SVE-B16B16_STREAMING-NEXT:    zip2 z0.h, z1.h, z0.h
+; SVE-B16B16_STREAMING-NEXT:    fcvtzs z0.s, p0/m, z0.s
+; SVE-B16B16_STREAMING-NEXT:    fcvtzs z2.s, p0/m, z2.s
+; SVE-B16B16_STREAMING-NEXT:    uzp1 z0.h, z2.h, z0.h
+; SVE-B16B16_STREAMING-NEXT:    ret
+  %r = call <vscale x 8 x i16> @llvm.experimental.constrained.fptoui(<vscale x 8 x bfloat> %a, metadata !"fpexcept.strict")
+  ret <vscale x 8 x i16> %r
+}
+
+;
+; constrained.fptrunc
+;
+
+define <vscale x 2 x bfloat> @fptrunc_nv2f32_to_nxv2bf16(<vscale x 2 x float> %a) {
+; CHECK-LABEL: fptrunc_nv2f32_to_nxv2bf16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 2 x bfloat> @llvm.experimental.constrained.fptrunc(<vscale x 2 x float> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+  ret <vscale x 2 x bfloat> %r
+}
+
+define <vscale x 4 x bfloat> @fptrunc_nv4f32_to_nxv4bf16(<vscale x 4 x float> %a) {
+; CHECK-LABEL: fptrunc_nv4f32_to_nxv4bf16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    ptrue p0.s
+; CHECK-NEXT:    bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 4 x bfloat> @llvm.experimental.constrained.fptrunc(<vscale x 4 x float> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+  ret <vscale x 4 x bfloat> %r
+}
+
+; TODO: Is there are viable lowering when FCVTX is not available?
+define <vscale x 2 x bfloat> @fptrunc_nv2f64_to_nxv2bf16(<vscale x 2 x double> %a) "target-features" = "+sve2" {
+; CHECK-LABEL: fptrunc_nv2f64_to_nxv2bf16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    fcvtx z0.s, p0/m, z0.d
+; CHECK-NEXT:    bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 2 x bfloat> @llvm.experimental.constrained.fptrunc(<vscale x 2 x double> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+  ret <vscale x 2 x bfloat> %r
+}
+
+;
+; constrained.frem (TODO)
+;
+
+;
+; constrained.fsub (TODO)
+;
+
+;
+; constrained.ldexp (TODO)
+;
+
+;
+; constrained.llrint (TODO)
+;
+
+;
+; constrained.llround (TODO)
+;
+
+;
+; constrained.lrint (TODO)
+;
+
+;
+; constrained.lround (TODO)
+;
+
+;
+; constrained.maximum (TODO)
+;
+
+;
+; constrained.maxnum (TODO)
+;
+
+;
+; constrained.minimum (TODO)
+;
+
+;
+; constrained.minnum (TODO)
+;
+
+;
+; constrained.nearbyint (TODO)
+;
+
+;
+; constrained.rint (TODO)
+;
+
+;
+; constrained.round (TODO)
+;
+
+;
+; constrained.roundeven (TODO)
+;
+
+;
+; constrained.sqrt (TODO)
+;
+
+;
+; constrained_sitofp
+;
+
+define <vscale x 8 x bfloat> @sitofp_nxv8i16_to_nxv8bf16(<vscale x 8 x i16> %a) {
+; CHECK-LABEL: sitofp_nxv8i16_to_nxv8bf16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    sunpklo z1.s, z0.h
+; CHECK-NEXT:    sunpkhi z0.s, z0.h
+; CHECK-NEXT:    ptrue p0.s
+; CHECK-NEXT:    scvtf z1.s, p0/m, z1.s
+; CHECK-NEXT:    scvtf z0.s, p0/m, z0.s
+; CHECK-NEXT:    bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT:    bfcvt z1.h, p0/m, z1.s
+; CHECK-NEXT:    uzp1 z0.h, z1.h, z0.h
+; CHECK-NEXT:    ret
+  %r = call <vscale x 8 x bfloat> @llvm.experimental.constrained.sitofp(<vscale x 8 x i16> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+  ret <vscale x 8 x bfloat> %r
+}
+
+define <vscale x 4 x bfloat> @sitofp_nxv4i32_to_nxv4bf16(<vscale x 4 x i32> %a) {
+; CHECK-LABEL: sitofp_nxv4i32_to_nxv4bf16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    ptrue p0.s
+; CHECK-NEXT:    scvtf z0.s, p0/m, z0.s
+; CHECK-NEXT:    bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 4 x bfloat> @llvm.experimental.constrained.sitofp(<vscale x 4 x i32> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+  ret <vscale x 4 x bfloat> %r
+}
+
+define <vscale x 2 x bfloat> @sitofp_nxv2i64_to_nxv2bf16(<vscale x 2 x i64> %a) {
+; CHECK-LABEL: sitofp_nxv2i64_to_nxv2bf16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    scvtf z0.s, p0/m, z0.d
+; CHECK-NEXT:    bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 2 x bfloat> @llvm.experimental.constrained.sitofp(<vscale x 2 x i64> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+  ret <vscale x 2 x bfloat> %r
+}
+
+;
+; constrained.trunc (TODO)
+;
+
+;
+; constrained_uitofp
+;
+
+define <vscale x 8 x bfloat> @uitofp_nxv8i16_to_nxv8bf16(<vscale x 8 x i16> %a) {
+; CHECK-LABEL: uitofp_nxv8i16_to_nxv8bf16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    uunpklo z1.s, z0.h
+; CHECK-NEXT:    uunpkhi z0.s, z0.h
+; CHECK-NEXT:    ptrue p0.s
+; CHECK-NEXT:    ucvtf z1.s, p0/m, z1.s
+; CHECK-NEXT:    ucvtf z0.s, p0/m, z0.s
+; CHECK-NEXT:    bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT:    bfcvt z1.h, p0/m, z1.s
+; CHECK-NEXT:    uzp1 z0.h, z1.h, z0.h
+; CHECK-NEXT:    ret
+  %r = call <vscale x 8 x bfloat> @llvm.experimental.constrained.uitofp(<vscale x 8 x i16> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+  ret <vscale x 8 x bfloat> %r
+}
+
+define <vscale x 4 x bfloat> @uitofp_nxv4i32_to_nxv4bf16(<vscale x 4 x i32> %a) {
+; CHECK-LABEL: uitofp_nxv4i32_to_nxv4bf16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    ptrue p0.s
+; CHECK-NEXT:    ucvtf z0.s, p0/m, z0.s
+; CHECK-NEXT:    bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 4 x bfloat> @llvm.experimental.constrained.uitofp(<vscale x 4 x i32> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+  ret <vscale x 4 x bfloat> %r
+}
+
+define <vscale x 2 x bfloat> @uitofp_nxv2i64_to_nxv2bf16(<vscale x 2 x i64> %a) {
+; CHECK-LABEL: uitofp_nxv2i64_to_nxv2bf16:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    ptrue p0.d
+; CHECK-NEXT:    ucvtf z0.s, p0/m, z0.d
+; CHECK-NEXT:    bfcvt z0.h, p0/m, z0.s
+; CHECK-NEXT:    ret
+  %r = call <vscale x 2 x bfloat> @llvm.experimental.constrained.uitofp(<vscale x 2 x i64> %a, metadata !"round.dynamic", metadata !"fpexcept.strict")
+  ret <vscale x 2 x bfloat> %r
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; BF16: {{.*}}
+; SVE-B16B16: {{.*}}



More information about the llvm-commits mailing list