[llvm] [NVPTX] Add intrinsics for ff/f16/bf16 to ue5m3 conversions (PR #218677)

Dharuni R Acharya via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 25 23:25:47 PDT 2026


https://github.com/DharuniRAcharya updated https://github.com/llvm/llvm-project/pull/218677

>From d8df46b3de73da0f27c8f0f4f4115dc19e57a9b7 Mon Sep 17 00:00:00 2001
From: DharuniRAcharya <dharunira at nvidia.com>
Date: Tue, 25 Aug 2026 12:55:40 +0000
Subject: [PATCH 1/2] [NVPTX] Add intrinsics for ff/f16/bf16 to ue5m3
 conversions

This patch adds the following intrinsics for ff/f16/bf16 to ue5m3 conversions introduced in PTX 9.4:

- cvt{.rn,.rz,.rp}{.satfinite}.ue5m3x2.f32
- cvt{.rn,.rz}{.satfinite}.scaled::n1::ue8m0.ue5m3x2.f32

- cvt{.rn,.rz,.rp}{.satfinite}.ue5m3x2{.f16x2,.bf16x2}
- cvt{.rn,.rz}{.satfinite}.scaled::n1::ue8m0.ue5m3x2{.f16x2,.bf16x2}

Signed-off-by: DharuniRAcharya <dharunira at nvidia.com>
---
 llvm/docs/NVPTXUsage.md                    |  35 +-
 llvm/include/llvm/IR/IntrinsicsNVVM.td     |  26 +
 llvm/lib/Target/NVPTX/NVPTXInstrInfo.td    |  32 ++
 llvm/lib/Target/NVPTX/NVPTXIntrinsics.td   |  37 +-
 llvm/test/CodeGen/NVPTX/convert-ue5m3x2.ll | 551 +++++++++++++++++++++
 5 files changed, 668 insertions(+), 13 deletions(-)

diff --git a/llvm/docs/NVPTXUsage.md b/llvm/docs/NVPTXUsage.md
index efad04b638494..48efb50b62e2b 100644
--- a/llvm/docs/NVPTXUsage.md
+++ b/llvm/docs/NVPTXUsage.md
@@ -1024,11 +1024,12 @@ The following table describes the rounding modes used across these intrinsics:
 
 (scale-factor)=
 
-Some conversions involve a scale factor which is provided as a packed 16-bit
-integer containing two scaling factors of type `ue8m0`, one for each input.
-For down conversion, inputs are divided by `scale_factor` and then the
-conversion is performed. For up-conversion, inputs are converted to destination
-type and then multiplied by `scale_factor`.
+Some conversions involve a scale factor of type `ue8m0`. For
+`scale.n2.ue8m0`, the operand is a packed 16-bit integer containing two
+`ue8m0` scale values, one for each input. For `scale.n1.ue8m0`, a single `ue8m0`
+scale factor is applied to both inputs. For down conversion, inputs are divided
+by `scale_factor` and then the conversion is performed. For up-conversion, 
+inputs are converted to destination type and then multiplied by `scale_factor`.
 
 #### `fp8` Conversion Intrinsics
 
@@ -1037,21 +1038,31 @@ type and then multiplied by `scale_factor`.
 ```llvm
 declare i16 @llvm.nvvm.ff.to{.e4m3x2, .e5m2x2}.rn{.relu}(float %a, float %b)
 declare i16 @llvm.nvvm.ff.to.ue8m0x2{.rz, .rp}{.satfinite}(float %a, float %b)
+declare i16 @llvm.nvvm.ff.to.ue5m3x2{.rn, .rz, .rp}{.satfinite}(float %a, float %b)
+declare i16 @llvm.nvvm.ff.to.ue5m3x2{.rn, .rz}{.satfinite}.scale.n1.ue8m0(float %a, float %b, i16 %scale_factor)
 declare i16 @llvm.f16x2.to{.e4m3x2, .e5m2x2}.rn{.relu}(<2 x half> %a)
+declare i16 @llvm.nvvm.f16x2.to.ue5m3x2{.rn, .rz, .rp}{.satfinite}(<2 x half> %a)
+declare i16 @llvm.nvvm.f16x2.to.ue5m3x2{.rn, .rz}{.satfinite}.scale.n1.ue8m0(<2 x half> %a, i16 %scale_factor)
 declare i16 @llvm.bf16x2.to{.e4m3x2, .e5m2x2}.rn{.relu}.satfinite(<2 x bfloat> %a)
 declare i16 @llvm.bf16x2.to.ue8m0x2{.rz, .rp}{.satfinite}(<2 x bfloat> %a)
+declare i16 @llvm.nvvm.bf16x2.to.ue5m3x2{.rn, .rz, .rp}{.satfinite}(<2 x bfloat> %a)
+declare i16 @llvm.nvvm.bf16x2.to.ue5m3x2{.rn, .rz}{.satfinite}.scale.n1.ue8m0(<2 x bfloat> %a, i16 %scale_factor)
 declare <2 x half> @llvm.nvvm{.e4m3x2, .e5m2x2}.to.f16x2.rn{.relu}(i16 %a)
+declare <2 x half> @llvm.nvvm.ue5m3x2.to.f16x2.rn(i16 %a)
 declare <2 x bfloat> @llvm.nvvm{.e4m3x2, .e5m2x2}.to.bf16x2.rn{.relu}{.satfinite}.scale.n2.ue8m0(i16 %a, i16 %scale_factor)
 declare <2 x bfloat> @llvm.nvvm.ue8m0x2.to.bf16x2(i16 %a)
+declare <2 x bfloat> @llvm.nvvm.ue5m3x2.to.bf16x2.rn{.satfinite}(i16 %a)
+declare <2 x bfloat> @llvm.nvvm.ue5m3x2.to.bf16x2.rn{.satfinite}.scale.n2.ue8m0(i16 %a, i16 %scale_factor)
 declare <4 x i8> @llvm.nvvm.f32x4.to{.e4m3x4, .e5m2x4}.rs{.relu}.satfinite(<4 x f32> %a, i32 %rnd_bits)
 ```
 
 ##### Overview:
 
-These intrinsics perform conversions involving the `e4m3` and `e5m2` narrow
-floating-point formats. In case of two inputs, the value converted from input
-`%a` is stored in the upper 8-bits of the result, and the value converted
-from input `%b` is stored in the lower 8-bits of the result.
+These intrinsics perform conversions involving the `e4m3`, `e5m2`, `ue8m0`,
+and `ue5m3` narrow floating-point formats. In case of two inputs, the value
+converted from input `%a` is stored in the upper 8-bits of the result, and
+the value converted from input `%b` is stored in the lower 8-bits of the
+result.
 
 For rounding modes, see {ref}`narrow-fp-rounding-modes`.
 
@@ -1060,9 +1071,9 @@ The `relu` modifier clamps negative results to 0.
 When `satfinite` is specified, if the absolute value of input (ignoring sign)
 is greater than `MAX_NORM` of the specified destination format, then the
 result is sign-preserved `MAX_NORM` of the destination format and a positive
-`MAX_NORM` in `.ue8m0x2` for which the destination sign is not supported.
-Also, if the input value is `NaN`, then the result is `NaN` in the
-specified destination format. The `satfinite` modifier is assumed to be
+`MAX_NORM` in `.ue8m0x2`/`.ue5m3x2` for which the destination sign is not
+supported. Also, if the input value is `NaN`, then the result is `NaN` in
+the specified destination format. The `satfinite` modifier is assumed to be
 present for conversions involving `e4m3` and `e5m2` types as the
 destination.
 
diff --git a/llvm/include/llvm/IR/IntrinsicsNVVM.td b/llvm/include/llvm/IR/IntrinsicsNVVM.td
index 69905899f3007..4cc536f88be7f 100644
--- a/llvm/include/llvm/IR/IntrinsicsNVVM.td
+++ b/llvm/include/llvm/IR/IntrinsicsNVVM.td
@@ -2016,6 +2016,32 @@ let TargetPrefix = "nvvm" in {
         : PureIntrinsic<[llvm_v2bf16_ty], [llvm_i16_ty, llvm_i16_ty]>;
   }
 
+  foreach rnd = ["rn", "rz", "rp"] in {
+    foreach satfinite = ["", "_satfinite"] in {
+      def int_nvvm_ff_to_ue5m3x2_ # rnd # satfinite
+          : PureIntrinsic<[llvm_i16_ty], [llvm_float_ty, llvm_float_ty]>;
+
+      def int_nvvm_f16x2_to_ue5m3x2_ # rnd # satfinite
+          : PureIntrinsic<[llvm_i16_ty], [llvm_v2f16_ty]>;
+
+      def int_nvvm_bf16x2_to_ue5m3x2_ # rnd # satfinite
+          : PureIntrinsic<[llvm_i16_ty], [llvm_v2bf16_ty]>;
+    }
+  }
+
+  foreach rnd = ["rn", "rz"] in {
+    foreach satfinite = ["", "_satfinite"] in {
+      def int_nvvm_ff_to_ue5m3x2_ # rnd # satfinite # _scale_n1_ue8m0
+          : PureIntrinsic<[llvm_i16_ty], [llvm_float_ty, llvm_float_ty, llvm_i16_ty]>;
+
+      def int_nvvm_f16x2_to_ue5m3x2_ # rnd # satfinite # _scale_n1_ue8m0
+          : PureIntrinsic<[llvm_i16_ty], [llvm_v2f16_ty, llvm_i16_ty]>;
+
+      def int_nvvm_bf16x2_to_ue5m3x2_ # rnd # satfinite # _scale_n1_ue8m0
+          : PureIntrinsic<[llvm_i16_ty], [llvm_v2bf16_ty, llvm_i16_ty]>;
+    }
+  }
+
   //
   // Bar.Sync
   //
diff --git a/llvm/lib/Target/NVPTX/NVPTXInstrInfo.td b/llvm/lib/Target/NVPTX/NVPTXInstrInfo.td
index 4c3c7792f277b..34e9a698d9d47 100644
--- a/llvm/lib/Target/NVPTX/NVPTXInstrInfo.td
+++ b/llvm/lib/Target/NVPTX/NVPTXInstrInfo.td
@@ -57,6 +57,8 @@ def CvtRS_RELU     : PatLeaf<(i32 0x4A)>;
 
 def CvtNONE_SATFINITE  : PatLeaf<(i32 0x80)>;
 def CvtRN_SATFINITE   : PatLeaf<(i32 0x85)>;
+def CvtRZ_SATFINITE   : PatLeaf<(i32 0x86)>;
+def CvtRP_SATFINITE   : PatLeaf<(i32 0x88)>;
 def CvtRN_RELU_SATFINITE : PatLeaf<(i32 0xC5)>;
 
 def CvtMode : Operand<i32> {
@@ -936,6 +938,36 @@ let Predicates = [hasS2F6X2ConversionSupport] in {
             "cvt${mode:base}${mode:satfinite}.scaled::n2::ue8m0.bf16x2.ue5m3x2">,
     Requires<[hasUE5M3TypeSupport]>;
 
+  def CVT_ue5m3x2_f32 : BasicFlagsNVPTXInst<(outs B16:$dst),
+            (ins B32:$src1, B32:$src2), (ins CvtMode:$mode),
+            "cvt${mode:base}${mode:satfinite}.ue5m3x2.f32">,
+    Requires<[hasUE5M3TypeSupport]>;
+  def CVT_ue5m3x2_f16x2 : BasicFlagsNVPTXInst<(outs B16:$dst),
+            (ins B32:$src), (ins CvtMode:$mode),
+            "cvt${mode:base}${mode:satfinite}.ue5m3x2.f16x2">,
+    Requires<[hasUE5M3TypeSupport]>;
+  def CVT_ue5m3x2_bf16x2 : BasicFlagsNVPTXInst<(outs B16:$dst),
+            (ins B32:$src), (ins CvtMode:$mode),
+            "cvt${mode:base}${mode:satfinite}.ue5m3x2.bf16x2">,
+    Requires<[hasUE5M3TypeSupport]>;
+
+  class CVT_TO_UE5M3X2_SCALE_N1<dag ins, string SrcType, string SrcOps>
+      : NVPTXInst<(outs B16:$dst), ins,
+            "{{ \n\t" #
+            ".reg .b8 \t%b8_in; \n\t" #
+            "cvt.u8.u16 \t%b8_in, $scale; \n\t" #
+            "cvt${mode:base}${mode:satfinite}.scaled::n1::ue8m0.ue5m3x2." #
+            SrcType # " \t$dst, " # SrcOps # ", %b8_in; \n\t" #
+            "}}", []>,
+        Requires<[hasUE5M3TypeSupport]>;
+
+  def CVT_ue5m3x2_f32_scale_n1
+      : CVT_TO_UE5M3X2_SCALE_N1<(ins B32:$src1, B32:$src2, B16:$scale, CvtMode:$mode),
+                                "f32", "$src1, $src2">;
+  foreach type = ["f16x2", "bf16x2"] in
+    def CVT_ue5m3x2_ # type # _scale_n1
+        : CVT_TO_UE5M3X2_SCALE_N1<(ins B32:$src1, B16:$scale, CvtMode:$mode), type, "$src1">;
+        
 }
 
 def fpround_oneuse : OneUse1<fpround>;
diff --git a/llvm/lib/Target/NVPTX/NVPTXIntrinsics.td b/llvm/lib/Target/NVPTX/NVPTXIntrinsics.td
index 5e01b251ff9fd..7e59761bebda9 100644
--- a/llvm/lib/Target/NVPTX/NVPTXIntrinsics.td
+++ b/llvm/lib/Target/NVPTX/NVPTXIntrinsics.td
@@ -2588,7 +2588,7 @@ let Predicates = [hasNarrowFPConversionSupport] in {
             (CVT_bf16x2_ue8m0x2 $a)>;
 }
 
-// ue5m3x2 to f16x2 / bf16x2 conversions.
+// ue5m3x2 conversions.
 let Predicates = [hasUE5M3TypeSupport] in {
   def : Pat<(int_nvvm_ue5m3x2_to_f16x2_rn i16:$a),
             (CVT_f16x2_ue5m3x2 $a, CvtRN)>;
@@ -2602,6 +2602,41 @@ let Predicates = [hasUE5M3TypeSupport] in {
             (CVT_bf16x2_ue5m3x2_scale_n2 $a, $b, CvtRN)>;
   def : Pat<(int_nvvm_ue5m3x2_to_bf16x2_rn_satfinite_scale_n2_ue8m0 i16:$a, i16:$b),
             (CVT_bf16x2_ue5m3x2_scale_n2 $a, $b, CvtRN_SATFINITE)>;
+
+  foreach rnd = ["rn", "rz", "rp"] in {
+    foreach satfinite = ["", "_satfinite"] in {
+      defvar intrin = !cast<Intrinsic>("int_nvvm_ff_to_ue5m3x2_" # rnd # satfinite);
+      defvar mode = !cast<PatLeaf>("Cvt" # !toupper(rnd) # !toupper(satfinite));
+      def : Pat<(intrin f32:$a, f32:$b), (CVT_ue5m3x2_f32 $a, $b, mode)>;
+    }
+  }
+  foreach rnd = ["rn", "rz"] in {
+    foreach satfinite = ["", "_satfinite"] in {
+      defvar intrin = !cast<Intrinsic>("int_nvvm_ff_to_ue5m3x2_" # rnd # satfinite # "_scale_n1_ue8m0");
+      defvar mode = !cast<PatLeaf>("Cvt" # !toupper(rnd) # !toupper(satfinite));
+      def : Pat<(intrin f32:$a, f32:$b, i16:$c),
+                (CVT_ue5m3x2_f32_scale_n1 $a, $b, $c, mode)>;
+    }
+  }
+
+  foreach type = ["f16x2", "bf16x2"] in {
+    foreach rnd = ["rn", "rz", "rp"] in {
+      foreach satfinite = ["", "_satfinite"] in {
+        defvar intrin = !cast<Intrinsic>("int_nvvm_" # type # "_to_ue5m3x2_" # rnd # satfinite);
+        defvar cvt_inst = !cast<NVPTXInst>("CVT_ue5m3x2_" # type);
+        defvar mode = !cast<PatLeaf>("Cvt" # !toupper(rnd) # !toupper(satfinite));
+        def : Pat<(intrin B32:$a), (cvt_inst $a, mode)>;
+      }
+    }
+    foreach rnd = ["rn", "rz"] in {
+      foreach satfinite = ["", "_satfinite"] in {
+        defvar intrin = !cast<Intrinsic>("int_nvvm_" # type # "_to_ue5m3x2_" # rnd # satfinite # "_scale_n1_ue8m0");
+        defvar cvt_inst = !cast<NVPTXInst>("CVT_ue5m3x2_" # type # "_scale_n1");
+        defvar mode = !cast<PatLeaf>("Cvt" # !toupper(rnd) # !toupper(satfinite));
+        def : Pat<(intrin B32:$a, i16:$b), (cvt_inst $a, $b, mode)>;
+      }
+    }
+  }
 } // let Predicates = [hasUE5M3TypeSupport]
 
 def SDT_CVT_F32X4_TO_FPX4_RS_VEC :
diff --git a/llvm/test/CodeGen/NVPTX/convert-ue5m3x2.ll b/llvm/test/CodeGen/NVPTX/convert-ue5m3x2.ll
index f24b80e52f0bf..c2ccf34a61415 100644
--- a/llvm/test/CodeGen/NVPTX/convert-ue5m3x2.ll
+++ b/llvm/test/CodeGen/NVPTX/convert-ue5m3x2.ll
@@ -78,3 +78,554 @@ define <2 x bfloat> @test_ue5m3x2_to_bf16x2_rn_satfinite_scale_ue8m0(i16 %a, i16
   %val = call <2 x bfloat> @llvm.nvvm.ue5m3x2.to.bf16x2.rn.satfinite.scale.n2.ue8m0(i16 %a, i16 %b)
   ret <2 x bfloat> %val
 }
+
+define i16 @test_ff_to_ue5m3x2_rn(float %a, float %b) {
+; CHECK-LABEL: test_ff_to_ue5m3x2_rn(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<4>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_ff_to_ue5m3x2_rn_param_0];
+; CHECK-NEXT:    ld.param::func.b32 %r2, [test_ff_to_ue5m3x2_rn_param_1];
+; CHECK-NEXT:    cvt.rn.ue5m3x2.f32 %rs1, %r1, %r2;
+; CHECK-NEXT:    cvt.u32.u16 %r3, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r3;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.ff.to.ue5m3x2.rn(float %a, float %b)
+  ret i16 %val
+}
+
+define i16 @test_ff_to_ue5m3x2_rz(float %a, float %b) {
+; CHECK-LABEL: test_ff_to_ue5m3x2_rz(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<4>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_ff_to_ue5m3x2_rz_param_0];
+; CHECK-NEXT:    ld.param::func.b32 %r2, [test_ff_to_ue5m3x2_rz_param_1];
+; CHECK-NEXT:    cvt.rz.ue5m3x2.f32 %rs1, %r1, %r2;
+; CHECK-NEXT:    cvt.u32.u16 %r3, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r3;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.ff.to.ue5m3x2.rz(float %a, float %b)
+  ret i16 %val
+}
+
+define i16 @test_ff_to_ue5m3x2_rp(float %a, float %b) {
+; CHECK-LABEL: test_ff_to_ue5m3x2_rp(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<4>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_ff_to_ue5m3x2_rp_param_0];
+; CHECK-NEXT:    ld.param::func.b32 %r2, [test_ff_to_ue5m3x2_rp_param_1];
+; CHECK-NEXT:    cvt.rp.ue5m3x2.f32 %rs1, %r1, %r2;
+; CHECK-NEXT:    cvt.u32.u16 %r3, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r3;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.ff.to.ue5m3x2.rp(float %a, float %b)
+  ret i16 %val
+}
+
+define i16 @test_ff_to_ue5m3x2_rn_satfinite(float %a, float %b) {
+; CHECK-LABEL: test_ff_to_ue5m3x2_rn_satfinite(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<4>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_ff_to_ue5m3x2_rn_satfinite_param_0];
+; CHECK-NEXT:    ld.param::func.b32 %r2, [test_ff_to_ue5m3x2_rn_satfinite_param_1];
+; CHECK-NEXT:    cvt.rn.satfinite.ue5m3x2.f32 %rs1, %r1, %r2;
+; CHECK-NEXT:    cvt.u32.u16 %r3, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r3;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.ff.to.ue5m3x2.rn.satfinite(float %a, float %b)
+  ret i16 %val
+}
+
+define i16 @test_ff_to_ue5m3x2_rz_satfinite(float %a, float %b) {
+; CHECK-LABEL: test_ff_to_ue5m3x2_rz_satfinite(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<4>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_ff_to_ue5m3x2_rz_satfinite_param_0];
+; CHECK-NEXT:    ld.param::func.b32 %r2, [test_ff_to_ue5m3x2_rz_satfinite_param_1];
+; CHECK-NEXT:    cvt.rz.satfinite.ue5m3x2.f32 %rs1, %r1, %r2;
+; CHECK-NEXT:    cvt.u32.u16 %r3, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r3;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.ff.to.ue5m3x2.rz.satfinite(float %a, float %b)
+  ret i16 %val
+}
+
+define i16 @test_ff_to_ue5m3x2_rp_satfinite(float %a, float %b) {
+; CHECK-LABEL: test_ff_to_ue5m3x2_rp_satfinite(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<4>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_ff_to_ue5m3x2_rp_satfinite_param_0];
+; CHECK-NEXT:    ld.param::func.b32 %r2, [test_ff_to_ue5m3x2_rp_satfinite_param_1];
+; CHECK-NEXT:    cvt.rp.satfinite.ue5m3x2.f32 %rs1, %r1, %r2;
+; CHECK-NEXT:    cvt.u32.u16 %r3, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r3;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.ff.to.ue5m3x2.rp.satfinite(float %a, float %b)
+  ret i16 %val
+}
+
+define i16 @test_ff_to_ue5m3x2_rn_scale_n1_ue8m0(float %a, float %b, i16 %scale) {
+; CHECK-LABEL: test_ff_to_ue5m3x2_rn_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<4>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_ff_to_ue5m3x2_rn_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b32 %r2, [test_ff_to_ue5m3x2_rn_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_ff_to_ue5m3x2_rn_scale_n1_ue8m0_param_2];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rn.scaled::n1::ue8m0.ue5m3x2.f32 %rs2, %r1, %r2, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r3, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r3;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.ff.to.ue5m3x2.rn.scale.n1.ue8m0(float %a, float %b, i16 %scale)
+  ret i16 %val
+}
+
+define i16 @test_ff_to_ue5m3x2_rz_scale_n1_ue8m0(float %a, float %b, i16 %scale) {
+; CHECK-LABEL: test_ff_to_ue5m3x2_rz_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<4>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_ff_to_ue5m3x2_rz_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b32 %r2, [test_ff_to_ue5m3x2_rz_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_ff_to_ue5m3x2_rz_scale_n1_ue8m0_param_2];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rz.scaled::n1::ue8m0.ue5m3x2.f32 %rs2, %r1, %r2, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r3, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r3;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.ff.to.ue5m3x2.rz.scale.n1.ue8m0(float %a, float %b, i16 %scale)
+  ret i16 %val
+}
+
+define i16 @test_ff_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0(float %a, float %b, i16 %scale) {
+; CHECK-LABEL: test_ff_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<4>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_ff_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b32 %r2, [test_ff_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_ff_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0_param_2];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rn.satfinite.scaled::n1::ue8m0.ue5m3x2.f32 %rs2, %r1, %r2, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r3, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r3;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.ff.to.ue5m3x2.rn.satfinite.scale.n1.ue8m0(float %a, float %b, i16 %scale)
+  ret i16 %val
+}
+
+define i16 @test_ff_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0(float %a, float %b, i16 %scale) {
+; CHECK-LABEL: test_ff_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<4>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_ff_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b32 %r2, [test_ff_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_ff_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0_param_2];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rz.satfinite.scaled::n1::ue8m0.ue5m3x2.f32 %rs2, %r1, %r2, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r3, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r3;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.ff.to.ue5m3x2.rz.satfinite.scale.n1.ue8m0(float %a, float %b, i16 %scale)
+  ret i16 %val
+}
+
+; f16x2 -> ue5m3x2
+define i16 @test_f16x2_to_ue5m3x2_rn(<2 x half> %a) {
+; CHECK-LABEL: test_f16x2_to_ue5m3x2_rn(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_f16x2_to_ue5m3x2_rn_param_0];
+; CHECK-NEXT:    cvt.rn.ue5m3x2.f16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.f16x2.to.ue5m3x2.rn(<2 x half> %a)
+  ret i16 %val
+}
+
+define i16 @test_f16x2_to_ue5m3x2_rz(<2 x half> %a) {
+; CHECK-LABEL: test_f16x2_to_ue5m3x2_rz(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_f16x2_to_ue5m3x2_rz_param_0];
+; CHECK-NEXT:    cvt.rz.ue5m3x2.f16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.f16x2.to.ue5m3x2.rz(<2 x half> %a)
+  ret i16 %val
+}
+
+define i16 @test_f16x2_to_ue5m3x2_rp(<2 x half> %a) {
+; CHECK-LABEL: test_f16x2_to_ue5m3x2_rp(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_f16x2_to_ue5m3x2_rp_param_0];
+; CHECK-NEXT:    cvt.rp.ue5m3x2.f16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.f16x2.to.ue5m3x2.rp(<2 x half> %a)
+  ret i16 %val
+}
+
+define i16 @test_f16x2_to_ue5m3x2_rn_satfinite(<2 x half> %a) {
+; CHECK-LABEL: test_f16x2_to_ue5m3x2_rn_satfinite(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_f16x2_to_ue5m3x2_rn_satfinite_param_0];
+; CHECK-NEXT:    cvt.rn.satfinite.ue5m3x2.f16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.f16x2.to.ue5m3x2.rn.satfinite(<2 x half> %a)
+  ret i16 %val
+}
+
+define i16 @test_f16x2_to_ue5m3x2_rz_satfinite(<2 x half> %a) {
+; CHECK-LABEL: test_f16x2_to_ue5m3x2_rz_satfinite(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_f16x2_to_ue5m3x2_rz_satfinite_param_0];
+; CHECK-NEXT:    cvt.rz.satfinite.ue5m3x2.f16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.f16x2.to.ue5m3x2.rz.satfinite(<2 x half> %a)
+  ret i16 %val
+}
+
+define i16 @test_f16x2_to_ue5m3x2_rp_satfinite(<2 x half> %a) {
+; CHECK-LABEL: test_f16x2_to_ue5m3x2_rp_satfinite(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_f16x2_to_ue5m3x2_rp_satfinite_param_0];
+; CHECK-NEXT:    cvt.rp.satfinite.ue5m3x2.f16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.f16x2.to.ue5m3x2.rp.satfinite(<2 x half> %a)
+  ret i16 %val
+}
+
+define i16 @test_f16x2_to_ue5m3x2_rn_scale_n1_ue8m0(<2 x half> %a, i16 %scale) {
+; CHECK-LABEL: test_f16x2_to_ue5m3x2_rn_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_f16x2_to_ue5m3x2_rn_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_f16x2_to_ue5m3x2_rn_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rn.scaled::n1::ue8m0.ue5m3x2.f16x2 %rs2, %r1, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.f16x2.to.ue5m3x2.rn.scale.n1.ue8m0(<2 x half> %a, i16 %scale)
+  ret i16 %val
+}
+
+define i16 @test_f16x2_to_ue5m3x2_rz_scale_n1_ue8m0(<2 x half> %a, i16 %scale) {
+; CHECK-LABEL: test_f16x2_to_ue5m3x2_rz_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_f16x2_to_ue5m3x2_rz_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_f16x2_to_ue5m3x2_rz_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rz.scaled::n1::ue8m0.ue5m3x2.f16x2 %rs2, %r1, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.f16x2.to.ue5m3x2.rz.scale.n1.ue8m0(<2 x half> %a, i16 %scale)
+  ret i16 %val
+}
+
+define i16 @test_f16x2_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0(<2 x half> %a, i16 %scale) {
+; CHECK-LABEL: test_f16x2_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_f16x2_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_f16x2_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rn.satfinite.scaled::n1::ue8m0.ue5m3x2.f16x2 %rs2, %r1, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.f16x2.to.ue5m3x2.rn.satfinite.scale.n1.ue8m0(<2 x half> %a, i16 %scale)
+  ret i16 %val
+}
+
+define i16 @test_f16x2_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0(<2 x half> %a, i16 %scale) {
+; CHECK-LABEL: test_f16x2_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_f16x2_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_f16x2_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rz.satfinite.scaled::n1::ue8m0.ue5m3x2.f16x2 %rs2, %r1, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.f16x2.to.ue5m3x2.rz.satfinite.scale.n1.ue8m0(<2 x half> %a, i16 %scale)
+  ret i16 %val
+}
+
+define i16 @test_bf16x2_to_ue5m3x2_rn(<2 x bfloat> %a) {
+; CHECK-LABEL: test_bf16x2_to_ue5m3x2_rn(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_bf16x2_to_ue5m3x2_rn_param_0];
+; CHECK-NEXT:    cvt.rn.ue5m3x2.bf16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.bf16x2.to.ue5m3x2.rn(<2 x bfloat> %a)
+  ret i16 %val
+}
+
+define i16 @test_bf16x2_to_ue5m3x2_rz(<2 x bfloat> %a) {
+; CHECK-LABEL: test_bf16x2_to_ue5m3x2_rz(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_bf16x2_to_ue5m3x2_rz_param_0];
+; CHECK-NEXT:    cvt.rz.ue5m3x2.bf16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.bf16x2.to.ue5m3x2.rz(<2 x bfloat> %a)
+  ret i16 %val
+}
+
+define i16 @test_bf16x2_to_ue5m3x2_rp(<2 x bfloat> %a) {
+; CHECK-LABEL: test_bf16x2_to_ue5m3x2_rp(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_bf16x2_to_ue5m3x2_rp_param_0];
+; CHECK-NEXT:    cvt.rp.ue5m3x2.bf16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.bf16x2.to.ue5m3x2.rp(<2 x bfloat> %a)
+  ret i16 %val
+}
+
+define i16 @test_bf16x2_to_ue5m3x2_rn_satfinite(<2 x bfloat> %a) {
+; CHECK-LABEL: test_bf16x2_to_ue5m3x2_rn_satfinite(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_bf16x2_to_ue5m3x2_rn_satfinite_param_0];
+; CHECK-NEXT:    cvt.rn.satfinite.ue5m3x2.bf16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.bf16x2.to.ue5m3x2.rn.satfinite(<2 x bfloat> %a)
+  ret i16 %val
+}
+
+define i16 @test_bf16x2_to_ue5m3x2_rz_satfinite(<2 x bfloat> %a) {
+; CHECK-LABEL: test_bf16x2_to_ue5m3x2_rz_satfinite(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_bf16x2_to_ue5m3x2_rz_satfinite_param_0];
+; CHECK-NEXT:    cvt.rz.satfinite.ue5m3x2.bf16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.bf16x2.to.ue5m3x2.rz.satfinite(<2 x bfloat> %a)
+  ret i16 %val
+}
+
+define i16 @test_bf16x2_to_ue5m3x2_rp_satfinite(<2 x bfloat> %a) {
+; CHECK-LABEL: test_bf16x2_to_ue5m3x2_rp_satfinite(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<2>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_bf16x2_to_ue5m3x2_rp_satfinite_param_0];
+; CHECK-NEXT:    cvt.rp.satfinite.ue5m3x2.bf16x2 %rs1, %r1;
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs1;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.bf16x2.to.ue5m3x2.rp.satfinite(<2 x bfloat> %a)
+  ret i16 %val
+}
+
+define i16 @test_bf16x2_to_ue5m3x2_rn_scale_n1_ue8m0(<2 x bfloat> %a, i16 %scale) {
+; CHECK-LABEL: test_bf16x2_to_ue5m3x2_rn_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_bf16x2_to_ue5m3x2_rn_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_bf16x2_to_ue5m3x2_rn_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rn.scaled::n1::ue8m0.ue5m3x2.bf16x2 %rs2, %r1, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.bf16x2.to.ue5m3x2.rn.scale.n1.ue8m0(<2 x bfloat> %a, i16 %scale)
+  ret i16 %val
+}
+
+define i16 @test_bf16x2_to_ue5m3x2_rz_scale_n1_ue8m0(<2 x bfloat> %a, i16 %scale) {
+; CHECK-LABEL: test_bf16x2_to_ue5m3x2_rz_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_bf16x2_to_ue5m3x2_rz_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_bf16x2_to_ue5m3x2_rz_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rz.scaled::n1::ue8m0.ue5m3x2.bf16x2 %rs2, %r1, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.bf16x2.to.ue5m3x2.rz.scale.n1.ue8m0(<2 x bfloat> %a, i16 %scale)
+  ret i16 %val
+}
+
+define i16 @test_bf16x2_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0(<2 x bfloat> %a, i16 %scale) {
+; CHECK-LABEL: test_bf16x2_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_bf16x2_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_bf16x2_to_ue5m3x2_rn_satfinite_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rn.satfinite.scaled::n1::ue8m0.ue5m3x2.bf16x2 %rs2, %r1, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.bf16x2.to.ue5m3x2.rn.satfinite.scale.n1.ue8m0(<2 x bfloat> %a, i16 %scale)
+  ret i16 %val
+}
+
+define i16 @test_bf16x2_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0(<2 x bfloat> %a, i16 %scale) {
+; CHECK-LABEL: test_bf16x2_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0(
+; CHECK:       {
+; CHECK-NEXT:    .reg .b16 %rs<3>;
+; CHECK-NEXT:    .reg .b32 %r<3>;
+; CHECK-EMPTY:
+; CHECK-NEXT:  // %bb.0:
+; CHECK-NEXT:    ld.param::func.b32 %r1, [test_bf16x2_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0_param_0];
+; CHECK-NEXT:    ld.param::func.b16 %rs1, [test_bf16x2_to_ue5m3x2_rz_satfinite_scale_n1_ue8m0_param_1];
+; CHECK-NEXT:    {
+; CHECK-NEXT:    .reg .b8 %b8_in;
+; CHECK-NEXT:    cvt.u8.u16 %b8_in, %rs1;
+; CHECK-NEXT:    cvt.rz.satfinite.scaled::n1::ue8m0.ue5m3x2.bf16x2 %rs2, %r1, %b8_in;
+; CHECK-NEXT:    }
+; CHECK-NEXT:    cvt.u32.u16 %r2, %rs2;
+; CHECK-NEXT:    st.param::func.b32 [func_retval0], %r2;
+; CHECK-NEXT:    ret;
+  %val = call i16 @llvm.nvvm.bf16x2.to.ue5m3x2.rz.satfinite.scale.n1.ue8m0(<2 x bfloat> %a, i16 %scale)
+  ret i16 %val
+}

>From 5a195a6e614f730d316463028b65d594dbf5be07 Mon Sep 17 00:00:00 2001
From: DharuniRAcharya <dharunira at nvidia.com>
Date: Wed, 26 Aug 2026 06:25:22 +0000
Subject: [PATCH 2/2] Address comments

---
 llvm/include/llvm/IR/IntrinsicsNVVM.td   | 26 ++++++-----
 llvm/lib/Target/NVPTX/NVPTXInstrInfo.td  | 26 +++++------
 llvm/lib/Target/NVPTX/NVPTXIntrinsics.td | 56 ++++++++++++------------
 3 files changed, 53 insertions(+), 55 deletions(-)

diff --git a/llvm/include/llvm/IR/IntrinsicsNVVM.td b/llvm/include/llvm/IR/IntrinsicsNVVM.td
index 4cc536f88be7f..2d3332e80050d 100644
--- a/llvm/include/llvm/IR/IntrinsicsNVVM.td
+++ b/llvm/include/llvm/IR/IntrinsicsNVVM.td
@@ -2016,30 +2016,32 @@ let TargetPrefix = "nvvm" in {
         : PureIntrinsic<[llvm_v2bf16_ty], [llvm_i16_ty, llvm_i16_ty]>;
   }
 
-  foreach rnd = ["rn", "rz", "rp"] in {
-    foreach satfinite = ["", "_satfinite"] in {
+  foreach satfinite = ["", "_satfinite"] in {
+    foreach rnd = ["rn", "rz"] in {
       def int_nvvm_ff_to_ue5m3x2_ # rnd # satfinite
           : PureIntrinsic<[llvm_i16_ty], [llvm_float_ty, llvm_float_ty]>;
+      def int_nvvm_ff_to_ue5m3x2_ # rnd # satfinite # _scale_n1_ue8m0
+          : PureIntrinsic<[llvm_i16_ty], [llvm_float_ty, llvm_float_ty, llvm_i16_ty]>;
 
       def int_nvvm_f16x2_to_ue5m3x2_ # rnd # satfinite
           : PureIntrinsic<[llvm_i16_ty], [llvm_v2f16_ty]>;
+      def int_nvvm_f16x2_to_ue5m3x2_ # rnd # satfinite # _scale_n1_ue8m0
+          : PureIntrinsic<[llvm_i16_ty], [llvm_v2f16_ty, llvm_i16_ty]>;
 
       def int_nvvm_bf16x2_to_ue5m3x2_ # rnd # satfinite
           : PureIntrinsic<[llvm_i16_ty], [llvm_v2bf16_ty]>;
+      def int_nvvm_bf16x2_to_ue5m3x2_ # rnd # satfinite # _scale_n1_ue8m0
+          : PureIntrinsic<[llvm_i16_ty], [llvm_v2bf16_ty, llvm_i16_ty]>;
     }
-  }
 
-  foreach rnd = ["rn", "rz"] in {
-    foreach satfinite = ["", "_satfinite"] in {
-      def int_nvvm_ff_to_ue5m3x2_ # rnd # satfinite # _scale_n1_ue8m0
-          : PureIntrinsic<[llvm_i16_ty], [llvm_float_ty, llvm_float_ty, llvm_i16_ty]>;
+    def int_nvvm_ff_to_ue5m3x2_rp # satfinite
+        : PureIntrinsic<[llvm_i16_ty], [llvm_float_ty, llvm_float_ty]>;
 
-      def int_nvvm_f16x2_to_ue5m3x2_ # rnd # satfinite # _scale_n1_ue8m0
-          : PureIntrinsic<[llvm_i16_ty], [llvm_v2f16_ty, llvm_i16_ty]>;
+    def int_nvvm_f16x2_to_ue5m3x2_rp # satfinite
+        : PureIntrinsic<[llvm_i16_ty], [llvm_v2f16_ty]>;
 
-      def int_nvvm_bf16x2_to_ue5m3x2_ # rnd # satfinite # _scale_n1_ue8m0
-          : PureIntrinsic<[llvm_i16_ty], [llvm_v2bf16_ty, llvm_i16_ty]>;
-    }
+    def int_nvvm_bf16x2_to_ue5m3x2_rp # satfinite
+        : PureIntrinsic<[llvm_i16_ty], [llvm_v2bf16_ty]>;
   }
 
   //
diff --git a/llvm/lib/Target/NVPTX/NVPTXInstrInfo.td b/llvm/lib/Target/NVPTX/NVPTXInstrInfo.td
index 34e9a698d9d47..7b0c6c131e22d 100644
--- a/llvm/lib/Target/NVPTX/NVPTXInstrInfo.td
+++ b/llvm/lib/Target/NVPTX/NVPTXInstrInfo.td
@@ -938,19 +938,6 @@ let Predicates = [hasS2F6X2ConversionSupport] in {
             "cvt${mode:base}${mode:satfinite}.scaled::n2::ue8m0.bf16x2.ue5m3x2">,
     Requires<[hasUE5M3TypeSupport]>;
 
-  def CVT_ue5m3x2_f32 : BasicFlagsNVPTXInst<(outs B16:$dst),
-            (ins B32:$src1, B32:$src2), (ins CvtMode:$mode),
-            "cvt${mode:base}${mode:satfinite}.ue5m3x2.f32">,
-    Requires<[hasUE5M3TypeSupport]>;
-  def CVT_ue5m3x2_f16x2 : BasicFlagsNVPTXInst<(outs B16:$dst),
-            (ins B32:$src), (ins CvtMode:$mode),
-            "cvt${mode:base}${mode:satfinite}.ue5m3x2.f16x2">,
-    Requires<[hasUE5M3TypeSupport]>;
-  def CVT_ue5m3x2_bf16x2 : BasicFlagsNVPTXInst<(outs B16:$dst),
-            (ins B32:$src), (ins CvtMode:$mode),
-            "cvt${mode:base}${mode:satfinite}.ue5m3x2.bf16x2">,
-    Requires<[hasUE5M3TypeSupport]>;
-
   class CVT_TO_UE5M3X2_SCALE_N1<dag ins, string SrcType, string SrcOps>
       : NVPTXInst<(outs B16:$dst), ins,
             "{{ \n\t" #
@@ -961,13 +948,22 @@ let Predicates = [hasS2F6X2ConversionSupport] in {
             "}}", []>,
         Requires<[hasUE5M3TypeSupport]>;
 
+  def CVT_ue5m3x2_f32 : BasicFlagsNVPTXInst<(outs B16:$dst),
+            (ins B32:$src1, B32:$src2), (ins CvtMode:$mode),
+            "cvt${mode:base}${mode:satfinite}.ue5m3x2.f32">,
+    Requires<[hasUE5M3TypeSupport]>;
   def CVT_ue5m3x2_f32_scale_n1
       : CVT_TO_UE5M3X2_SCALE_N1<(ins B32:$src1, B32:$src2, B16:$scale, CvtMode:$mode),
                                 "f32", "$src1, $src2">;
-  foreach type = ["f16x2", "bf16x2"] in
+
+  foreach type = ["f16x2", "bf16x2"] in {
+    def CVT_ue5m3x2_ # type : BasicFlagsNVPTXInst<(outs B16:$dst),
+              (ins B32:$src), (ins CvtMode:$mode),
+              "cvt${mode:base}${mode:satfinite}.ue5m3x2." # type>,
+      Requires<[hasUE5M3TypeSupport]>;
     def CVT_ue5m3x2_ # type # _scale_n1
         : CVT_TO_UE5M3X2_SCALE_N1<(ins B32:$src1, B16:$scale, CvtMode:$mode), type, "$src1">;
-        
+  }
 }
 
 def fpround_oneuse : OneUse1<fpround>;
diff --git a/llvm/lib/Target/NVPTX/NVPTXIntrinsics.td b/llvm/lib/Target/NVPTX/NVPTXIntrinsics.td
index 7e59761bebda9..cb09885f6163f 100644
--- a/llvm/lib/Target/NVPTX/NVPTXIntrinsics.td
+++ b/llvm/lib/Target/NVPTX/NVPTXIntrinsics.td
@@ -2603,38 +2603,38 @@ let Predicates = [hasUE5M3TypeSupport] in {
   def : Pat<(int_nvvm_ue5m3x2_to_bf16x2_rn_satfinite_scale_n2_ue8m0 i16:$a, i16:$b),
             (CVT_bf16x2_ue5m3x2_scale_n2 $a, $b, CvtRN_SATFINITE)>;
 
-  foreach rnd = ["rn", "rz", "rp"] in {
-    foreach satfinite = ["", "_satfinite"] in {
-      defvar intrin = !cast<Intrinsic>("int_nvvm_ff_to_ue5m3x2_" # rnd # satfinite);
-      defvar mode = !cast<PatLeaf>("Cvt" # !toupper(rnd) # !toupper(satfinite));
-      def : Pat<(intrin f32:$a, f32:$b), (CVT_ue5m3x2_f32 $a, $b, mode)>;
-    }
-  }
-  foreach rnd = ["rn", "rz"] in {
-    foreach satfinite = ["", "_satfinite"] in {
-      defvar intrin = !cast<Intrinsic>("int_nvvm_ff_to_ue5m3x2_" # rnd # satfinite # "_scale_n1_ue8m0");
-      defvar mode = !cast<PatLeaf>("Cvt" # !toupper(rnd) # !toupper(satfinite));
-      def : Pat<(intrin f32:$a, f32:$b, i16:$c),
+  foreach satfinite = ["", "satfinite"] in {
+    defvar sat = !if(!empty(satfinite), "", "_" # satfinite);
+    foreach rnd = ["rn", "rz"] in {
+      defvar mode = !cast<PatLeaf>("Cvt" # !toupper(rnd) # !toupper(sat));
+
+      defvar ff_intrin = !cast<Intrinsic>("int_nvvm_ff_to_ue5m3x2_" # rnd # sat);
+      def : Pat<(ff_intrin f32:$a, f32:$b), (CVT_ue5m3x2_f32 $a, $b, mode)>;
+
+      defvar ff_scale_intrin = !cast<Intrinsic>("int_nvvm_ff_to_ue5m3x2_" # rnd # sat # "_scale_n1_ue8m0");
+      def : Pat<(ff_scale_intrin f32:$a, f32:$b, i16:$c),
                 (CVT_ue5m3x2_f32_scale_n1 $a, $b, $c, mode)>;
-    }
-  }
 
-  foreach type = ["f16x2", "bf16x2"] in {
-    foreach rnd = ["rn", "rz", "rp"] in {
-      foreach satfinite = ["", "_satfinite"] in {
-        defvar intrin = !cast<Intrinsic>("int_nvvm_" # type # "_to_ue5m3x2_" # rnd # satfinite);
-        defvar cvt_inst = !cast<NVPTXInst>("CVT_ue5m3x2_" # type);
-        defvar mode = !cast<PatLeaf>("Cvt" # !toupper(rnd) # !toupper(satfinite));
-        def : Pat<(intrin B32:$a), (cvt_inst $a, mode)>;
+      foreach type = ["f16x2", "bf16x2"] in {
+        defvar intrin = !cast<Intrinsic>("int_nvvm_" # type # "_to_ue5m3x2_" # rnd # sat);
+        def : Pat<(intrin B32:$a),
+                  (!cast<NVPTXInst>("CVT_ue5m3x2_" # type) $a, mode)>;
+
+        defvar scale_intrin = !cast<Intrinsic>("int_nvvm_" # type # "_to_ue5m3x2_" # rnd # sat # "_scale_n1_ue8m0");
+        def : Pat<(scale_intrin B32:$a, i16:$b),
+                  (!cast<NVPTXInst>("CVT_ue5m3x2_" # type # "_scale_n1") $a, $b, mode)>;
       }
     }
-    foreach rnd = ["rn", "rz"] in {
-      foreach satfinite = ["", "_satfinite"] in {
-        defvar intrin = !cast<Intrinsic>("int_nvvm_" # type # "_to_ue5m3x2_" # rnd # satfinite # "_scale_n1_ue8m0");
-        defvar cvt_inst = !cast<NVPTXInst>("CVT_ue5m3x2_" # type # "_scale_n1");
-        defvar mode = !cast<PatLeaf>("Cvt" # !toupper(rnd) # !toupper(satfinite));
-        def : Pat<(intrin B32:$a, i16:$b), (cvt_inst $a, $b, mode)>;
-      }
+
+    defvar rp_mode = !cast<PatLeaf>("CvtRP" # !toupper(sat));
+
+    defvar ff_rp_intrin = !cast<Intrinsic>("int_nvvm_ff_to_ue5m3x2_rp" # sat);
+    def : Pat<(ff_rp_intrin f32:$a, f32:$b), (CVT_ue5m3x2_f32 $a, $b, rp_mode)>;
+
+    foreach type = ["f16x2", "bf16x2"] in {
+      defvar rp_intrin = !cast<Intrinsic>("int_nvvm_" # type # "_to_ue5m3x2_rp" # sat);
+      def : Pat<(rp_intrin B32:$a),
+                (!cast<NVPTXInst>("CVT_ue5m3x2_" # type) $a, rp_mode)>;
     }
   }
 } // let Predicates = [hasUE5M3TypeSupport]



More information about the llvm-commits mailing list