[llvm] [NVPTX] Add support for f32x2 mixed-precision add/sub (PR #221957)
Durgadoss R via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 10 02:27:22 PDT 2026
================
@@ -0,0 +1,264 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s -mtriple=nvptx64 -mcpu=sm_107f -mattr=+ptx94 | FileCheck %s
+; RUN: %if ptxas-sm_107f && ptxas-isa-9.4 %{ llc < %s -mtriple=nvptx64 -mcpu=sm_107f -mattr=+ptx94 | %ptxas-verify -arch=sm_107f %}
+
+;
+; Cases where the mixed-precision instruction must not be selected.
+;
+
+;
+; F16x2
+;
+
+; multiple uses
+
+define <2 x half> @add_rz_f16x2_f32x2_extra_use(<2 x float> %a, <2 x float> %b, ptr %p) {
+; CHECK-LABEL: add_rz_f16x2_f32x2_extra_use(
+; CHECK: {
+; CHECK-NEXT: .reg .b32 %r<4>;
+; CHECK-NEXT: .reg .b64 %rd<5>;
+; CHECK-EMPTY:
+; CHECK-NEXT: // %bb.0:
+; CHECK-NEXT: ld.param::func.b64 %rd1, [add_rz_f16x2_f32x2_extra_use_param_0];
+; CHECK-NEXT: ld.param::func.b64 %rd2, [add_rz_f16x2_f32x2_extra_use_param_1];
+; CHECK-NEXT: add.rz.ftz.f32x2 %rd3, %rd1, %rd2;
+; CHECK-NEXT: ld.param::func.b64 %rd4, [add_rz_f16x2_f32x2_extra_use_param_2];
+; CHECK-NEXT: st.b64 [%rd4], %rd3;
+; CHECK-NEXT: mov.b64 {%r1, %r2}, %rd3;
+; CHECK-NEXT: cvt.rz.f16x2.f32 %r3, %r2, %r1;
+; CHECK-NEXT: st.param::func.b32 [func_retval0], %r3;
+; CHECK-NEXT: ret;
+ %sum = call <2 x float> @llvm.nvvm.fadd.ftz.v2f32(<2 x float> %a, <2 x float> %b, i32 0)
+ store <2 x float> %sum, ptr %p
----------------
durga4github wrote:
nit: let us move the `; extra use` comment from line 13 to here.. so that it is easy to contrast with the example in the valid donwconvert file.
https://github.com/llvm/llvm-project/pull/221957
More information about the llvm-commits
mailing list