[llvm] [CodeGen] Add scalarization fallback for multi-result scalable intrinsics (PR #218634)
Mattéo Rizza Murgier via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 14 02:47:10 PDT 2026
================
@@ -0,0 +1,140 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=pre-isel-intrinsic-lowering -vector-library=sleefgnuabi -S -mattr=+sve < %s | FileCheck %s
+target triple = "aarch64"
+
+; No vector variants exist for these types, so the intrinsics get scalarized.
+define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_sincos_f32(<vscale x 2 x float> %input) {
+; CHECK-LABEL: define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_sincos_f32(
+; CHECK-SAME: <vscale x 2 x float> [[INPUT:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP2:%.*]] = mul nuw i64 [[TMP1]], 2
+; CHECK-NEXT: br label %[[BB3:.*]]
+; CHECK: [[BB3]]:
+; CHECK-NEXT: [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP13:%.*]], %[[BB3]] ]
+; CHECK-NEXT: [[TMP5:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP10:%.*]], %[[BB3]] ]
+; CHECK-NEXT: [[TMP6:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP12:%.*]], %[[BB3]] ]
+; CHECK-NEXT: [[TMP7:%.*]] = extractelement <vscale x 2 x float> [[INPUT]], i64 [[TMP4]]
+; CHECK-NEXT: [[TMP8:%.*]] = call { float, float } @llvm.sincos.f32(float [[TMP7]])
+; CHECK-NEXT: [[TMP9:%.*]] = extractvalue { float, float } [[TMP8]], 0
+; CHECK-NEXT: [[TMP10]] = insertelement <vscale x 2 x float> [[TMP5]], float [[TMP9]], i64 [[TMP4]]
+; CHECK-NEXT: [[TMP11:%.*]] = extractvalue { float, float } [[TMP8]], 1
+; CHECK-NEXT: [[TMP12]] = insertelement <vscale x 2 x float> [[TMP6]], float [[TMP11]], i64 [[TMP4]]
+; CHECK-NEXT: [[TMP13]] = add i64 [[TMP4]], 1
+; CHECK-NEXT: [[TMP14:%.*]] = icmp eq i64 [[TMP13]], [[TMP2]]
+; CHECK-NEXT: br i1 [[TMP14]], label %[[BB15:.*]], label %[[BB3]]
+; CHECK: [[BB15]]:
+; CHECK-NEXT: [[TMP16:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } poison, <vscale x 2 x float> [[TMP10]], 0
+; CHECK-NEXT: [[TMP17:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP16]], <vscale x 2 x float> [[TMP12]], 1
+; CHECK-NEXT: ret { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP17]]
+;
+ %output = call { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.sincos.nxv2f32(<vscale x 2 x float> %input)
+ ret { <vscale x 2 x float>, <vscale x 2 x float> } %output
+}
+
+define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_sincospi_f32(<vscale x 2 x float> %input) {
+; CHECK-LABEL: define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_sincospi_f32(
+; CHECK-SAME: <vscale x 2 x float> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP2:%.*]] = mul nuw i64 [[TMP1]], 2
+; CHECK-NEXT: br label %[[BB3:.*]]
+; CHECK: [[BB3]]:
+; CHECK-NEXT: [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP13:%.*]], %[[BB3]] ]
+; CHECK-NEXT: [[TMP5:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP10:%.*]], %[[BB3]] ]
+; CHECK-NEXT: [[TMP6:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP12:%.*]], %[[BB3]] ]
+; CHECK-NEXT: [[TMP7:%.*]] = extractelement <vscale x 2 x float> [[INPUT]], i64 [[TMP4]]
+; CHECK-NEXT: [[TMP8:%.*]] = call { float, float } @llvm.sincospi.f32(float [[TMP7]])
+; CHECK-NEXT: [[TMP9:%.*]] = extractvalue { float, float } [[TMP8]], 0
+; CHECK-NEXT: [[TMP10]] = insertelement <vscale x 2 x float> [[TMP5]], float [[TMP9]], i64 [[TMP4]]
+; CHECK-NEXT: [[TMP11:%.*]] = extractvalue { float, float } [[TMP8]], 1
+; CHECK-NEXT: [[TMP12]] = insertelement <vscale x 2 x float> [[TMP6]], float [[TMP11]], i64 [[TMP4]]
+; CHECK-NEXT: [[TMP13]] = add i64 [[TMP4]], 1
+; CHECK-NEXT: [[TMP14:%.*]] = icmp eq i64 [[TMP13]], [[TMP2]]
+; CHECK-NEXT: br i1 [[TMP14]], label %[[BB15:.*]], label %[[BB3]]
+; CHECK: [[BB15]]:
+; CHECK-NEXT: [[TMP16:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } poison, <vscale x 2 x float> [[TMP10]], 0
+; CHECK-NEXT: [[TMP17:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP16]], <vscale x 2 x float> [[TMP12]], 1
+; CHECK-NEXT: ret { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP17]]
+;
+ %output = call { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.sincospi.nxv2f32(<vscale x 2 x float> %input)
+ ret { <vscale x 2 x float>, <vscale x 2 x float> } %output
+}
+
+define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_modf_f32(<vscale x 2 x float> %input) {
+; CHECK-LABEL: define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_modf_f32(
+; CHECK-SAME: <vscale x 2 x float> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP2:%.*]] = mul nuw i64 [[TMP1]], 2
+; CHECK-NEXT: br label %[[BB3:.*]]
+; CHECK: [[BB3]]:
+; CHECK-NEXT: [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP13:%.*]], %[[BB3]] ]
+; CHECK-NEXT: [[TMP5:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP10:%.*]], %[[BB3]] ]
+; CHECK-NEXT: [[TMP6:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP12:%.*]], %[[BB3]] ]
+; CHECK-NEXT: [[TMP7:%.*]] = extractelement <vscale x 2 x float> [[INPUT]], i64 [[TMP4]]
+; CHECK-NEXT: [[TMP8:%.*]] = call { float, float } @llvm.modf.f32(float [[TMP7]])
+; CHECK-NEXT: [[TMP9:%.*]] = extractvalue { float, float } [[TMP8]], 0
+; CHECK-NEXT: [[TMP10]] = insertelement <vscale x 2 x float> [[TMP5]], float [[TMP9]], i64 [[TMP4]]
+; CHECK-NEXT: [[TMP11:%.*]] = extractvalue { float, float } [[TMP8]], 1
+; CHECK-NEXT: [[TMP12]] = insertelement <vscale x 2 x float> [[TMP6]], float [[TMP11]], i64 [[TMP4]]
+; CHECK-NEXT: [[TMP13]] = add i64 [[TMP4]], 1
+; CHECK-NEXT: [[TMP14:%.*]] = icmp eq i64 [[TMP13]], [[TMP2]]
+; CHECK-NEXT: br i1 [[TMP14]], label %[[BB15:.*]], label %[[BB3]]
+; CHECK: [[BB15]]:
+; CHECK-NEXT: [[TMP16:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } poison, <vscale x 2 x float> [[TMP10]], 0
+; CHECK-NEXT: [[TMP17:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP16]], <vscale x 2 x float> [[TMP12]], 1
+; CHECK-NEXT: ret { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP17]]
+;
+ %output = call { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.modf.nxv2f32(<vscale x 2 x float> %input)
+ ret { <vscale x 2 x float>, <vscale x 2 x float> } %output
+}
+
+; A matching libcall exists, scalarization is skipped.
+define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_sincos_f64(<vscale x 2 x double> %input) {
+; CHECK-LABEL: define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_sincos_f64(
+; CHECK-SAME: <vscale x 2 x double> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[OUTPUT:%.*]] = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincos.nxv2f64(<vscale x 2 x double> [[INPUT]])
+; CHECK-NEXT: ret { <vscale x 2 x double>, <vscale x 2 x double> } [[OUTPUT]]
+;
+ %output = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincos.nxv2f64(<vscale x 2 x double> %input)
+ ret { <vscale x 2 x double>, <vscale x 2 x double> } %output
+}
+
+define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_sincospi_f64(<vscale x 2 x double> %input) {
+; CHECK-LABEL: define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_sincospi_f64(
+; CHECK-SAME: <vscale x 2 x double> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[OUTPUT:%.*]] = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincospi.nxv2f64(<vscale x 2 x double> [[INPUT]])
+; CHECK-NEXT: ret { <vscale x 2 x double>, <vscale x 2 x double> } [[OUTPUT]]
+;
+ %output = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincospi.nxv2f64(<vscale x 2 x double> %input)
+ ret { <vscale x 2 x double>, <vscale x 2 x double> } %output
+}
+
+define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_modf_f64(<vscale x 2 x double> %input) {
+; CHECK-LABEL: define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_modf_f64(
+; CHECK-SAME: <vscale x 2 x double> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[OUTPUT:%.*]] = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.modf.nxv2f64(<vscale x 2 x double> [[INPUT]])
+; CHECK-NEXT: ret { <vscale x 2 x double>, <vscale x 2 x double> } [[OUTPUT]]
+;
+ %output = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.modf.nxv2f64(<vscale x 2 x double> %input)
+ ret { <vscale x 2 x double>, <vscale x 2 x double> } %output
+}
+
+; Fixed-length vectors are left for the vector legalizer to unroll.
+define { <2 x float>, <2 x float> } @fixed_vec_sincos(<2 x float> %input) {
+; CHECK-LABEL: define { <2 x float>, <2 x float> } @fixed_vec_sincos(
+; CHECK-SAME: <2 x float> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[OUTPUT:%.*]] = call { <2 x float>, <2 x float> } @llvm.sincos.v2f32(<2 x float> [[INPUT]])
+; CHECK-NEXT: ret { <2 x float>, <2 x float> } [[OUTPUT]]
+;
+ %output = call { <2 x float>, <2 x float> } @llvm.sincos.v2f32(<2 x float> %input)
+ ret { <2 x float>, <2 x float> } %output
+}
+
+declare { <2 x float>, <2 x float> } @llvm.sincos.v2f32(<2 x float>) #0
+declare { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.sincos.nxv2f32(<vscale x 2 x float>) #0
+declare { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincos.nxv2f64(<vscale x 2 x double>) #0
+declare { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.modf.nxv2f32(<vscale x 2 x float>) #0
+declare { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.modf.nxv2f64(<vscale x 2 x double>) #0
+declare { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.sincospi.nxv2f32(<vscale x 2 x float>) #0
+declare { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincospi.nxv2f64(<vscale x 2 x double>) #0
+
----------------
matteo-rm wrote:
Tested with `half` and `bfloat`. The behavior is unchanged: the same assembly is produced by `llc` for `modf` and `sincos`, and `sincospi` errors with `no libcall available for fsincospi` in both cases.
https://github.com/llvm/llvm-project/pull/218634
More information about the llvm-commits
mailing list