[llvm] [CodeGen] Add scalarization fallback for multi-result scalable intrinsics (PR #218634)

Mattéo Rizza Murgier via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 25 01:23:56 PDT 2026


https://github.com/matteo-rm created https://github.com/llvm/llvm-project/pull/218634

Muti-result scalable intrinsics such as `sincos`, `sincospi` and `modf` currently have no fallback when no veclib implementation is available, resulting in a crash.

This PR adds support for those intrinsics in `PreISelIntrinsicLowering` and updates the scalarization util to support functions that return structs of vectors.

Example:

```llvm
; llc -mtriple=aarch64 -mattr=+sve crash.ll
define { <vscale x 2 x float>, <vscale x 2 x float> } @f(<vscale x 2 x float> %x) {
  %r = call { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.sincos.nxv2f32(<vscale x 2 x float> %x)
  ret { <vscale x 2 x float>, <vscale x 2 x float> } %r
}
```

NOTE: support for `frexp` which returns a multi-type struct will follow separately.

>From 4cd236f9134b96b7623cf638e67619af053c5c74 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Matt=C3=A9o=20Rizza=20Murgier?=
 <matteo.rizza-murgier at sipearl.com>
Date: Mon, 24 Aug 2026 18:31:35 +0200
Subject: [PATCH] [CodeGen] Add scalarization fallback for multi-result
 scalable intrinsics

---
 .../Transforms/Utils/LowerVectorIntrinsics.h  |   5 +-
 llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp |  32 ++++
 llvm/lib/CodeGen/TargetLoweringBase.cpp       |   6 +
 .../Utils/LowerVectorIntrinsics.cpp           |  40 ++++-
 .../AArch64/expand-exp.ll                     |   4 +-
 .../AArch64/expand-fp-math.ll                 |  48 +++---
 .../AArch64/expand-log.ll                     |   4 +-
 .../AArch64/expand-multi-result.ll            | 140 ++++++++++++++++++
 8 files changed, 241 insertions(+), 38 deletions(-)
 create mode 100644 llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-multi-result.ll

diff --git a/llvm/include/llvm/Transforms/Utils/LowerVectorIntrinsics.h b/llvm/include/llvm/Transforms/Utils/LowerVectorIntrinsics.h
index 44c70a99b4306..7ed96823e908d 100644
--- a/llvm/include/llvm/Transforms/Utils/LowerVectorIntrinsics.h
+++ b/llvm/include/llvm/Transforms/Utils/LowerVectorIntrinsics.h
@@ -21,8 +21,9 @@ namespace llvm {
 class CallInst;
 class Module;
 
-/// Lower \p CI as a loop. \p CI is a unary intrinsic with a vector argument and
-/// is deleted and replaced with a loop.
+/// Lower \p CI as a loop. \p CI is a unary intrinsic with a vector argument,
+/// returning either a vector or a struct of vectors of the same type. \p CI is
+/// deleted and replaced with a loop.
 LLVM_ABI bool lowerUnaryVectorIntrinsicAsLoop(Module &M, CallInst *CI);
 
 } // namespace llvm
diff --git a/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp b/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
index 24788148b9f15..4f1d220f329f4 100644
--- a/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
+++ b/llvm/lib/CodeGen/PreISelIntrinsicLowering.cpp
@@ -20,6 +20,7 @@
 #include "llvm/CodeGen/ExpandVectorPredication.h"
 #include "llvm/CodeGen/LibcallLoweringInfo.h"
 #include "llvm/CodeGen/Passes.h"
+#include "llvm/CodeGen/RuntimeLibcallUtil.h"
 #include "llvm/CodeGen/TargetLowering.h"
 #include "llvm/CodeGen/TargetPassConfig.h"
 #include "llvm/IR/Function.h"
@@ -808,6 +809,37 @@ bool PreISelIntrinsicLowering::lowerIntrinsics(Module &M) const {
         return lowerUnaryVectorIntrinsicAsLoop(M, CI);
       });
       break;
+    case Intrinsic::modf:
+    case Intrinsic::sincos:
+    case Intrinsic::sincospi:
+      Changed |= forEachCall(F, [&](CallInst *CI) {
+        Type *Ty = CI->getArgOperand(0)->getType();
+        if (!TM || !isa<ScalableVectorType>(Ty))
+          return false;
+        const TargetLowering *TL = TM->getSubtargetImpl(F)->getTargetLowering();
+        unsigned Op = TL->IntrinsicIDToISD(F.getIntrinsicID());
+        assert(Op != ISD::DELETED_NODE && "unsupported intrinsic");
+        EVT VT = EVT::getEVT(Ty);
+        if (!TL->isOperationExpand(Op, VT))
+          return false;
+        // The vector legalizer can expand these to a vector math library call.
+        RTLIB::Libcall LC;
+        switch (Op) {
+        case ISD::FMODF:
+          LC = RTLIB::getMODF(VT);
+          break;
+        case ISD::FSINCOS:
+          LC = RTLIB::getSINCOS(VT);
+          break;
+        default:
+          LC = RTLIB::getSINCOSPI(VT);
+          break;
+        }
+        if (TL->getLibcallImpl(LC) != RTLIB::Unsupported)
+          return false;
+        return lowerUnaryVectorIntrinsicAsLoop(M, CI);
+      });
+      break;
     case Intrinsic::ptrauth_sign:
     case Intrinsic::ptrauth_auth:
       Changed |= expandPtrauthForEmuPAC(F);
diff --git a/llvm/lib/CodeGen/TargetLoweringBase.cpp b/llvm/lib/CodeGen/TargetLoweringBase.cpp
index 58a644094ab8f..cf12687e9115e 100644
--- a/llvm/lib/CodeGen/TargetLoweringBase.cpp
+++ b/llvm/lib/CodeGen/TargetLoweringBase.cpp
@@ -2051,8 +2051,14 @@ int TargetLoweringBase::IntrinsicIDToISD(Intrinsic::ID ID) const {
     return ISD::FLOG2;
   case Intrinsic::log10:
     return ISD::FLOG10;
+  case Intrinsic::modf:
+    return ISD::FMODF;
   case Intrinsic::sin:
     return ISD::FSIN;
+  case Intrinsic::sincos:
+    return ISD::FSINCOS;
+  case Intrinsic::sincospi:
+    return ISD::FSINCOSPI;
   case Intrinsic::sinh:
     return ISD::FSINH;
   case Intrinsic::tan:
diff --git a/llvm/lib/Transforms/Utils/LowerVectorIntrinsics.cpp b/llvm/lib/Transforms/Utils/LowerVectorIntrinsics.cpp
index 1317b873d8db7..6e5c80a129ccd 100644
--- a/llvm/lib/Transforms/Utils/LowerVectorIntrinsics.cpp
+++ b/llvm/lib/Transforms/Utils/LowerVectorIntrinsics.cpp
@@ -15,8 +15,11 @@
 using namespace llvm;
 
 bool llvm::lowerUnaryVectorIntrinsicAsLoop(Module &M, CallInst *CI) {
-  Type *ArgTy = CI->getArgOperand(0)->getType();
-  VectorType *VecTy = cast<VectorType>(ArgTy);
+  Type *RetTy = CI->getType();
+  auto *StructRetTy = dyn_cast<StructType>(RetTy);
+  unsigned NumResults = StructRetTy ? StructRetTy->getNumElements() : 1;
+  auto *VecTy = cast<VectorType>(StructRetTy ? StructRetTy->getElementType(0)
+                                             : RetTy);
 
   BasicBlock *PreLoopBB = CI->getParent();
   BasicBlock *PostLoopBB = nullptr;
@@ -38,18 +41,31 @@ bool llvm::lowerUnaryVectorIntrinsicAsLoop(Module &M, CallInst *CI) {
 
   PHINode *LoopIndex = LoopBuilder.CreatePHI(IdxTy, 2);
   LoopIndex->addIncoming(ConstantInt::get(IdxTy, 0U), PreLoopBB);
-  PHINode *Vec = LoopBuilder.CreatePHI(VecTy, 2);
-  Vec->addIncoming(CI->getArgOperand(0), PreLoopBB);
 
-  Value *Elem = LoopBuilder.CreateExtractElement(Vec, LoopIndex);
+  SmallVector<PHINode *, 2> ResultPhis(NumResults);
+  for (unsigned I = 0; I != NumResults; ++I) {
+    ResultPhis[I] = LoopBuilder.CreatePHI(VecTy, 2);
+    ResultPhis[I]->addIncoming(PoisonValue::get(VecTy), PreLoopBB);
+  }
+
+  Value *Elem =
+      LoopBuilder.CreateExtractElement(CI->getArgOperand(0), LoopIndex);
   Function *Fn = Intrinsic::getOrInsertDeclaration(&M, CI->getIntrinsicID(),
                                                    VecTy->getElementType());
 
   CallInst *ScalarCall = LoopBuilder.CreateCall(Fn, Elem);
   if (isa<FPMathOperator>(CI))
     ScalarCall->copyFastMathFlags(CI);
-  Value *NewVec = LoopBuilder.CreateInsertElement(Vec, ScalarCall, LoopIndex);
-  Vec->addIncoming(NewVec, LoopBB);
+
+  SmallVector<Value *, 2> NewVecs(NumResults);
+  for (unsigned I = 0; I != NumResults; ++I) {
+    Value *ScalarRes = ScalarCall;
+    if (StructRetTy)
+      ScalarRes = LoopBuilder.CreateExtractValue(ScalarCall, I);
+    NewVecs[I] =
+        LoopBuilder.CreateInsertElement(ResultPhis[I], ScalarRes, LoopIndex);
+    ResultPhis[I]->addIncoming(NewVecs[I], LoopBB);
+  }
 
   Value *One = ConstantInt::get(IdxTy, 1U);
   Value *NextLoopIndex = LoopBuilder.CreateAdd(LoopIndex, One);
@@ -59,7 +75,15 @@ bool llvm::lowerUnaryVectorIntrinsicAsLoop(Module &M, CallInst *CI) {
       LoopBuilder.CreateICmp(CmpInst::ICMP_EQ, NextLoopIndex, LoopEnd);
   LoopBuilder.CreateCondBr(ExitCond, PostLoopBB, LoopBB);
 
-  CI->replaceAllUsesWith(NewVec);
+  Value *Res = NewVecs[0];
+  if (StructRetTy) {
+    IRBuilder<> PostLoopBuilder(CI);
+    Res = PoisonValue::get(RetTy);
+    for (unsigned I = 0; I != NumResults; ++I)
+      Res = PostLoopBuilder.CreateInsertValue(Res, NewVecs[I], I);
+  }
+
+  CI->replaceAllUsesWith(Res);
   CI->eraseFromParent();
   return true;
 }
diff --git a/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-exp.ll b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-exp.ll
index d4545a541b43a..75b50fca98d27 100644
--- a/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-exp.ll
+++ b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-exp.ll
@@ -11,8 +11,8 @@ define <vscale x 4 x float> @scalable_vec_exp(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call nnan ninf afn float @llvm.exp.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
diff --git a/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-fp-math.ll b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-fp-math.ll
index 867d655ebf39c..30f58d3458140 100644
--- a/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-fp-math.ll
+++ b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-fp-math.ll
@@ -10,8 +10,8 @@ define <vscale x 4 x float> @scalable_vec_acos(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.acos.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
@@ -32,8 +32,8 @@ define <vscale x 4 x float> @scalable_vec_asin(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.asin.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
@@ -54,8 +54,8 @@ define <vscale x 4 x float> @scalable_vec_atan(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.atan.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
@@ -86,8 +86,8 @@ define <vscale x 4 x float> @scalable_vec_cos(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.cos.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
@@ -108,8 +108,8 @@ define <vscale x 4 x float> @scalable_vec_cosh(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.cosh.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
@@ -130,8 +130,8 @@ define <vscale x 4 x float> @scalable_vec_exp10(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.exp10.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
@@ -152,8 +152,8 @@ define <vscale x 4 x float> @scalable_vec_log2(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.log2.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
@@ -174,8 +174,8 @@ define <vscale x 4 x float> @scalable_vec_log10(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.log10.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
@@ -196,8 +196,8 @@ define <vscale x 4 x float> @scalable_vec_sin(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.sin.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
@@ -218,8 +218,8 @@ define <vscale x 4 x float> @scalable_vec_sinh(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.sinh.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
@@ -240,8 +240,8 @@ define <vscale x 4 x float> @scalable_vec_tan(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.tan.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
@@ -262,8 +262,8 @@ define <vscale x 4 x float> @scalable_vec_tanh(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.tanh.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
diff --git a/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-log.ll b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-log.ll
index 4925011201aee..c0f9021aba751 100644
--- a/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-log.ll
+++ b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-log.ll
@@ -10,8 +10,8 @@ define <vscale x 4 x float> @scalable_vec_log(<vscale x 4 x float> %input) {
 ; CHECK-NEXT:    br label %[[BB3:.*]]
 ; CHECK:       [[BB3]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP9:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ [[INPUT]], [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[TMP5]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 4 x float> [ poison, [[TMP0]] ], [ [[TMP8:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <vscale x 4 x float> [[INPUT]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call float @llvm.log.f32(float [[TMP6]])
 ; CHECK-NEXT:    [[TMP8]] = insertelement <vscale x 4 x float> [[TMP5]], float [[TMP7]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP9]] = add i64 [[TMP4]], 1
diff --git a/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-multi-result.ll b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-multi-result.ll
new file mode 100644
index 0000000000000..c83b609697d9d
--- /dev/null
+++ b/llvm/test/Transforms/PreISelIntrinsicLowering/AArch64/expand-multi-result.ll
@@ -0,0 +1,140 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=pre-isel-intrinsic-lowering -vector-library=sleefgnuabi -S -mattr=+sve < %s | FileCheck %s
+target triple = "aarch64"
+
+; No vector variants exist for these types, so the intrinsics get scalarized.
+define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_sincos_f32(<vscale x 2 x float> %input) {
+; CHECK-LABEL: define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_sincos_f32(
+; CHECK-SAME: <vscale x 2 x float> [[INPUT:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP2:%.*]] = mul nuw i64 [[TMP1]], 2
+; CHECK-NEXT:    br label %[[BB3:.*]]
+; CHECK:       [[BB3]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP13:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP10:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP12:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <vscale x 2 x float> [[INPUT]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP8:%.*]] = call { float, float } @llvm.sincos.f32(float [[TMP7]])
+; CHECK-NEXT:    [[TMP9:%.*]] = extractvalue { float, float } [[TMP8]], 0
+; CHECK-NEXT:    [[TMP10]] = insertelement <vscale x 2 x float> [[TMP5]], float [[TMP9]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP11:%.*]] = extractvalue { float, float } [[TMP8]], 1
+; CHECK-NEXT:    [[TMP12]] = insertelement <vscale x 2 x float> [[TMP6]], float [[TMP11]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP13]] = add i64 [[TMP4]], 1
+; CHECK-NEXT:    [[TMP14:%.*]] = icmp eq i64 [[TMP13]], [[TMP2]]
+; CHECK-NEXT:    br i1 [[TMP14]], label %[[BB15:.*]], label %[[BB3]]
+; CHECK:       [[BB15]]:
+; CHECK-NEXT:    [[TMP16:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } poison, <vscale x 2 x float> [[TMP10]], 0
+; CHECK-NEXT:    [[TMP17:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP16]], <vscale x 2 x float> [[TMP12]], 1
+; CHECK-NEXT:    ret { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP17]]
+;
+  %output = call { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.sincos.nxv2f32(<vscale x 2 x float> %input)
+  ret { <vscale x 2 x float>, <vscale x 2 x float> } %output
+}
+
+define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_sincospi_f32(<vscale x 2 x float> %input) {
+; CHECK-LABEL: define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_sincospi_f32(
+; CHECK-SAME: <vscale x 2 x float> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP2:%.*]] = mul nuw i64 [[TMP1]], 2
+; CHECK-NEXT:    br label %[[BB3:.*]]
+; CHECK:       [[BB3]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP13:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP10:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP12:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <vscale x 2 x float> [[INPUT]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP8:%.*]] = call { float, float } @llvm.sincospi.f32(float [[TMP7]])
+; CHECK-NEXT:    [[TMP9:%.*]] = extractvalue { float, float } [[TMP8]], 0
+; CHECK-NEXT:    [[TMP10]] = insertelement <vscale x 2 x float> [[TMP5]], float [[TMP9]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP11:%.*]] = extractvalue { float, float } [[TMP8]], 1
+; CHECK-NEXT:    [[TMP12]] = insertelement <vscale x 2 x float> [[TMP6]], float [[TMP11]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP13]] = add i64 [[TMP4]], 1
+; CHECK-NEXT:    [[TMP14:%.*]] = icmp eq i64 [[TMP13]], [[TMP2]]
+; CHECK-NEXT:    br i1 [[TMP14]], label %[[BB15:.*]], label %[[BB3]]
+; CHECK:       [[BB15]]:
+; CHECK-NEXT:    [[TMP16:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } poison, <vscale x 2 x float> [[TMP10]], 0
+; CHECK-NEXT:    [[TMP17:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP16]], <vscale x 2 x float> [[TMP12]], 1
+; CHECK-NEXT:    ret { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP17]]
+;
+  %output = call { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.sincospi.nxv2f32(<vscale x 2 x float> %input)
+  ret { <vscale x 2 x float>, <vscale x 2 x float> } %output
+}
+
+define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_modf_f32(<vscale x 2 x float> %input) {
+; CHECK-LABEL: define { <vscale x 2 x float>, <vscale x 2 x float> } @scalable_vec_modf_f32(
+; CHECK-SAME: <vscale x 2 x float> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP2:%.*]] = mul nuw i64 [[TMP1]], 2
+; CHECK-NEXT:    br label %[[BB3:.*]]
+; CHECK:       [[BB3]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = phi i64 [ 0, [[TMP0:%.*]] ], [ [[TMP13:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP5:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP10:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = phi <vscale x 2 x float> [ poison, [[TMP0]] ], [ [[TMP12:%.*]], %[[BB3]] ]
+; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <vscale x 2 x float> [[INPUT]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP8:%.*]] = call { float, float } @llvm.modf.f32(float [[TMP7]])
+; CHECK-NEXT:    [[TMP9:%.*]] = extractvalue { float, float } [[TMP8]], 0
+; CHECK-NEXT:    [[TMP10]] = insertelement <vscale x 2 x float> [[TMP5]], float [[TMP9]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP11:%.*]] = extractvalue { float, float } [[TMP8]], 1
+; CHECK-NEXT:    [[TMP12]] = insertelement <vscale x 2 x float> [[TMP6]], float [[TMP11]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP13]] = add i64 [[TMP4]], 1
+; CHECK-NEXT:    [[TMP14:%.*]] = icmp eq i64 [[TMP13]], [[TMP2]]
+; CHECK-NEXT:    br i1 [[TMP14]], label %[[BB15:.*]], label %[[BB3]]
+; CHECK:       [[BB15]]:
+; CHECK-NEXT:    [[TMP16:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } poison, <vscale x 2 x float> [[TMP10]], 0
+; CHECK-NEXT:    [[TMP17:%.*]] = insertvalue { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP16]], <vscale x 2 x float> [[TMP12]], 1
+; CHECK-NEXT:    ret { <vscale x 2 x float>, <vscale x 2 x float> } [[TMP17]]
+;
+  %output = call { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.modf.nxv2f32(<vscale x 2 x float> %input)
+  ret { <vscale x 2 x float>, <vscale x 2 x float> } %output
+}
+
+; A matching libcall exists, scalarization is skipped.
+define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_sincos_f64(<vscale x 2 x double> %input) {
+; CHECK-LABEL: define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_sincos_f64(
+; CHECK-SAME: <vscale x 2 x double> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[OUTPUT:%.*]] = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincos.nxv2f64(<vscale x 2 x double> [[INPUT]])
+; CHECK-NEXT:    ret { <vscale x 2 x double>, <vscale x 2 x double> } [[OUTPUT]]
+;
+  %output = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincos.nxv2f64(<vscale x 2 x double> %input)
+  ret { <vscale x 2 x double>, <vscale x 2 x double> } %output
+}
+
+define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_sincospi_f64(<vscale x 2 x double> %input) {
+; CHECK-LABEL: define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_sincospi_f64(
+; CHECK-SAME: <vscale x 2 x double> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[OUTPUT:%.*]] = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincospi.nxv2f64(<vscale x 2 x double> [[INPUT]])
+; CHECK-NEXT:    ret { <vscale x 2 x double>, <vscale x 2 x double> } [[OUTPUT]]
+;
+  %output = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincospi.nxv2f64(<vscale x 2 x double> %input)
+  ret { <vscale x 2 x double>, <vscale x 2 x double> } %output
+}
+
+define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_modf_f64(<vscale x 2 x double> %input) {
+; CHECK-LABEL: define { <vscale x 2 x double>, <vscale x 2 x double> } @scalable_vec_modf_f64(
+; CHECK-SAME: <vscale x 2 x double> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[OUTPUT:%.*]] = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.modf.nxv2f64(<vscale x 2 x double> [[INPUT]])
+; CHECK-NEXT:    ret { <vscale x 2 x double>, <vscale x 2 x double> } [[OUTPUT]]
+;
+  %output = call { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.modf.nxv2f64(<vscale x 2 x double> %input)
+  ret { <vscale x 2 x double>, <vscale x 2 x double> } %output
+}
+
+; Fixed-length vectors are left for the vector legalizer to unroll.
+define { <2 x float>, <2 x float> } @fixed_vec_sincos(<2 x float> %input) {
+; CHECK-LABEL: define { <2 x float>, <2 x float> } @fixed_vec_sincos(
+; CHECK-SAME: <2 x float> [[INPUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[OUTPUT:%.*]] = call { <2 x float>, <2 x float> } @llvm.sincos.v2f32(<2 x float> [[INPUT]])
+; CHECK-NEXT:    ret { <2 x float>, <2 x float> } [[OUTPUT]]
+;
+  %output = call { <2 x float>, <2 x float> } @llvm.sincos.v2f32(<2 x float> %input)
+  ret { <2 x float>, <2 x float> } %output
+}
+
+declare { <2 x float>, <2 x float> } @llvm.sincos.v2f32(<2 x float>) #0
+declare { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.sincos.nxv2f32(<vscale x 2 x float>) #0
+declare { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincos.nxv2f64(<vscale x 2 x double>) #0
+declare { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.modf.nxv2f32(<vscale x 2 x float>) #0
+declare { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.modf.nxv2f64(<vscale x 2 x double>) #0
+declare { <vscale x 2 x float>, <vscale x 2 x float> } @llvm.sincospi.nxv2f32(<vscale x 2 x float>) #0
+declare { <vscale x 2 x double>, <vscale x 2 x double> } @llvm.sincospi.nxv2f64(<vscale x 2 x double>) #0
+
+attributes #0 = { nocallback nofree nosync nounwind speculatable willreturn memory(none) }



More information about the llvm-commits mailing list