[llvm] [LoongArch] Remove inaccurate LASX conversion pattern and use [X]VFFINT.S.L instead (PR #207107)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Jul 2 02:29:24 PDT 2026
https://github.com/lrzlin updated https://github.com/llvm/llvm-project/pull/207107
>From 38efbef0ccbaaf1a4cdaeb71d4cca0f2470ff216 Mon Sep 17 00:00:00 2001
From: Lin Runze <linrunze at loongson.cn>
Date: Tue, 30 Jun 2026 16:11:53 +0800
Subject: [PATCH] [LoongArch] Remove inaccurate LASX conversion pattern and use
[X]VFFINT.S.L instead
---
.../LoongArch/LoongArchISelLowering.cpp | 132 ++++++++++++------
.../LoongArch/LoongArchLASXInstrInfo.td | 12 +-
.../Target/LoongArch/LoongArchLSXInstrInfo.td | 9 ++
.../LoongArch/lasx/ir-instruction/sitofp.ll | 5 +-
.../LoongArch/lasx/ir-instruction/uitofp.ll | 22 ++-
.../LoongArch/lsx/ir-instruction/sitofp.ll | 34 ++++-
6 files changed, 150 insertions(+), 64 deletions(-)
diff --git a/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp b/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp
index 9dc6d23711ec1..e944e43e4306d 100644
--- a/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp
+++ b/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp
@@ -464,8 +464,9 @@ LoongArchTargetLowering::LoongArchTargetLowering(const TargetMachine &TM,
for (MVT VT : {MVT::v16i16, MVT::v8i32, MVT::v4i64})
setOperationAction(ISD::BSWAP, VT, Legal);
for (MVT VT : {MVT::v8i32, MVT::v4i32, MVT::v4i64}) {
- setOperationAction({ISD::SINT_TO_FP, ISD::UINT_TO_FP}, VT, Legal);
setOperationAction({ISD::FP_TO_SINT, ISD::FP_TO_UINT}, VT, Legal);
+ setOperationAction(ISD::SINT_TO_FP, VT, Legal);
+ setOperationAction(ISD::UINT_TO_FP, VT, Custom);
}
for (MVT VT : {MVT::v8f32, MVT::v4f64}) {
setOperationAction({ISD::FADD, ISD::FSUB}, VT, Legal);
@@ -4178,6 +4179,12 @@ SDValue LoongArchTargetLowering::lowerUINT_TO_FP(SDValue Op,
EVT VT = Op.getValueType();
EVT Op0VT = Op0.getValueType();
+ if (VT.isVector()) {
+ if (VT.getScalarSizeInBits() != Op0VT.getScalarSizeInBits())
+ return SDValue();
+ return Op;
+ }
+
if ((DAG.SignBitIsZero(Op0) || Op->getFlags().hasNonNeg()) &&
!isOperationLegal(ISD::UINT_TO_FP, Op0VT) &&
isOperationLegal(ISD::SINT_TO_FP, Op0VT))
@@ -8151,15 +8158,91 @@ static SDValue ExtendSrcToDst(SDNode *N, SelectionDAG &DAG, unsigned ExtendOp) {
return DAG.getNode(N->getOpcode(), DL, VT, Extend);
}
+// Merge two 64 to 32 convert instructions into one,
+// e.g.
+// vffint.s.l $vr0, $vr1, $vr2
+// will convert 4 si64 into 4 float at once.
+// or
+// vftintrz.w.d $vr0, $vr1, $vr2
+// which will convert 4 double into 4 si32 at once.
+// also deal with their 256-bits LASX version.
+static SDValue MergeBlocksConvert(SDNode *N, SelectionDAG &DAG, unsigned Opcode,
+ unsigned BlockBits) {
+ SDLoc DL(N);
+ MVT DstVT = N->getSimpleValueType(0);
+ SDValue Src = N->getOperand(0);
+ MVT SrcVT = Src.getSimpleValueType();
+ unsigned SrcBits = SrcVT.getSizeInBits();
+
+ SmallVector<SDValue, 4> Blocks;
+ unsigned BlockNumElts = BlockBits / SrcVT.getScalarSizeInBits();
+ MVT BlockVT = MVT::getVectorVT(SrcVT.getScalarType(), BlockNumElts);
+ if (Src.getOpcode() == ISD::CONCAT_VECTORS &&
+ Src.getOperand(0).getValueType() == BlockVT) {
+ for (unsigned i = 0; i < Src.getNumOperands(); ++i)
+ Blocks.push_back(Src.getOperand(i));
+ } else if (SrcBits > BlockBits) {
+ // Wider than one register: extract each BlockBits-wide sub-vector.
+ for (unsigned i = 0; i < SrcBits / BlockBits; ++i)
+ Blocks.push_back(
+ DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, BlockVT, Src,
+ DAG.getVectorIdxConstant(i * BlockNumElts, DL)));
+ } else {
+ BlockBits = SrcBits;
+ Blocks.push_back(Src);
+ }
+
+ MVT NativeVecVT = MVT::getVectorVT(DstVT.getScalarType(),
+ BlockBits / DstVT.getScalarSizeInBits());
+ SmallVector<SDValue, 4> Parts;
+ for (unsigned i = 0; i < Blocks.size(); i += 2) {
+ SDValue Lo = Blocks[i];
+ SDValue Hi = Blocks.size() > 1 ? Blocks[i + 1] : Lo;
+ SDValue Res = DAG.getNode(Opcode, DL, NativeVecVT, Hi, Lo);
+
+ if (BlockBits == 256) {
+ SDValue Undef = DAG.getUNDEF(NativeVecVT);
+ SmallVector<int, 8> Mask = {0, 1, 4, 5, 2, 3, 6, 7};
+ Res = DAG.getVectorShuffle(NativeVecVT, DL, Res, Undef, Mask);
+ Res = DAG.getBitcast(NativeVecVT, Res);
+ }
+
+ Parts.push_back(Res);
+ }
+
+ if (Blocks.size() == 1)
+ return DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, DstVT, Parts[0],
+ DAG.getVectorIdxConstant(0, DL));
+ return DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Parts);
+}
+
static SDValue performSINT_TO_FPCombine(SDNode *N, SelectionDAG &DAG,
TargetLowering::DAGCombinerInfo &DCI,
const LoongArchSubtarget &Subtarget) {
SDLoc DL(N);
EVT VT = N->getValueType(0);
+ SDValue Src = N->getOperand(0);
+ EVT SrcVT = Src.getValueType();
- // Sign-extend src to avoid scalarization.
- if (VT.isVector())
- return ExtendSrcToDst(N, DAG, ISD::SIGN_EXTEND);
+ if (VT.isVector()) {
+ unsigned SrcEltBits = SrcVT.getScalarSizeInBits();
+ unsigned DstEltBits = VT.getScalarSizeInBits();
+ unsigned NumElts = VT.getVectorNumElements();
+ unsigned BlockBits = Subtarget.hasExtLASX() ? 256 : 128;
+
+ // Sign-extend src to avoid scalarization.
+ if (SrcEltBits <= DstEltBits)
+ return ExtendSrcToDst(N, DAG, ISD::SIGN_EXTEND);
+
+ if (SrcEltBits != 64 || DstEltBits != 32 || !isPowerOf2_32(NumElts))
+ return SDValue();
+
+ if (!SrcVT.isSimple() || !VT.isSimple())
+ return SDValue();
+
+ // Combine [x]vffint.s.l for vector si64 to float conversion.
+ return MergeBlocksConvert(N, DAG, LoongArchISD::VFFINT, BlockBits);
+ }
if (VT != MVT::f32 && VT != MVT::f64)
return SDValue();
@@ -8172,7 +8255,6 @@ static SDValue performSINT_TO_FPCombine(SDNode *N, SelectionDAG &DAG,
if (VT.getSizeInBits() != N->getOperand(0).getValueSizeInBits())
return SDValue();
- SDValue Src = N->getOperand(0);
// If the result of an integer load is only used by an integer-to-float
// conversion, use a fp load instead. This eliminates an integer-to-float-move
// (movgr2fr) instruction.
@@ -8257,45 +8339,7 @@ static SDValue performFP_TO_INTCombine(SDNode *N, SelectionDAG &DAG,
return DAG.getNode(ISD::TRUNCATE, DL, DstVT, Tmp);
}
- SmallVector<SDValue, 8> Blocks;
- unsigned BlockNumElts = BlockBits / 64;
- MVT BlockVT = MVT::getVectorVT(MVT::f64, BlockNumElts);
- if (Src.getOpcode() == ISD::CONCAT_VECTORS &&
- Src.getOperand(0).getValueType() == BlockVT) {
- for (unsigned i = 0; i < Src.getNumOperands(); i++)
- Blocks.push_back(Src.getOperand(i));
- } else if (SrcBits > BlockBits) {
- // Wider than one register: extract each BlockBits-wide sub-vector.
- for (unsigned i = 0; i < SrcBits / BlockBits; i++)
- Blocks.push_back(
- DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, BlockVT, Src,
- DAG.getVectorIdxConstant(i * BlockNumElts, DL)));
- } else {
- BlockBits = SrcBits;
- Blocks.push_back(Src);
- }
-
- MVT NativeVT = BlockBits == 256 ? MVT::v8i32 : MVT::v4i32;
- SmallVector<SDValue, 4> Parts;
- for (unsigned i = 0; i < Blocks.size(); i += 2) {
- SDValue Lo = Blocks[i];
- SDValue Hi = Blocks.size() > 1 ? Blocks[i + 1] : Lo;
- SDValue Res = DAG.getNode(LoongArchISD::VFTINTRZ, DL, NativeVT, Hi, Lo);
-
- if (BlockBits == 256) {
- SDValue Undef = DAG.getUNDEF(Res.getValueType());
- SmallVector<int, 8> Mask = {0, 1, 4, 5, 2, 3, 6, 7};
- Res = DAG.getVectorShuffle(Res.getValueType(), DL, Res, Undef, Mask);
- Res = DAG.getBitcast(NativeVT, Res);
- }
-
- Parts.push_back(Res);
- }
-
- if (Blocks.size() == 1)
- return DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, DstVT, Parts[0],
- DAG.getVectorIdxConstant(0, DL));
- return DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Parts);
+ return MergeBlocksConvert(N, DAG, LoongArchISD::VFTINTRZ, BlockBits);
}
// Try to widen AND, OR and XOR nodes to VT in order to remove casts around
diff --git a/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td b/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td
index 21d937d5b4275..c73b24c3e8b2a 100644
--- a/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td
+++ b/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td
@@ -2126,10 +2126,6 @@ def : Pat<(v4f64 (sint_to_fp v4i64:$vj)), (XVFFINT_D_L v4i64:$vj)>;
def : Pat<(v4f64 (sint_to_fp v4i32:$vj)),
(XVFFINT_D_L (VEXT2XV_D_W (SUBREG_TO_REG v4i32:$vj,
sub_128)))>;
-def : Pat<(v4f32 (sint_to_fp v4i64:$vj)),
- (EXTRACT_SUBREG (XVFCVT_S_D (XVPERMI_D (XVFFINT_D_L v4i64:$vj), 238),
- (XVFFINT_D_L v4i64:$vj)),
- sub_128)>;
// XVFFINT_{S_WU/D_LU}
def : Pat<(v8f32 (uint_to_fp v8i32:$vj)), (XVFFINT_S_WU v8i32:$vj)>;
@@ -2137,10 +2133,10 @@ def : Pat<(v4f64 (uint_to_fp v4i64:$vj)), (XVFFINT_D_LU v4i64:$vj)>;
def : Pat<(v4f64 (uint_to_fp v4i32:$vj)),
(XVFFINT_D_LU (VEXT2XV_DU_WU (SUBREG_TO_REG v4i32:$vj,
sub_128)))>;
-def : Pat<(v4f32 (uint_to_fp v4i64:$vj)),
- (EXTRACT_SUBREG (XVFCVT_S_D (XVPERMI_D (XVFFINT_D_LU v4i64:$vj), 238),
- (XVFFINT_D_LU v4i64:$vj)),
- sub_128)>;
+
+// XVFFINT_S_L
+def : Pat<(v8f32 (loongarch_vffint_s_l (v4i64 LASX256:$xj), (v4i64 LASX256:$xk))),
+ (XVFFINT_S_L LASX256:$xj, LASX256:$xk)>;
// XVFTINTRZ_{W_S/L_D}
def : Pat<(v8i32 (fp_to_sint v8f32:$vj)), (XVFTINTRZ_W_S v8f32:$vj)>;
diff --git a/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td b/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td
index d545976d58712..6a05d26d04168 100644
--- a/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td
+++ b/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td
@@ -40,6 +40,8 @@ def SDT_LoongArchVFCVTLH_D_S : SDTypeProfile<1, 1, [SDTCisVec<0>, SDTCisFP<0>,
SDTCisVec<1>, SDTCisFP<1>]>;
def SDT_LoongArchVFTINTRZ_W_D : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisInt<0>,
SDTCisVec<1>, SDTCisFP<1>, SDTCisSameAs<1, 2>]>;
+def SDT_LoongArchVFFINT_S_L : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisFP<0>,
+ SDTCisVec<1>, SDTCisInt<1>, SDTCisSameAs<1, 2>]>;
// Target nodes.
@@ -123,6 +125,9 @@ def loongarch_vsrar: SDNode<"LoongArchISD::VSRAR", SDT_LoongArchV2R>;
// Vector double-precision convert to 32-bit integer
def loongarch_vftintrz_w_d: SDNode<"LoongArchISD::VFTINTRZ", SDT_LoongArchVFTINTRZ_W_D>;
+// Vector 64-bit integer convert to single-precision
+def loongarch_vffint_s_l: SDNode<"LoongArchISD::VFFINT", SDT_LoongArchVFFINT_S_L>;
+
def immZExt1 : ImmLeaf<GRLenVT, [{return isUInt<1>(Imm);}]>;
def immZExt2 : ImmLeaf<GRLenVT, [{return isUInt<2>(Imm);}]>;
def immZExt3 : ImmLeaf<GRLenVT, [{return isUInt<3>(Imm);}]>;
@@ -2352,6 +2357,10 @@ def : Pat<(v2f64 (sint_to_fp v2i64:$vj)), (VFFINT_D_L v2i64:$vj)>;
def : Pat<(v4f32 (uint_to_fp v4i32:$vj)), (VFFINT_S_WU v4i32:$vj)>;
def : Pat<(v2f64 (uint_to_fp v2i64:$vj)), (VFFINT_D_LU v2i64:$vj)>;
+// VFFINT_S_L
+def : Pat<(v4f32 (loongarch_vffint_s_l (v2i64 LSX128:$vj), (v2i64 LSX128:$vk))),
+ (VFFINT_S_L LSX128:$vj, LSX128:$vk)>;
+
// VFTINTRZ_{W_S/L_D}
def : Pat<(v4i32 (fp_to_sint v4f32:$vj)), (VFTINTRZ_W_S v4f32:$vj)>;
def : Pat<(v2i64 (fp_to_sint v2f64:$vj)), (VFTINTRZ_L_D v2f64:$vj)>;
diff --git a/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/sitofp.ll b/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/sitofp.ll
index 5e3261a22c404..b0cbc6e34192c 100644
--- a/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/sitofp.ll
+++ b/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/sitofp.ll
@@ -32,9 +32,8 @@ define void @sitofp_v4i64_v4f32(ptr %res, ptr %in){
; CHECK-LABEL: sitofp_v4i64_v4f32:
; CHECK: # %bb.0:
; CHECK-NEXT: xvld $xr0, $a1, 0
-; CHECK-NEXT: xvffint.d.l $xr0, $xr0
-; CHECK-NEXT: xvpermi.d $xr1, $xr0, 238
-; CHECK-NEXT: xvfcvt.s.d $xr0, $xr1, $xr0
+; CHECK-NEXT: xvffint.s.l $xr0, $xr0, $xr0
+; CHECK-NEXT: xvpermi.d $xr0, $xr0, 8
; CHECK-NEXT: vst $vr0, $a0, 0
; CHECK-NEXT: ret
%v0 = load <4 x i64>, ptr %in
diff --git a/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/uitofp.ll b/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/uitofp.ll
index 8d5ec3f64ebcf..340e45dbfdb84 100644
--- a/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/uitofp.ll
+++ b/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/uitofp.ll
@@ -32,9 +32,17 @@ define void @uitofp_v4i64_v4f32(ptr %res, ptr %in){
; CHECK-LABEL: uitofp_v4i64_v4f32:
; CHECK: # %bb.0:
; CHECK-NEXT: xvld $xr0, $a1, 0
-; CHECK-NEXT: xvffint.d.lu $xr0, $xr0
-; CHECK-NEXT: xvpermi.d $xr1, $xr0, 238
-; CHECK-NEXT: xvfcvt.s.d $xr0, $xr1, $xr0
+; CHECK-NEXT: lu12i.w $a1, 325632
+; CHECK-NEXT: vreplgr2vr.w $vr1, $a1
+; CHECK-NEXT: xvsrli.d $xr2, $xr0, 32
+; CHECK-NEXT: xvffint.s.l $xr2, $xr2, $xr2
+; CHECK-NEXT: xvpermi.d $xr2, $xr2, 216
+; CHECK-NEXT: vfmul.s $vr1, $vr2, $vr1
+; CHECK-NEXT: xvldi $xr2, -1777
+; CHECK-NEXT: xvand.v $xr0, $xr0, $xr2
+; CHECK-NEXT: xvffint.s.l $xr0, $xr0, $xr0
+; CHECK-NEXT: xvpermi.d $xr0, $xr0, 216
+; CHECK-NEXT: vfadd.s $vr0, $vr1, $vr0
; CHECK-NEXT: vst $vr0, $a0, 0
; CHECK-NEXT: ret
%v0 = load <4 x i64>, ptr %in
@@ -48,7 +56,7 @@ define void @uitofp_v4i32_v4f64(ptr %res, ptr %in){
; CHECK: # %bb.0:
; CHECK-NEXT: vld $vr0, $a1, 0
; CHECK-NEXT: vext2xv.du.wu $xr0, $xr0
-; CHECK-NEXT: xvffint.d.lu $xr0, $xr0
+; CHECK-NEXT: xvffint.d.l $xr0, $xr0
; CHECK-NEXT: xvst $xr0, $a0, 0
; CHECK-NEXT: ret
%v0 = load <4 x i32>, ptr %in
@@ -76,7 +84,7 @@ define <2 x double> @uitofp_v16i8_v2f64(<16 x i8> %a) {
; CHECK-NEXT: vext2xv.hu.bu $xr0, $xr0
; CHECK-NEXT: vext2xv.wu.hu $xr0, $xr0
; CHECK-NEXT: vext2xv.du.wu $xr0, $xr0
-; CHECK-NEXT: xvffint.d.lu $xr0, $xr0
+; CHECK-NEXT: xvffint.d.l $xr0, $xr0
; CHECK-NEXT: # kill: def $vr0 killed $vr0 killed $xr0
; CHECK-NEXT: ret
%cvt = uitofp <16 x i8> %a to <16 x double>
@@ -89,7 +97,7 @@ define <4 x double> @uitofp_v4i8_v4f64(<16 x i8> %a) {
; CHECK: # %bb.0:
; CHECK-NEXT: # kill: def $vr0 killed $vr0 def $xr0
; CHECK-NEXT: vext2xv.du.bu $xr0, $xr0
-; CHECK-NEXT: xvffint.d.lu $xr0, $xr0
+; CHECK-NEXT: xvffint.d.l $xr0, $xr0
; CHECK-NEXT: ret
%shuf = shufflevector <16 x i8> %a, <16 x i8> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
%cvt = uitofp <4 x i8> %shuf to <4 x double>
@@ -103,7 +111,7 @@ define <4 x double> @uitofp_v16i8_v4f64(<16 x i8> %a) {
; CHECK-NEXT: vext2xv.hu.bu $xr0, $xr0
; CHECK-NEXT: vext2xv.wu.hu $xr0, $xr0
; CHECK-NEXT: vext2xv.du.wu $xr0, $xr0
-; CHECK-NEXT: xvffint.d.lu $xr0, $xr0
+; CHECK-NEXT: xvffint.d.l $xr0, $xr0
; CHECK-NEXT: ret
%cvt = uitofp <16 x i8> %a to <16 x double>
%shuf = shufflevector <16 x double> %cvt, <16 x double> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
diff --git a/llvm/test/CodeGen/LoongArch/lsx/ir-instruction/sitofp.ll b/llvm/test/CodeGen/LoongArch/lsx/ir-instruction/sitofp.ll
index 18a3bcd70b465..a3b5a17622114 100644
--- a/llvm/test/CodeGen/LoongArch/lsx/ir-instruction/sitofp.ll
+++ b/llvm/test/CodeGen/LoongArch/lsx/ir-instruction/sitofp.ll
@@ -1,6 +1,6 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 3
-; RUN: llc --mtriple=loongarch32 --mattr=+32s,+lsx < %s | FileCheck %s
-; RUN: llc --mtriple=loongarch64 --mattr=+lsx < %s | FileCheck %s
+; RUN: llc --mtriple=loongarch32 --mattr=+32s,+lsx < %s | FileCheck %s --check-prefixes=CHECK,LA32
+; RUN: llc --mtriple=loongarch64 --mattr=+lsx < %s | FileCheck %s --check-prefixes=CHECK,LA64
define void @sitofp_v4i32_v4f32(ptr %res, ptr %in){
; CHECK-LABEL: sitofp_v4i32_v4f32:
@@ -28,6 +28,27 @@ define void @sitofp_v2i64_v2f64(ptr %res, ptr %in){
ret void
}
+define void @sitofp_v2i64_v2f32(ptr %res, ptr %in){
+; LA32-LABEL: sitofp_v2i64_v2f32:
+; LA32: # %bb.0:
+; LA32-NEXT: vld $vr0, $a1, 0
+; LA32-NEXT: vffint.s.l $vr0, $vr0, $vr0
+; LA32-NEXT: vstelm.w $vr0, $a0, 4, 1
+; LA32-NEXT: vstelm.w $vr0, $a0, 0, 0
+; LA32-NEXT: ret
+;
+; LA64-LABEL: sitofp_v2i64_v2f32:
+; LA64: # %bb.0:
+; LA64-NEXT: vld $vr0, $a1, 0
+; LA64-NEXT: vffint.s.l $vr0, $vr0, $vr0
+; LA64-NEXT: vstelm.d $vr0, $a0, 0, 0
+; LA64-NEXT: ret
+ %v0 = load <2 x i64>, ptr %in
+ %v1 = sitofp <2 x i64> %v0 to <2 x float>
+ store <2 x float> %v1, ptr %res
+ ret void
+}
+
define <4 x double> @sitofp_v4i64_v4f64(<4 x i64> %a) {
; CHECK-LABEL: sitofp_v4i64_v4f64:
; CHECK: # %bb.0:
@@ -38,6 +59,15 @@ define <4 x double> @sitofp_v4i64_v4f64(<4 x i64> %a) {
ret <4 x double> %cvt
}
+define <4 x float> @sitofp_v4i64_v4f32(<4 x i64> %a) {
+; CHECK-LABEL: sitofp_v4i64_v4f32:
+; CHECK: # %bb.0:
+; CHECK-NEXT: vffint.s.l $vr0, $vr1, $vr0
+; CHECK-NEXT: ret
+ %cvt = sitofp <4 x i64> %a to <4 x float>
+ ret <4 x float> %cvt
+}
+
define <4 x double> @sitofp_v4i32_v4f64(<4 x i32> %a) {
; CHECK-LABEL: sitofp_v4i32_v4f64:
; CHECK: # %bb.0:
More information about the llvm-commits
mailing list