[llvm] [LoongArch] Remove inaccurate LASX conversion pattern and use [X]VFFINT.S.L instead (PR #207107)

via llvm-commits llvm-commits at lists.llvm.org
Thu Jul 2 02:29:24 PDT 2026


https://github.com/lrzlin updated https://github.com/llvm/llvm-project/pull/207107

>From 38efbef0ccbaaf1a4cdaeb71d4cca0f2470ff216 Mon Sep 17 00:00:00 2001
From: Lin Runze <linrunze at loongson.cn>
Date: Tue, 30 Jun 2026 16:11:53 +0800
Subject: [PATCH] [LoongArch] Remove inaccurate LASX conversion pattern and use
 [X]VFFINT.S.L instead

---
 .../LoongArch/LoongArchISelLowering.cpp       | 132 ++++++++++++------
 .../LoongArch/LoongArchLASXInstrInfo.td       |  12 +-
 .../Target/LoongArch/LoongArchLSXInstrInfo.td |   9 ++
 .../LoongArch/lasx/ir-instruction/sitofp.ll   |   5 +-
 .../LoongArch/lasx/ir-instruction/uitofp.ll   |  22 ++-
 .../LoongArch/lsx/ir-instruction/sitofp.ll    |  34 ++++-
 6 files changed, 150 insertions(+), 64 deletions(-)

diff --git a/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp b/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp
index 9dc6d23711ec1..e944e43e4306d 100644
--- a/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp
+++ b/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp
@@ -464,8 +464,9 @@ LoongArchTargetLowering::LoongArchTargetLowering(const TargetMachine &TM,
     for (MVT VT : {MVT::v16i16, MVT::v8i32, MVT::v4i64})
       setOperationAction(ISD::BSWAP, VT, Legal);
     for (MVT VT : {MVT::v8i32, MVT::v4i32, MVT::v4i64}) {
-      setOperationAction({ISD::SINT_TO_FP, ISD::UINT_TO_FP}, VT, Legal);
       setOperationAction({ISD::FP_TO_SINT, ISD::FP_TO_UINT}, VT, Legal);
+      setOperationAction(ISD::SINT_TO_FP, VT, Legal);
+      setOperationAction(ISD::UINT_TO_FP, VT, Custom);
     }
     for (MVT VT : {MVT::v8f32, MVT::v4f64}) {
       setOperationAction({ISD::FADD, ISD::FSUB}, VT, Legal);
@@ -4178,6 +4179,12 @@ SDValue LoongArchTargetLowering::lowerUINT_TO_FP(SDValue Op,
   EVT VT = Op.getValueType();
   EVT Op0VT = Op0.getValueType();
 
+  if (VT.isVector()) {
+    if (VT.getScalarSizeInBits() != Op0VT.getScalarSizeInBits())
+      return SDValue();
+    return Op;
+  }
+
   if ((DAG.SignBitIsZero(Op0) || Op->getFlags().hasNonNeg()) &&
       !isOperationLegal(ISD::UINT_TO_FP, Op0VT) &&
       isOperationLegal(ISD::SINT_TO_FP, Op0VT))
@@ -8151,15 +8158,91 @@ static SDValue ExtendSrcToDst(SDNode *N, SelectionDAG &DAG, unsigned ExtendOp) {
   return DAG.getNode(N->getOpcode(), DL, VT, Extend);
 }
 
+// Merge two 64 to 32 convert instructions into one,
+// e.g.
+//   vffint.s.l $vr0, $vr1, $vr2
+// will convert 4 si64 into 4 float at once.
+// or
+//   vftintrz.w.d $vr0, $vr1, $vr2
+// which will convert 4 double into 4 si32 at once.
+// also deal with their 256-bits LASX version.
+static SDValue MergeBlocksConvert(SDNode *N, SelectionDAG &DAG, unsigned Opcode,
+                                  unsigned BlockBits) {
+  SDLoc DL(N);
+  MVT DstVT = N->getSimpleValueType(0);
+  SDValue Src = N->getOperand(0);
+  MVT SrcVT = Src.getSimpleValueType();
+  unsigned SrcBits = SrcVT.getSizeInBits();
+
+  SmallVector<SDValue, 4> Blocks;
+  unsigned BlockNumElts = BlockBits / SrcVT.getScalarSizeInBits();
+  MVT BlockVT = MVT::getVectorVT(SrcVT.getScalarType(), BlockNumElts);
+  if (Src.getOpcode() == ISD::CONCAT_VECTORS &&
+      Src.getOperand(0).getValueType() == BlockVT) {
+    for (unsigned i = 0; i < Src.getNumOperands(); ++i)
+      Blocks.push_back(Src.getOperand(i));
+  } else if (SrcBits > BlockBits) {
+    // Wider than one register: extract each BlockBits-wide sub-vector.
+    for (unsigned i = 0; i < SrcBits / BlockBits; ++i)
+      Blocks.push_back(
+          DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, BlockVT, Src,
+                      DAG.getVectorIdxConstant(i * BlockNumElts, DL)));
+  } else {
+    BlockBits = SrcBits;
+    Blocks.push_back(Src);
+  }
+
+  MVT NativeVecVT = MVT::getVectorVT(DstVT.getScalarType(),
+                                  BlockBits / DstVT.getScalarSizeInBits());
+  SmallVector<SDValue, 4> Parts;
+  for (unsigned i = 0; i < Blocks.size(); i += 2) {
+    SDValue Lo = Blocks[i];
+    SDValue Hi = Blocks.size() > 1 ? Blocks[i + 1] : Lo;
+    SDValue Res = DAG.getNode(Opcode, DL, NativeVecVT, Hi, Lo);
+
+    if (BlockBits == 256) {
+      SDValue Undef = DAG.getUNDEF(NativeVecVT);
+      SmallVector<int, 8> Mask = {0, 1, 4, 5, 2, 3, 6, 7};
+      Res = DAG.getVectorShuffle(NativeVecVT, DL, Res, Undef, Mask);
+      Res = DAG.getBitcast(NativeVecVT, Res);
+    }
+
+    Parts.push_back(Res);
+  }
+
+  if (Blocks.size() == 1)
+    return DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, DstVT, Parts[0],
+                       DAG.getVectorIdxConstant(0, DL));
+  return DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Parts);
+}
+
 static SDValue performSINT_TO_FPCombine(SDNode *N, SelectionDAG &DAG,
                                         TargetLowering::DAGCombinerInfo &DCI,
                                         const LoongArchSubtarget &Subtarget) {
   SDLoc DL(N);
   EVT VT = N->getValueType(0);
+  SDValue Src = N->getOperand(0);
+  EVT SrcVT = Src.getValueType();
 
-  // Sign-extend src to avoid scalarization.
-  if (VT.isVector())
-    return ExtendSrcToDst(N, DAG, ISD::SIGN_EXTEND);
+  if (VT.isVector()) {
+    unsigned SrcEltBits = SrcVT.getScalarSizeInBits();
+    unsigned DstEltBits = VT.getScalarSizeInBits();
+    unsigned NumElts = VT.getVectorNumElements();
+    unsigned BlockBits = Subtarget.hasExtLASX() ? 256 : 128;
+
+    // Sign-extend src to avoid scalarization.
+    if (SrcEltBits <= DstEltBits)
+      return ExtendSrcToDst(N, DAG, ISD::SIGN_EXTEND);
+
+    if (SrcEltBits != 64 || DstEltBits != 32 || !isPowerOf2_32(NumElts))
+      return SDValue();
+
+    if (!SrcVT.isSimple() || !VT.isSimple())
+      return SDValue();
+
+    // Combine [x]vffint.s.l for vector si64 to float conversion.
+    return MergeBlocksConvert(N, DAG, LoongArchISD::VFFINT, BlockBits);
+  }
 
   if (VT != MVT::f32 && VT != MVT::f64)
     return SDValue();
@@ -8172,7 +8255,6 @@ static SDValue performSINT_TO_FPCombine(SDNode *N, SelectionDAG &DAG,
   if (VT.getSizeInBits() != N->getOperand(0).getValueSizeInBits())
     return SDValue();
 
-  SDValue Src = N->getOperand(0);
   // If the result of an integer load is only used by an integer-to-float
   // conversion, use a fp load instead. This eliminates an integer-to-float-move
   // (movgr2fr) instruction.
@@ -8257,45 +8339,7 @@ static SDValue performFP_TO_INTCombine(SDNode *N, SelectionDAG &DAG,
     return DAG.getNode(ISD::TRUNCATE, DL, DstVT, Tmp);
   }
 
-  SmallVector<SDValue, 8> Blocks;
-  unsigned BlockNumElts = BlockBits / 64;
-  MVT BlockVT = MVT::getVectorVT(MVT::f64, BlockNumElts);
-  if (Src.getOpcode() == ISD::CONCAT_VECTORS &&
-      Src.getOperand(0).getValueType() == BlockVT) {
-    for (unsigned i = 0; i < Src.getNumOperands(); i++)
-      Blocks.push_back(Src.getOperand(i));
-  } else if (SrcBits > BlockBits) {
-    // Wider than one register: extract each BlockBits-wide sub-vector.
-    for (unsigned i = 0; i < SrcBits / BlockBits; i++)
-      Blocks.push_back(
-          DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, BlockVT, Src,
-                      DAG.getVectorIdxConstant(i * BlockNumElts, DL)));
-  } else {
-    BlockBits = SrcBits;
-    Blocks.push_back(Src);
-  }
-
-  MVT NativeVT = BlockBits == 256 ? MVT::v8i32 : MVT::v4i32;
-  SmallVector<SDValue, 4> Parts;
-  for (unsigned i = 0; i < Blocks.size(); i += 2) {
-    SDValue Lo = Blocks[i];
-    SDValue Hi = Blocks.size() > 1 ? Blocks[i + 1] : Lo;
-    SDValue Res = DAG.getNode(LoongArchISD::VFTINTRZ, DL, NativeVT, Hi, Lo);
-
-    if (BlockBits == 256) {
-      SDValue Undef = DAG.getUNDEF(Res.getValueType());
-      SmallVector<int, 8> Mask = {0, 1, 4, 5, 2, 3, 6, 7};
-      Res = DAG.getVectorShuffle(Res.getValueType(), DL, Res, Undef, Mask);
-      Res = DAG.getBitcast(NativeVT, Res);
-    }
-
-    Parts.push_back(Res);
-  }
-
-  if (Blocks.size() == 1)
-    return DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, DstVT, Parts[0],
-                       DAG.getVectorIdxConstant(0, DL));
-  return DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Parts);
+  return MergeBlocksConvert(N, DAG, LoongArchISD::VFTINTRZ, BlockBits);
 }
 
 // Try to widen AND, OR and XOR nodes to VT in order to remove casts around
diff --git a/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td b/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td
index 21d937d5b4275..c73b24c3e8b2a 100644
--- a/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td
+++ b/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td
@@ -2126,10 +2126,6 @@ def : Pat<(v4f64 (sint_to_fp v4i64:$vj)), (XVFFINT_D_L v4i64:$vj)>;
 def : Pat<(v4f64 (sint_to_fp v4i32:$vj)),
           (XVFFINT_D_L (VEXT2XV_D_W (SUBREG_TO_REG v4i32:$vj,
                                                    sub_128)))>;
-def : Pat<(v4f32 (sint_to_fp v4i64:$vj)),
-          (EXTRACT_SUBREG (XVFCVT_S_D (XVPERMI_D (XVFFINT_D_L v4i64:$vj), 238),
-                                      (XVFFINT_D_L v4i64:$vj)),
-                          sub_128)>;
 
 // XVFFINT_{S_WU/D_LU}
 def : Pat<(v8f32 (uint_to_fp v8i32:$vj)), (XVFFINT_S_WU v8i32:$vj)>;
@@ -2137,10 +2133,10 @@ def : Pat<(v4f64 (uint_to_fp v4i64:$vj)), (XVFFINT_D_LU v4i64:$vj)>;
 def : Pat<(v4f64 (uint_to_fp v4i32:$vj)),
           (XVFFINT_D_LU (VEXT2XV_DU_WU (SUBREG_TO_REG v4i32:$vj,
                                                       sub_128)))>;
-def : Pat<(v4f32 (uint_to_fp v4i64:$vj)),
-          (EXTRACT_SUBREG (XVFCVT_S_D (XVPERMI_D (XVFFINT_D_LU v4i64:$vj), 238),
-                                       (XVFFINT_D_LU v4i64:$vj)),
-                          sub_128)>;
+
+// XVFFINT_S_L
+def : Pat<(v8f32 (loongarch_vffint_s_l (v4i64 LASX256:$xj), (v4i64 LASX256:$xk))),
+          (XVFFINT_S_L LASX256:$xj, LASX256:$xk)>;
 
 // XVFTINTRZ_{W_S/L_D}
 def : Pat<(v8i32 (fp_to_sint v8f32:$vj)), (XVFTINTRZ_W_S v8f32:$vj)>;
diff --git a/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td b/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td
index d545976d58712..6a05d26d04168 100644
--- a/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td
+++ b/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td
@@ -40,6 +40,8 @@ def SDT_LoongArchVFCVTLH_D_S : SDTypeProfile<1, 1, [SDTCisVec<0>, SDTCisFP<0>,
                                                     SDTCisVec<1>, SDTCisFP<1>]>;
 def SDT_LoongArchVFTINTRZ_W_D : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisInt<0>,
                                                      SDTCisVec<1>, SDTCisFP<1>, SDTCisSameAs<1, 2>]>;
+def SDT_LoongArchVFFINT_S_L : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisFP<0>,
+                                                  SDTCisVec<1>, SDTCisInt<1>, SDTCisSameAs<1, 2>]>;
 
 // Target nodes.
 
@@ -123,6 +125,9 @@ def loongarch_vsrar: SDNode<"LoongArchISD::VSRAR", SDT_LoongArchV2R>;
 // Vector double-precision convert to 32-bit integer
 def loongarch_vftintrz_w_d: SDNode<"LoongArchISD::VFTINTRZ", SDT_LoongArchVFTINTRZ_W_D>;
 
+// Vector 64-bit integer convert to single-precision
+def loongarch_vffint_s_l: SDNode<"LoongArchISD::VFFINT", SDT_LoongArchVFFINT_S_L>;
+
 def immZExt1 : ImmLeaf<GRLenVT, [{return isUInt<1>(Imm);}]>;
 def immZExt2 : ImmLeaf<GRLenVT, [{return isUInt<2>(Imm);}]>;
 def immZExt3 : ImmLeaf<GRLenVT, [{return isUInt<3>(Imm);}]>;
@@ -2352,6 +2357,10 @@ def : Pat<(v2f64 (sint_to_fp v2i64:$vj)), (VFFINT_D_L v2i64:$vj)>;
 def : Pat<(v4f32 (uint_to_fp v4i32:$vj)), (VFFINT_S_WU v4i32:$vj)>;
 def : Pat<(v2f64 (uint_to_fp v2i64:$vj)), (VFFINT_D_LU v2i64:$vj)>;
 
+// VFFINT_S_L
+def : Pat<(v4f32 (loongarch_vffint_s_l (v2i64 LSX128:$vj), (v2i64 LSX128:$vk))),
+          (VFFINT_S_L LSX128:$vj, LSX128:$vk)>;
+
 // VFTINTRZ_{W_S/L_D}
 def : Pat<(v4i32 (fp_to_sint v4f32:$vj)), (VFTINTRZ_W_S v4f32:$vj)>;
 def : Pat<(v2i64 (fp_to_sint v2f64:$vj)), (VFTINTRZ_L_D v2f64:$vj)>;
diff --git a/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/sitofp.ll b/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/sitofp.ll
index 5e3261a22c404..b0cbc6e34192c 100644
--- a/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/sitofp.ll
+++ b/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/sitofp.ll
@@ -32,9 +32,8 @@ define void @sitofp_v4i64_v4f32(ptr %res, ptr %in){
 ; CHECK-LABEL: sitofp_v4i64_v4f32:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    xvld $xr0, $a1, 0
-; CHECK-NEXT:    xvffint.d.l $xr0, $xr0
-; CHECK-NEXT:    xvpermi.d $xr1, $xr0, 238
-; CHECK-NEXT:    xvfcvt.s.d $xr0, $xr1, $xr0
+; CHECK-NEXT:    xvffint.s.l $xr0, $xr0, $xr0
+; CHECK-NEXT:    xvpermi.d $xr0, $xr0, 8
 ; CHECK-NEXT:    vst $vr0, $a0, 0
 ; CHECK-NEXT:    ret
   %v0 = load <4 x i64>, ptr %in
diff --git a/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/uitofp.ll b/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/uitofp.ll
index 8d5ec3f64ebcf..340e45dbfdb84 100644
--- a/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/uitofp.ll
+++ b/llvm/test/CodeGen/LoongArch/lasx/ir-instruction/uitofp.ll
@@ -32,9 +32,17 @@ define void @uitofp_v4i64_v4f32(ptr %res, ptr %in){
 ; CHECK-LABEL: uitofp_v4i64_v4f32:
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    xvld $xr0, $a1, 0
-; CHECK-NEXT:    xvffint.d.lu $xr0, $xr0
-; CHECK-NEXT:    xvpermi.d $xr1, $xr0, 238
-; CHECK-NEXT:    xvfcvt.s.d $xr0, $xr1, $xr0
+; CHECK-NEXT:    lu12i.w $a1, 325632
+; CHECK-NEXT:    vreplgr2vr.w $vr1, $a1
+; CHECK-NEXT:    xvsrli.d $xr2, $xr0, 32
+; CHECK-NEXT:    xvffint.s.l $xr2, $xr2, $xr2
+; CHECK-NEXT:    xvpermi.d $xr2, $xr2, 216
+; CHECK-NEXT:    vfmul.s $vr1, $vr2, $vr1
+; CHECK-NEXT:    xvldi $xr2, -1777
+; CHECK-NEXT:    xvand.v $xr0, $xr0, $xr2
+; CHECK-NEXT:    xvffint.s.l $xr0, $xr0, $xr0
+; CHECK-NEXT:    xvpermi.d $xr0, $xr0, 216
+; CHECK-NEXT:    vfadd.s $vr0, $vr1, $vr0
 ; CHECK-NEXT:    vst $vr0, $a0, 0
 ; CHECK-NEXT:    ret
   %v0 = load <4 x i64>, ptr %in
@@ -48,7 +56,7 @@ define void @uitofp_v4i32_v4f64(ptr %res, ptr %in){
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    vld $vr0, $a1, 0
 ; CHECK-NEXT:    vext2xv.du.wu $xr0, $xr0
-; CHECK-NEXT:    xvffint.d.lu $xr0, $xr0
+; CHECK-NEXT:    xvffint.d.l $xr0, $xr0
 ; CHECK-NEXT:    xvst $xr0, $a0, 0
 ; CHECK-NEXT:    ret
   %v0 = load <4 x i32>, ptr %in
@@ -76,7 +84,7 @@ define <2 x double> @uitofp_v16i8_v2f64(<16 x i8> %a) {
 ; CHECK-NEXT:    vext2xv.hu.bu $xr0, $xr0
 ; CHECK-NEXT:    vext2xv.wu.hu $xr0, $xr0
 ; CHECK-NEXT:    vext2xv.du.wu $xr0, $xr0
-; CHECK-NEXT:    xvffint.d.lu $xr0, $xr0
+; CHECK-NEXT:    xvffint.d.l $xr0, $xr0
 ; CHECK-NEXT:    # kill: def $vr0 killed $vr0 killed $xr0
 ; CHECK-NEXT:    ret
   %cvt = uitofp <16 x i8> %a to <16 x double>
@@ -89,7 +97,7 @@ define <4 x double> @uitofp_v4i8_v4f64(<16 x i8> %a) {
 ; CHECK:       # %bb.0:
 ; CHECK-NEXT:    # kill: def $vr0 killed $vr0 def $xr0
 ; CHECK-NEXT:    vext2xv.du.bu $xr0, $xr0
-; CHECK-NEXT:    xvffint.d.lu $xr0, $xr0
+; CHECK-NEXT:    xvffint.d.l $xr0, $xr0
 ; CHECK-NEXT:    ret
   %shuf = shufflevector <16 x i8> %a, <16 x i8> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
   %cvt = uitofp <4 x i8> %shuf to <4 x double>
@@ -103,7 +111,7 @@ define <4 x double> @uitofp_v16i8_v4f64(<16 x i8> %a) {
 ; CHECK-NEXT:    vext2xv.hu.bu $xr0, $xr0
 ; CHECK-NEXT:    vext2xv.wu.hu $xr0, $xr0
 ; CHECK-NEXT:    vext2xv.du.wu $xr0, $xr0
-; CHECK-NEXT:    xvffint.d.lu $xr0, $xr0
+; CHECK-NEXT:    xvffint.d.l $xr0, $xr0
 ; CHECK-NEXT:    ret
   %cvt = uitofp <16 x i8> %a to <16 x double>
   %shuf = shufflevector <16 x double> %cvt, <16 x double> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
diff --git a/llvm/test/CodeGen/LoongArch/lsx/ir-instruction/sitofp.ll b/llvm/test/CodeGen/LoongArch/lsx/ir-instruction/sitofp.ll
index 18a3bcd70b465..a3b5a17622114 100644
--- a/llvm/test/CodeGen/LoongArch/lsx/ir-instruction/sitofp.ll
+++ b/llvm/test/CodeGen/LoongArch/lsx/ir-instruction/sitofp.ll
@@ -1,6 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 3
-; RUN: llc --mtriple=loongarch32 --mattr=+32s,+lsx < %s | FileCheck %s
-; RUN: llc --mtriple=loongarch64 --mattr=+lsx < %s | FileCheck %s
+; RUN: llc --mtriple=loongarch32 --mattr=+32s,+lsx < %s | FileCheck %s --check-prefixes=CHECK,LA32
+; RUN: llc --mtriple=loongarch64 --mattr=+lsx < %s | FileCheck %s --check-prefixes=CHECK,LA64
 
 define void @sitofp_v4i32_v4f32(ptr %res, ptr %in){
 ; CHECK-LABEL: sitofp_v4i32_v4f32:
@@ -28,6 +28,27 @@ define void @sitofp_v2i64_v2f64(ptr %res, ptr %in){
   ret void
 }
 
+define void @sitofp_v2i64_v2f32(ptr %res, ptr %in){
+; LA32-LABEL: sitofp_v2i64_v2f32:
+; LA32:       # %bb.0:
+; LA32-NEXT:    vld $vr0, $a1, 0
+; LA32-NEXT:    vffint.s.l $vr0, $vr0, $vr0
+; LA32-NEXT:    vstelm.w $vr0, $a0, 4, 1
+; LA32-NEXT:    vstelm.w $vr0, $a0, 0, 0
+; LA32-NEXT:    ret
+;
+; LA64-LABEL: sitofp_v2i64_v2f32:
+; LA64:       # %bb.0:
+; LA64-NEXT:    vld $vr0, $a1, 0
+; LA64-NEXT:    vffint.s.l $vr0, $vr0, $vr0
+; LA64-NEXT:    vstelm.d $vr0, $a0, 0, 0
+; LA64-NEXT:    ret
+  %v0 = load <2 x i64>, ptr %in
+  %v1 = sitofp <2 x i64> %v0 to <2 x float>
+  store <2 x float> %v1, ptr %res
+  ret void
+}
+
 define <4 x double> @sitofp_v4i64_v4f64(<4 x i64> %a) {
 ; CHECK-LABEL: sitofp_v4i64_v4f64:
 ; CHECK:       # %bb.0:
@@ -38,6 +59,15 @@ define <4 x double> @sitofp_v4i64_v4f64(<4 x i64> %a) {
   ret <4 x double> %cvt
 }
 
+define <4 x float> @sitofp_v4i64_v4f32(<4 x i64> %a) {
+; CHECK-LABEL: sitofp_v4i64_v4f32:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vffint.s.l $vr0, $vr1, $vr0
+; CHECK-NEXT:    ret
+  %cvt = sitofp <4 x i64> %a to <4 x float>
+  ret <4 x float> %cvt
+}
+
 define <4 x double> @sitofp_v4i32_v4f64(<4 x i32> %a) {
 ; CHECK-LABEL: sitofp_v4i32_v4f64:
 ; CHECK:       # %bb.0:



More information about the llvm-commits mailing list