[llvm] [SelectionDAG] Soft promote the element operand of INSERT_VECTOR_ELT (PR #223612)
Paweł Bylica via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 15 00:14:25 PDT 2026
https://github.com/chfast created https://github.com/llvm/llvm-project/pull/223612
When f16/bf16 is soft promoted but a vector of that type is legal,
INSERT_VECTOR_ELT is left with a soft promoted i16 element operand.
SoftPromoteHalfOperand had no case for it and failed with "Do not know
how to soft promote this operator's operand!".
Legalize it like BUILD_VECTOR and the inverse EXTRACT_VECTOR_ELT: bitcast
the vector to its integer counterpart, insert the promoted element there
and bitcast the result back. This also covers scalable vectors.
This is reachable on MIPS MSA and on RISC-V with Zvfhmin/Zvfbfmin but
without Zfhmin/Zfbfmin. X86, Hexagon and WebAssembly avoid it by custom
lowering INSERT_VECTOR_ELT on the scalar type; that still takes
precedence and can be removed separately.
Fixes https://github.com/llvm/llvm-project/issues/198104.
Assisted-by: Claude Code
>From 2418733c342446449c0e2840d998ccba105c8c14 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Pawe=C5=82=20Bylica?= <pawel at hepcolgum.band>
Date: Tue, 15 Sep 2026 01:29:36 +0200
Subject: [PATCH 1/2] [SelectionDAG][test] Add tests for INSERT_VECTOR_ELT with
a soft promoted half
Pre-commit tests demonstrating the crash. When f16/bf16 is soft promoted
but a vector of that type is legal, INSERT_VECTOR_ELT is left with a soft
promoted element operand and SoftPromoteHalfOperand has no case for it:
LLVM ERROR: Do not know how to soft promote this operator's operand!
Reachable on MIPS MSA (fixed vectors) and on RISC-V with Zvfhmin/Zvfbfmin
but without Zfhmin/Zfbfmin (scalable vectors and bf16).
Assisted-by: Claude Code
---
.../CodeGen/Mips/msa/f16vec-insertelement.ll | 33 +++++++++++++
.../RISCV/rvv/insertelt-fp-soft-promote.ll | 49 +++++++++++++++++++
2 files changed, 82 insertions(+)
create mode 100644 llvm/test/CodeGen/Mips/msa/f16vec-insertelement.ll
create mode 100644 llvm/test/CodeGen/RISCV/rvv/insertelt-fp-soft-promote.ll
diff --git a/llvm/test/CodeGen/Mips/msa/f16vec-insertelement.ll b/llvm/test/CodeGen/Mips/msa/f16vec-insertelement.ll
new file mode 100644
index 00000000000000..efd0f33ed50bda
--- /dev/null
+++ b/llvm/test/CodeGen/Mips/msa/f16vec-insertelement.ll
@@ -0,0 +1,33 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=mipsel-unknown-linux-gnu -mcpu=mips32r5 -mattr=+fp64,+msa < %s | FileCheck %s --check-prefix=MIPS32
+; RUN: llc -mtriple=mips64el-unknown-linux-gnuabi64 -mcpu=mips64r5 -mattr=+fp64,+msa < %s | FileCheck %s --check-prefix=MIPS64
+
+; Test that a scalar f16 can be inserted into an f16 vector without crashing.
+; f16 is soft promoted on MIPS while v8f16 is a legal MSA type, so the
+; INSERT_VECTOR_ELT ends up with a soft promoted element operand. With a
+; constant index DAGCombine usually folds the insert into a BUILD_VECTOR
+; first; the variable index cases reach the type legalizer directly.
+
+define <8 x half> @insert_v8f16_idx(<8 x half> %v, half %e, i32 %idx) nounwind {
+;
+ %r = insertelement <8 x half> %v, half %e, i32 %idx
+ ret <8 x half> %r
+}
+
+define <8 x half> @insert_v8f16_imm(<8 x half> %v, half %e) nounwind {
+;
+ %r = insertelement <8 x half> %v, half %e, i32 3
+ ret <8 x half> %r
+}
+
+define <4 x half> @insert_v4f16_idx(<4 x half> %v, half %e, i32 %idx) nounwind {
+;
+ %r = insertelement <4 x half> %v, half %e, i32 %idx
+ ret <4 x half> %r
+}
+
+define <2 x half> @insert_v2f16_idx(<2 x half> %v, half %e, i32 %idx) nounwind {
+;
+ %r = insertelement <2 x half> %v, half %e, i32 %idx
+ ret <2 x half> %r
+}
diff --git a/llvm/test/CodeGen/RISCV/rvv/insertelt-fp-soft-promote.ll b/llvm/test/CodeGen/RISCV/rvv/insertelt-fp-soft-promote.ll
new file mode 100644
index 00000000000000..29c1b566e12a25
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/rvv/insertelt-fp-soft-promote.ll
@@ -0,0 +1,49 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=riscv32 -mattr=+v,+zvfhmin,+zvfbfmin -target-abi=ilp32d \
+; RUN: -verify-machineinstrs < %s | FileCheck %s
+; RUN: llc -mtriple=riscv64 -mattr=+v,+zvfhmin,+zvfbfmin -target-abi=lp64d \
+; RUN: -verify-machineinstrs < %s | FileCheck %s
+
+; Without Zfhmin/Zfbfmin the scalar f16/bf16 types are soft promoted while the
+; vector types are legal, so INSERT_VECTOR_ELT has to be legalized with a soft
+; promoted element operand.
+
+define <vscale x 4 x half> @insertelt_nxv4f16_0(<vscale x 4 x half> %v, half %elt) {
+ %r = insertelement <vscale x 4 x half> %v, half %elt, i32 0
+ ret <vscale x 4 x half> %r
+}
+
+define <vscale x 4 x half> @insertelt_nxv4f16_imm(<vscale x 4 x half> %v, half %elt) {
+ %r = insertelement <vscale x 4 x half> %v, half %elt, i32 3
+ ret <vscale x 4 x half> %r
+}
+
+define <vscale x 4 x half> @insertelt_nxv4f16_idx(<vscale x 4 x half> %v, half %elt, i32 zeroext %idx) {
+ %r = insertelement <vscale x 4 x half> %v, half %elt, i32 %idx
+ ret <vscale x 4 x half> %r
+}
+
+define <vscale x 4 x bfloat> @insertelt_nxv4bf16_0(<vscale x 4 x bfloat> %v, bfloat %elt) {
+ %r = insertelement <vscale x 4 x bfloat> %v, bfloat %elt, i32 0
+ ret <vscale x 4 x bfloat> %r
+}
+
+define <vscale x 4 x bfloat> @insertelt_nxv4bf16_imm(<vscale x 4 x bfloat> %v, bfloat %elt) {
+ %r = insertelement <vscale x 4 x bfloat> %v, bfloat %elt, i32 3
+ ret <vscale x 4 x bfloat> %r
+}
+
+define <vscale x 4 x bfloat> @insertelt_nxv4bf16_idx(<vscale x 4 x bfloat> %v, bfloat %elt, i32 zeroext %idx) {
+ %r = insertelement <vscale x 4 x bfloat> %v, bfloat %elt, i32 %idx
+ ret <vscale x 4 x bfloat> %r
+}
+
+define <4 x half> @insertelt_v4f16_idx(<4 x half> %v, half %elt, i32 zeroext %idx) {
+ %r = insertelement <4 x half> %v, half %elt, i32 %idx
+ ret <4 x half> %r
+}
+
+define <4 x bfloat> @insertelt_v4bf16_idx(<4 x bfloat> %v, bfloat %elt, i32 zeroext %idx) {
+ %r = insertelement <4 x bfloat> %v, bfloat %elt, i32 %idx
+ ret <4 x bfloat> %r
+}
>From 8b34d34de33dd699c72fad4614c36213d92e78cc Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Pawe=C5=82=20Bylica?= <pawel at hepcolgum.band>
Date: Tue, 15 Sep 2026 01:30:06 +0200
Subject: [PATCH 2/2] [SelectionDAG] Soft promote the element operand of
INSERT_VECTOR_ELT
When f16/bf16 is soft promoted but a vector of that type is legal,
INSERT_VECTOR_ELT is left with a soft promoted i16 element operand.
SoftPromoteHalfOperand had no case for it and failed with "Do not know
how to soft promote this operator's operand!".
Legalize it like BUILD_VECTOR and the inverse EXTRACT_VECTOR_ELT: bitcast
the vector to its integer counterpart, insert the promoted element there
and bitcast the result back. This also covers scalable vectors.
This is reachable on MIPS MSA and on RISC-V with Zvfhmin/Zvfbfmin but
without Zfhmin/Zfbfmin. X86, Hexagon and WebAssembly avoid it by custom
lowering INSERT_VECTOR_ELT on the scalar type; that still takes
precedence and can be removed separately.
Fixes #198104.
Assisted-by: Claude Code
---
.../SelectionDAG/LegalizeFloatTypes.cpp | 13 ++
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 1 +
.../CodeGen/Mips/msa/f16vec-insertelement.ll | 137 ++++++++++++++++++
.../RISCV/rvv/insertelt-fp-soft-promote.ll | 62 ++++++++
4 files changed, 213 insertions(+)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
index b18da75e46baec..b7efd80d93fd49 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeFloatTypes.cpp
@@ -2994,6 +2994,9 @@ bool DAGTypeLegalizer::SoftPromoteHalfOperand(SDNode *N, unsigned OpNo) {
case ISD::BUILD_VECTOR:
Res = SoftPromoteHalfOp_BUILD_VECTOR(N);
break;
+ case ISD::INSERT_VECTOR_ELT:
+ Res = SoftPromoteHalfOp_INSERT_VECTOR_ELT(N, OpNo);
+ break;
case ISD::FAKE_USE:
Res = SoftPromoteHalfOp_FAKE_USE(N, OpNo);
break;
@@ -3070,6 +3073,16 @@ SDValue DAGTypeLegalizer::SoftPromoteHalfOp_BUILD_VECTOR(SDNode *N) {
return DAG.getBitcast(VT, Res);
}
+SDValue DAGTypeLegalizer::SoftPromoteHalfOp_INSERT_VECTOR_ELT(SDNode *N,
+ unsigned OpNo) {
+ assert(OpNo == 1 && "Only Operand 1 must need promotion here");
+ SDValue Vec = BitConvertVectorToIntegerVector(N->getOperand(0));
+ SDValue Elt = GetSoftPromotedHalf(N->getOperand(OpNo));
+ SDValue Res = DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(N),
+ Vec.getValueType(), Vec, Elt, N->getOperand(2));
+ return DAG.getBitcast(N->getValueType(0), Res);
+}
+
SDValue DAGTypeLegalizer::SoftPromoteHalfOp_FAKE_USE(SDNode *N, unsigned OpNo) {
assert(OpNo == 1 && "Only Operand 1 must need promotion here");
SDValue Op = GetSoftPromotedHalf(N->getOperand(OpNo));
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index 6fc6d61c6a38df..d486b3f50cc901 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -787,6 +787,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
bool SoftPromoteHalfOperand(SDNode *N, unsigned OpNo);
SDValue SoftPromoteHalfOp_BITCAST(SDNode *N);
SDValue SoftPromoteHalfOp_BUILD_VECTOR(SDNode *N);
+ SDValue SoftPromoteHalfOp_INSERT_VECTOR_ELT(SDNode *N, unsigned OpNo);
SDValue SoftPromoteHalfOp_FAKE_USE(SDNode *N, unsigned OpNo);
SDValue SoftPromoteHalfOp_FCOPYSIGN(SDNode *N, unsigned OpNo);
SDValue SoftPromoteHalfOp_FP_EXTEND(SDNode *N);
diff --git a/llvm/test/CodeGen/Mips/msa/f16vec-insertelement.ll b/llvm/test/CodeGen/Mips/msa/f16vec-insertelement.ll
index efd0f33ed50bda..ee24ea9b5d7401 100644
--- a/llvm/test/CodeGen/Mips/msa/f16vec-insertelement.ll
+++ b/llvm/test/CodeGen/Mips/msa/f16vec-insertelement.ll
@@ -10,24 +10,161 @@
define <8 x half> @insert_v8f16_idx(<8 x half> %v, half %e, i32 %idx) nounwind {
;
+; MIPS32-LABEL: insert_v8f16_idx:
+; MIPS32: # %bb.0:
+; MIPS32-NEXT: insert.w $w0[0], $6
+; MIPS32-NEXT: insert.w $w0[1], $7
+; MIPS32-NEXT: lw $1, 16($sp)
+; MIPS32-NEXT: insert.w $w0[2], $1
+; MIPS32-NEXT: lw $1, 20($sp)
+; MIPS32-NEXT: insert.w $w0[3], $1
+; MIPS32-NEXT: lhu $1, 24($sp)
+; MIPS32-NEXT: lw $2, 28($sp)
+; MIPS32-NEXT: sll $2, $2, 1
+; MIPS32-NEXT: sld.b $w0, $w0[$2]
+; MIPS32-NEXT: insert.h $w0[0], $1
+; MIPS32-NEXT: neg $1, $2
+; MIPS32-NEXT: sld.b $w0, $w0[$1]
+; MIPS32-NEXT: jr $ra
+; MIPS32-NEXT: st.h $w0, 0($4)
+;
+; MIPS64-LABEL: insert_v8f16_idx:
+; MIPS64: # %bb.0:
+; MIPS64-NEXT: insert.d $w0[0], $4
+; MIPS64-NEXT: insert.d $w0[1], $5
+; MIPS64-NEXT: dext $1, $7, 0, 32
+; MIPS64-NEXT: sll $2, $6, 0
+; MIPS64-NEXT: dsll $1, $1, 1
+; MIPS64-NEXT: sld.b $w0, $w0[$1]
+; MIPS64-NEXT: insert.h $w0[0], $2
+; MIPS64-NEXT: dneg $1, $1
+; MIPS64-NEXT: sld.b $w0, $w0[$1]
+; MIPS64-NEXT: copy_s.d $2, $w0[0]
+; MIPS64-NEXT: jr $ra
+; MIPS64-NEXT: copy_s.d $3, $w0[1]
%r = insertelement <8 x half> %v, half %e, i32 %idx
ret <8 x half> %r
}
define <8 x half> @insert_v8f16_imm(<8 x half> %v, half %e) nounwind {
;
+; MIPS32-LABEL: insert_v8f16_imm:
+; MIPS32: # %bb.0:
+; MIPS32-NEXT: insert.w $w0[0], $6
+; MIPS32-NEXT: insert.w $w0[1], $7
+; MIPS32-NEXT: lw $1, 16($sp)
+; MIPS32-NEXT: insert.w $w0[2], $1
+; MIPS32-NEXT: lw $1, 20($sp)
+; MIPS32-NEXT: insert.w $w0[3], $1
+; MIPS32-NEXT: lhu $1, 24($sp)
+; MIPS32-NEXT: insert.h $w0[3], $1
+; MIPS32-NEXT: jr $ra
+; MIPS32-NEXT: st.h $w0, 0($4)
+;
+; MIPS64-LABEL: insert_v8f16_imm:
+; MIPS64: # %bb.0:
+; MIPS64-NEXT: insert.d $w0[0], $4
+; MIPS64-NEXT: insert.d $w0[1], $5
+; MIPS64-NEXT: sll $1, $6, 0
+; MIPS64-NEXT: insert.h $w0[3], $1
+; MIPS64-NEXT: copy_s.d $2, $w0[0]
+; MIPS64-NEXT: jr $ra
+; MIPS64-NEXT: copy_s.d $3, $w0[1]
%r = insertelement <8 x half> %v, half %e, i32 3
ret <8 x half> %r
}
define <4 x half> @insert_v4f16_idx(<4 x half> %v, half %e, i32 %idx) nounwind {
;
+; MIPS32-LABEL: insert_v4f16_idx:
+; MIPS32: # %bb.0:
+; MIPS32-NEXT: addiu $sp, $sp, -32
+; MIPS32-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill
+; MIPS32-NEXT: sw $fp, 24($sp) # 4-byte Folded Spill
+; MIPS32-NEXT: move $fp, $sp
+; MIPS32-NEXT: addiu $1, $zero, -16
+; MIPS32-NEXT: and $sp, $sp, $1
+; MIPS32-NEXT: sw $7, 4($sp)
+; MIPS32-NEXT: sw $6, 0($sp)
+; MIPS32-NEXT: lhu $1, 48($fp)
+; MIPS32-NEXT: lw $2, 52($fp)
+; MIPS32-NEXT: ld.h $w0, 0($sp)
+; MIPS32-NEXT: sll $2, $2, 1
+; MIPS32-NEXT: sld.b $w0, $w0[$2]
+; MIPS32-NEXT: insert.h $w0[0], $1
+; MIPS32-NEXT: neg $1, $2
+; MIPS32-NEXT: sld.b $w0, $w0[$1]
+; MIPS32-NEXT: copy_s.w $1, $w0[0]
+; MIPS32-NEXT: copy_s.w $2, $w0[1]
+; MIPS32-NEXT: sw $2, 4($4)
+; MIPS32-NEXT: sw $1, 0($4)
+; MIPS32-NEXT: move $sp, $fp
+; MIPS32-NEXT: lw $fp, 24($sp) # 4-byte Folded Reload
+; MIPS32-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload
+; MIPS32-NEXT: jr $ra
+; MIPS32-NEXT: addiu $sp, $sp, 32
+;
+; MIPS64-LABEL: insert_v4f16_idx:
+; MIPS64: # %bb.0:
+; MIPS64-NEXT: daddiu $sp, $sp, -16
+; MIPS64-NEXT: sd $4, 0($sp)
+; MIPS64-NEXT: dext $1, $6, 0, 32
+; MIPS64-NEXT: sll $2, $5, 0
+; MIPS64-NEXT: ld.h $w0, 0($sp)
+; MIPS64-NEXT: dsll $1, $1, 1
+; MIPS64-NEXT: sld.b $w0, $w0[$1]
+; MIPS64-NEXT: insert.h $w0[0], $2
+; MIPS64-NEXT: dneg $1, $1
+; MIPS64-NEXT: sld.b $w0, $w0[$1]
+; MIPS64-NEXT: copy_s.d $2, $w0[0]
+; MIPS64-NEXT: jr $ra
+; MIPS64-NEXT: daddiu $sp, $sp, 16
%r = insertelement <4 x half> %v, half %e, i32 %idx
ret <4 x half> %r
}
define <2 x half> @insert_v2f16_idx(<2 x half> %v, half %e, i32 %idx) nounwind {
;
+; MIPS32-LABEL: insert_v2f16_idx:
+; MIPS32: # %bb.0:
+; MIPS32-NEXT: addiu $sp, $sp, -32
+; MIPS32-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill
+; MIPS32-NEXT: sw $fp, 24($sp) # 4-byte Folded Spill
+; MIPS32-NEXT: move $fp, $sp
+; MIPS32-NEXT: addiu $1, $zero, -16
+; MIPS32-NEXT: and $sp, $sp, $1
+; MIPS32-NEXT: sw $5, 0($sp)
+; MIPS32-NEXT: ld.h $w0, 0($sp)
+; MIPS32-NEXT: sll $1, $7, 1
+; MIPS32-NEXT: sld.b $w0, $w0[$1]
+; MIPS32-NEXT: insert.h $w0[0], $6
+; MIPS32-NEXT: neg $1, $1
+; MIPS32-NEXT: sld.b $w0, $w0[$1]
+; MIPS32-NEXT: copy_s.w $1, $w0[0]
+; MIPS32-NEXT: sw $1, 0($4)
+; MIPS32-NEXT: move $sp, $fp
+; MIPS32-NEXT: lw $fp, 24($sp) # 4-byte Folded Reload
+; MIPS32-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload
+; MIPS32-NEXT: jr $ra
+; MIPS32-NEXT: addiu $sp, $sp, 32
+;
+; MIPS64-LABEL: insert_v2f16_idx:
+; MIPS64: # %bb.0:
+; MIPS64-NEXT: daddiu $sp, $sp, -16
+; MIPS64-NEXT: sll $1, $5, 0
+; MIPS64-NEXT: sw $1, 0($sp)
+; MIPS64-NEXT: dext $1, $7, 0, 32
+; MIPS64-NEXT: sll $2, $6, 0
+; MIPS64-NEXT: ld.h $w0, 0($sp)
+; MIPS64-NEXT: dsll $1, $1, 1
+; MIPS64-NEXT: sld.b $w0, $w0[$1]
+; MIPS64-NEXT: insert.h $w0[0], $2
+; MIPS64-NEXT: dneg $1, $1
+; MIPS64-NEXT: sld.b $w0, $w0[$1]
+; MIPS64-NEXT: copy_s.w $1, $w0[0]
+; MIPS64-NEXT: sw $1, 0($4)
+; MIPS64-NEXT: jr $ra
+; MIPS64-NEXT: daddiu $sp, $sp, 16
%r = insertelement <2 x half> %v, half %e, i32 %idx
ret <2 x half> %r
}
diff --git a/llvm/test/CodeGen/RISCV/rvv/insertelt-fp-soft-promote.ll b/llvm/test/CodeGen/RISCV/rvv/insertelt-fp-soft-promote.ll
index 29c1b566e12a25..af810bd1567946 100644
--- a/llvm/test/CodeGen/RISCV/rvv/insertelt-fp-soft-promote.ll
+++ b/llvm/test/CodeGen/RISCV/rvv/insertelt-fp-soft-promote.ll
@@ -9,41 +9,103 @@
; promoted element operand.
define <vscale x 4 x half> @insertelt_nxv4f16_0(<vscale x 4 x half> %v, half %elt) {
+; CHECK-LABEL: insertelt_nxv4f16_0:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmv.x.w a0, fa0
+; CHECK-NEXT: vsetvli a1, zero, e16, m1, tu, ma
+; CHECK-NEXT: vmv.s.x v8, a0
+; CHECK-NEXT: ret
%r = insertelement <vscale x 4 x half> %v, half %elt, i32 0
ret <vscale x 4 x half> %r
}
define <vscale x 4 x half> @insertelt_nxv4f16_imm(<vscale x 4 x half> %v, half %elt) {
+; CHECK-LABEL: insertelt_nxv4f16_imm:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmv.x.w a0, fa0
+; CHECK-NEXT: vsetivli zero, 4, e16, m1, tu, ma
+; CHECK-NEXT: vmv.s.x v9, a0
+; CHECK-NEXT: vslideup.vi v8, v9, 3
+; CHECK-NEXT: ret
%r = insertelement <vscale x 4 x half> %v, half %elt, i32 3
ret <vscale x 4 x half> %r
}
define <vscale x 4 x half> @insertelt_nxv4f16_idx(<vscale x 4 x half> %v, half %elt, i32 zeroext %idx) {
+; CHECK-LABEL: insertelt_nxv4f16_idx:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmv.x.w a1, fa0
+; CHECK-NEXT: vsetvli a2, zero, e16, m1, ta, ma
+; CHECK-NEXT: vmv.s.x v9, a1
+; CHECK-NEXT: addi a1, a0, 1
+; CHECK-NEXT: vsetvli zero, a1, e16, m1, tu, ma
+; CHECK-NEXT: vslideup.vx v8, v9, a0
+; CHECK-NEXT: ret
%r = insertelement <vscale x 4 x half> %v, half %elt, i32 %idx
ret <vscale x 4 x half> %r
}
define <vscale x 4 x bfloat> @insertelt_nxv4bf16_0(<vscale x 4 x bfloat> %v, bfloat %elt) {
+; CHECK-LABEL: insertelt_nxv4bf16_0:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmv.x.w a0, fa0
+; CHECK-NEXT: vsetvli a1, zero, e16, m1, tu, ma
+; CHECK-NEXT: vmv.s.x v8, a0
+; CHECK-NEXT: ret
%r = insertelement <vscale x 4 x bfloat> %v, bfloat %elt, i32 0
ret <vscale x 4 x bfloat> %r
}
define <vscale x 4 x bfloat> @insertelt_nxv4bf16_imm(<vscale x 4 x bfloat> %v, bfloat %elt) {
+; CHECK-LABEL: insertelt_nxv4bf16_imm:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmv.x.w a0, fa0
+; CHECK-NEXT: vsetivli zero, 4, e16, m1, tu, ma
+; CHECK-NEXT: vmv.s.x v9, a0
+; CHECK-NEXT: vslideup.vi v8, v9, 3
+; CHECK-NEXT: ret
%r = insertelement <vscale x 4 x bfloat> %v, bfloat %elt, i32 3
ret <vscale x 4 x bfloat> %r
}
define <vscale x 4 x bfloat> @insertelt_nxv4bf16_idx(<vscale x 4 x bfloat> %v, bfloat %elt, i32 zeroext %idx) {
+; CHECK-LABEL: insertelt_nxv4bf16_idx:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmv.x.w a1, fa0
+; CHECK-NEXT: vsetvli a2, zero, e16, m1, ta, ma
+; CHECK-NEXT: vmv.s.x v9, a1
+; CHECK-NEXT: addi a1, a0, 1
+; CHECK-NEXT: vsetvli zero, a1, e16, m1, tu, ma
+; CHECK-NEXT: vslideup.vx v8, v9, a0
+; CHECK-NEXT: ret
%r = insertelement <vscale x 4 x bfloat> %v, bfloat %elt, i32 %idx
ret <vscale x 4 x bfloat> %r
}
define <4 x half> @insertelt_v4f16_idx(<4 x half> %v, half %elt, i32 zeroext %idx) {
+; CHECK-LABEL: insertelt_v4f16_idx:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmv.x.w a1, fa0
+; CHECK-NEXT: vsetivli zero, 4, e16, m1, ta, ma
+; CHECK-NEXT: vmv.s.x v9, a1
+; CHECK-NEXT: addi a1, a0, 1
+; CHECK-NEXT: vsetvli zero, a1, e16, mf2, tu, ma
+; CHECK-NEXT: vslideup.vx v8, v9, a0
+; CHECK-NEXT: ret
%r = insertelement <4 x half> %v, half %elt, i32 %idx
ret <4 x half> %r
}
define <4 x bfloat> @insertelt_v4bf16_idx(<4 x bfloat> %v, bfloat %elt, i32 zeroext %idx) {
+; CHECK-LABEL: insertelt_v4bf16_idx:
+; CHECK: # %bb.0:
+; CHECK-NEXT: fmv.x.w a1, fa0
+; CHECK-NEXT: vsetivli zero, 4, e16, m1, ta, ma
+; CHECK-NEXT: vmv.s.x v9, a1
+; CHECK-NEXT: addi a1, a0, 1
+; CHECK-NEXT: vsetvli zero, a1, e16, mf2, tu, ma
+; CHECK-NEXT: vslideup.vx v8, v9, a0
+; CHECK-NEXT: ret
%r = insertelement <4 x bfloat> %v, bfloat %elt, i32 %idx
ret <4 x bfloat> %r
}
More information about the llvm-commits
mailing list