[llvm] [IR][REVEC] Define llvm.vector.repeat intrinsic (PR #208212)
Gaƫtan Bossu via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 01:41:11 PDT 2026
https://github.com/gbossu updated https://github.com/llvm/llvm-project/pull/208212
>From 9e2d7d492eaf721670e16d2dd3d1463f2ff63e17 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Fri, 3 Jul 2026 11:25:48 +0000
Subject: [PATCH 01/19] [IR] Define llvm.vector.broadcast intrinsic
This is used broadcast a smaller vector into a wider one. The patch adds
basic legalisation support and ISel for AArch64.
---
llvm/include/llvm/CodeGen/ISDOpcodes.h | 5 +
llvm/include/llvm/IR/Intrinsics.td | 7 +
.../include/llvm/Target/TargetSelectionDAG.td | 4 +
.../SelectionDAG/LegalizeIntegerTypes.cpp | 30 ++
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 2 +
.../lib/CodeGen/SelectionDAG/SelectionDAG.cpp | 12 +
.../SelectionDAG/SelectionDAGBuilder.cpp | 6 +
.../SelectionDAG/SelectionDAGDumper.cpp | 1 +
llvm/lib/Target/AArch64/SVEInstrFormats.td | 42 ++-
.../sve-vector-broadcast-unsupported.ll | 9 +
.../CodeGen/AArch64/sve-vector-broadcast.ll | 318 ++++++++++++++++++
11 files changed, 430 insertions(+), 6 deletions(-)
create mode 100644 llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
create mode 100644 llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index 1af501d4c8dd0..fa215c4ce262f 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -637,6 +637,11 @@ enum NodeType {
/// Result[J] = EXTRACT_SUBVECTOR(Interleaved, J * getVectorMinNumElements())
VECTOR_INTERLEAVE,
+ /// VECTOR_BROADCAST(SRC_SUBVEC)
+ /// Duplicate a vector in a larger vector. The element count of the result
+ /// type is expected to be a multiple of the input vector's.
+ VECTOR_BROADCAST,
+
/// VECTOR_REVERSE(VECTOR) - Returns a vector, of the same type as VECTOR,
/// whose elements are shuffled using the following algorithm:
/// RESULT[i] = VECTOR[VECTOR.ElementCount - 1 - i]
diff --git a/llvm/include/llvm/IR/Intrinsics.td b/llvm/include/llvm/IR/Intrinsics.td
index 26b7c772eb244..36499fcb560d8 100644
--- a/llvm/include/llvm/IR/Intrinsics.td
+++ b/llvm/include/llvm/IR/Intrinsics.td
@@ -2816,6 +2816,13 @@ foreach n = 2...8 in {
[IntrNoMem, IntrSpeculatable]>;
}
+// vector_broadcast( SrcVector )
+// Broadcast a vector in a larger one.
+// This is the equivalent of a vector splat for vector types.
+def int_vector_broadcast : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
+ [llvm_anyvector_ty],
+ [IntrNoMem, IntrSpeculatable]>;
+
//===-------------- Intrinsics to perform partial reduction ---------------===//
def int_vector_partial_reduce_add : DefaultAttrsIntrinsic<[LLVMMatchType<0>],
diff --git a/llvm/include/llvm/Target/TargetSelectionDAG.td b/llvm/include/llvm/Target/TargetSelectionDAG.td
index b0fd9c733e5b6..cab55a327fa53 100644
--- a/llvm/include/llvm/Target/TargetSelectionDAG.td
+++ b/llvm/include/llvm/Target/TargetSelectionDAG.td
@@ -936,6 +936,10 @@ def vector_insert_subvec : SDNode<"ISD::INSERT_SUBVECTOR",
def extract_subvector : SDNode<"ISD::EXTRACT_SUBVECTOR", SDTSubVecExtract, []>;
def insert_subvector : SDNode<"ISD::INSERT_SUBVECTOR", SDTSubVecInsert, []>;
+def vector_broadcast : SDNode<"ISD::VECTOR_BROADCAST",
+ SDTypeProfile<1, 1, [SDTCisVec<1>, SDTCisVec<0>]>,
+ []>;
+
def find_last_active
: SDNode<"ISD::VECTOR_FIND_LAST_ACTIVE",
SDTypeProfile<1, 1, [SDTCisInt<0>, SDTCisVec<1>]>, []>;
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
index a90bb06fb424d..5509199255560 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
@@ -127,6 +127,9 @@ void DAGTypeLegalizer::PromoteIntegerResult(SDNode *N, unsigned ResNo) {
case ISD::VECTOR_SPLICE_RIGHT:
Res = PromoteIntRes_VECTOR_SPLICE(N);
break;
+ case ISD::VECTOR_BROADCAST:
+ Res = PromoteIntRes_VECTOR_BROADCAST(N);
+ break;
case ISD::VECTOR_INTERLEAVE:
case ISD::VECTOR_DEINTERLEAVE:
Res = PromoteIntRes_VECTOR_INTERLEAVE_DEINTERLEAVE(N);
@@ -2157,6 +2160,9 @@ bool DAGTypeLegalizer::PromoteIntegerOperand(SDNode *N, unsigned OpNo) {
case ISD::PARTIAL_REDUCE_SUMLA:
Res = PromoteIntOp_PARTIAL_REDUCE_MLA(N);
break;
+ case ISD::VECTOR_BROADCAST:
+ Res = PromoteIntOp_VECTOR_BROADCAST(N);
+ break;
case ISD::LOOP_DEPENDENCE_RAW_MASK:
case ISD::LOOP_DEPENDENCE_WAR_MASK:
Res = PromoteIntOp_LOOP_DEPENDENCE_MASK(N);
@@ -3018,6 +3024,17 @@ SDValue DAGTypeLegalizer::PromoteIntOp_LOOP_DEPENDENCE_MASK(SDNode *N) {
return SDValue(DAG.UpdateNodeOperands(N, NewOps), 0);
}
+SDValue DAGTypeLegalizer::PromoteIntOp_VECTOR_BROADCAST(SDNode *N) {
+ SDLoc DL(N);
+ SDValue Src = GetPromotedInteger(N->getOperand(0));
+ EVT SrcVT = Src.getValueType();
+ EVT OrigVT = N->getValueType(0);
+ EVT NewVT = EVT::getVectorVT(*DAG.getContext(), SrcVT.getVectorElementType(),
+ OrigVT.getVectorElementCount());
+ SDValue Res = DAG.getNode(ISD::VECTOR_BROADCAST, DL, NewVT, Src);
+ return DAG.getNode(ISD::TRUNCATE, DL, OrigVT, Res);
+}
+
//===----------------------------------------------------------------------===//
// Integer Result Expansion
//===----------------------------------------------------------------------===//
@@ -6106,6 +6123,19 @@ SDValue DAGTypeLegalizer::PromoteIntRes_VECTOR_SPLICE(SDNode *N) {
return DAG.getNode(N->getOpcode(), dl, OutVT, V0, V1, N->getOperand(2));
}
+SDValue DAGTypeLegalizer::PromoteIntRes_VECTOR_BROADCAST(SDNode *N) {
+ SDLoc DL(N);
+
+ EVT OutVT = N->getValueType(0);
+ EVT NOutVT = TLI.getTypeToTransformTo(*DAG.getContext(), OutVT);
+ assert(NOutVT.isVector() && "This type must be promoted to a vector type");
+ EVT NInVT = N->getOperand(0).getValueType().changeVectorElementType(
+ *DAG.getContext(), NOutVT.getVectorElementType());
+
+ SDValue Op = DAG.getNode(ISD::ANY_EXTEND, DL, NInVT, N->getOperand(0));
+ return DAG.getNode(N->getOpcode(), DL, NOutVT, Op);
+}
+
SDValue DAGTypeLegalizer::PromoteIntRes_VECTOR_INTERLEAVE_DEINTERLEAVE(SDNode *N) {
SDLoc DL(N);
unsigned Factor = N->getNumOperands();
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index c9583d99ddbf4..e1b1468c4e29a 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -286,6 +286,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue PromoteIntRes_VECTOR_REVERSE(SDNode *N);
SDValue PromoteIntRes_VECTOR_SHUFFLE(SDNode *N);
SDValue PromoteIntRes_VECTOR_SPLICE(SDNode *N);
+ SDValue PromoteIntRes_VECTOR_BROADCAST(SDNode *N);
SDValue PromoteIntRes_VECTOR_INTERLEAVE_DEINTERLEAVE(SDNode *N);
SDValue PromoteIntRes_BUILD_VECTOR(SDNode *N);
SDValue PromoteIntRes_ScalarOp(SDNode *N);
@@ -419,6 +420,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue PromoteIntOp_GET_ACTIVE_LANE_MASK(SDNode *N);
SDValue PromoteIntOp_VECTOR_MATCH(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_PARTIAL_REDUCE_MLA(SDNode *N);
+ SDValue PromoteIntOp_VECTOR_BROADCAST(SDNode *N);
SDValue PromoteIntOp_LOOP_DEPENDENCE_MASK(SDNode *N);
SDValue PromoteIntOp_MaskedBinOp(SDNode *N, unsigned OpNo);
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
index 0a3eb65965826..a2fef38e9d4ca 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
@@ -9185,6 +9185,18 @@ SDValue SelectionDAG::getNode(unsigned Opcode, const SDLoc &DL, EVT VT,
}
break;
}
+ case ISD::VECTOR_BROADCAST: {
+ [[maybe_unused]] EVT InputVT = N1.getValueType();
+ assert(InputVT.isVector() && VT.isVector() &&
+ "Expected the input and output of the VECTOR_BROADCAST node to be "
+ "vectors!");
+ assert(VT.getVectorElementCount().hasKnownScalarFactor(
+ InputVT.getVectorElementCount()) &&
+ "Expected the element count of the output of the VECTOR_BROADCAST "
+ "node to be a positive integer multiple of the element count of the "
+ "source operand!");
+ break;
+ }
case ISD::BITCAST:
// Fold bit_convert nodes from a type to themselves.
if (N1.getValueType() == VT)
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
index 065347d903033..aa4ec35b1955a 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
@@ -8649,6 +8649,12 @@ void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
case Intrinsic::vector_deinterleave8:
visitVectorDeinterleave(I, 8);
return;
+ case Intrinsic::vector_broadcast: {
+ SDValue Vec = getValue(I.getOperand(0));
+ EVT ResultVT = TLI.getValueType(DAG.getDataLayout(), I.getType());
+ setValue(&I, DAG.getNode(ISD::VECTOR_BROADCAST, sdl, ResultVT, Vec));
+ return;
+ }
case Intrinsic::experimental_vector_compress:
setValue(&I, DAG.getNode(ISD::VECTOR_COMPRESS, sdl,
getValue(I.getArgOperand(0)).getValueType(),
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
index 32e0cafd9f71c..2819a2dd78ad4 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
@@ -360,6 +360,7 @@ std::string SDNode::getOperationName(const SelectionDAG *G) const {
case ISD::EXTRACT_SUBVECTOR: return "extract_subvector";
case ISD::VECTOR_DEINTERLEAVE: return "vector_deinterleave";
case ISD::VECTOR_INTERLEAVE: return "vector_interleave";
+ case ISD::VECTOR_BROADCAST: return "vector_broadcast";
case ISD::SCALAR_TO_VECTOR: return "scalar_to_vector";
case ISD::VECTOR_SHUFFLE: return "vector_shuffle";
case ISD::VECTOR_SPLICE_LEFT: return "vector_splice_left";
diff --git a/llvm/lib/Target/AArch64/SVEInstrFormats.td b/llvm/lib/Target/AArch64/SVEInstrFormats.td
index e9191c95eee35..0528eaa854f3e 100644
--- a/llvm/lib/Target/AArch64/SVEInstrFormats.td
+++ b/llvm/lib/Target/AArch64/SVEInstrFormats.td
@@ -1553,12 +1553,21 @@ multiclass sve_int_perm_dup_i<string asm> {
// Duplicate an extracted vector element across a vector.
- def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (nxv16i8 ZPR:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) ZPR:$vec, sve_elm_idx_extdup_b:$index)>;
- def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (v16i8 V128:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_b:$index)>;
- def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (v8i8 V64:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_b:$index)>;
+ foreach VT = [nxv16i8] in {
+ def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.Packed ZPR:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) ZPR:$vec, sve_elm_idx_extdup_b:$index)>;
+ def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.ZSub V128:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_b:$index)>;
+ def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_b:$index)>;
+
+ // Broadcast a whole 128-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
+ (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
+ // Broadcast a whole 64-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
+ (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
+ }
foreach VT = [nxv8i16, nxv2f16, nxv4f16, nxv8f16, nxv2bf16, nxv4bf16, nxv8bf16] in {
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.Packed ZPR:$vec), sve_elm_idx_extdup_h:$index)))),
@@ -1567,6 +1576,13 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _H) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_h:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_h:$index)))),
(!cast<Instruction>(NAME # _H) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_h:$index)>;
+
+ // Broadcast a whole 128-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
+ (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
+ // Broadcast a whole 64-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
+ (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
foreach VT = [nxv4i32, nxv2f32, nxv4f32 ] in {
@@ -1576,6 +1592,13 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _S) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_s:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_s:$index)))),
(!cast<Instruction>(NAME # _S) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_s:$index)>;
+
+ // Broadcast a whole 128-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
+ (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
+ // Broadcast a whole 64-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
+ (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
foreach VT = [nxv2i64, nxv2f64] in {
@@ -1585,6 +1608,13 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_d:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_d:$index)))),
(!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_d:$index)>;
+
+ // Broadcast a whole 128-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
+ (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
+ // Broadcast a whole 64-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
+ (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
// When extracting from an unpacked vector the index must be scaled to account
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
new file mode 100644
index 0000000000000..49152c702505a
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
@@ -0,0 +1,9 @@
+; RUN: not --crash llc -mtriple=aarch64-linux-gnu -mattr=+sve < %s 2>&1 | FileCheck %s
+
+; CHECK: LLVM ERROR: Do not know how to widen this operator's operand!
+
+; TODO: Support broadcasts from 32-bit vec
+define <vscale x 8 x half> @broadcast_single_f16(<2 x half> %a) {
+ %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<2 x half> %a)
+ ret <vscale x 8 x half> %out
+}
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
new file mode 100644
index 0000000000000..f0ef8b2c2e350
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -0,0 +1,318 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve < %s | FileCheck %s
+
+
+define <vscale x 16 x i8> @broadcast_quad_i8(<16 x i8> %a) {
+; CHECK-LABEL: broadcast_quad_i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v16i8(<16 x i8> %a)
+ ret <vscale x 16 x i8> %out
+}
+
+define <vscale x 16 x i8> @broadcast_double_i8(<8 x i8> %a) {
+; CHECK-LABEL: broadcast_double_i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %out = call <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v8i8(<8 x i8> %a)
+ ret <vscale x 16 x i8> %out
+}
+
+define <vscale x 8 x i16> @broadcast_double_i8_to_double_sve(<8 x i8> %a) {
+; CHECK-LABEL: broadcast_double_i8_to_double_sve:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ushll v0.8h, v0.8b, #0
+; CHECK-NEXT: ptrue p0.h
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: sxtb z0.h, p0/m, z0.h
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x i8> @llvm.vector.broadcast.nxv8i8.v8i8(<8 x i8> %a)
+ %out.legal = sext <vscale x 8 x i8> %out to <vscale x 8 x i16>
+ ret <vscale x 8 x i16> %out.legal
+}
+
+define <vscale x 8 x i16> @broadcast_quad_i16(<8 x i16> %a) {
+; CHECK-LABEL: broadcast_quad_i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x i16> @llvm.vector.broadcast.nxv8i16.v8i16(<8 x i16> %a)
+ ret <vscale x 8 x i16> %out
+}
+
+define <vscale x 8 x i16> @broadcast_double_i16(<4 x i16> %a) {
+; CHECK-LABEL: broadcast_double_i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x i16> @llvm.vector.broadcast.nxv8i16.v4i16(<4 x i16> %a)
+ ret <vscale x 8 x i16> %out
+}
+
+define <vscale x 4 x i32> @broadcast_double_i16_to_double_sve(<4 x i16> %a) {
+; CHECK-LABEL: broadcast_double_i16_to_double_sve:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ushll v0.4s, v0.4h, #0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.s, z0.s, #0xffff
+; CHECK-NEXT: ret
+ %out = call <vscale x 4 x i16> @llvm.vector.broadcast.nxv4i16.v4i16(<4 x i16> %a)
+ %out.legal = zext <vscale x 4 x i16> %out to <vscale x 4 x i32>
+ ret <vscale x 4 x i32> %out.legal
+}
+
+define <vscale x 4 x i32> @broadcast_quad_i32(<4 x i32> %a) {
+; CHECK-LABEL: broadcast_quad_i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 4 x i32> @llvm.vector.broadcast.nxv4i32.v4i32(<4 x i32> %a)
+ ret <vscale x 4 x i32> %out
+}
+
+define <vscale x 2 x i64> @broadcast_quad_i64(<2 x i64> %a) {
+; CHECK-LABEL: broadcast_quad_i64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v2i64(<2 x i64> %a)
+ ret <vscale x 2 x i64> %out
+}
+
+define <vscale x 8 x half> @broadcast_quad_f16(<8 x half> %a) {
+; CHECK-LABEL: broadcast_quad_f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<8 x half> %a)
+ ret <vscale x 8 x half> %out
+}
+
+define <vscale x 8 x half> @broadcast_double_f16(<4 x half> %a) {
+; CHECK-LABEL: broadcast_double_f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<4 x half> %a)
+ ret <vscale x 8 x half> %out
+}
+
+define <vscale x 4 x half> @broadcast_double_f16_to_double_sve(<4 x half> %a) {
+; CHECK-LABEL: broadcast_double_f16_to_double_sve:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<4 x half> %a)
+ ret <vscale x 4 x half> %out
+}
+
+define <vscale x 8 x bfloat> @broadcast_quad_bf16(<8 x bfloat> %a) #0 {
+; CHECK-LABEL: broadcast_quad_bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x bfloat> @llvm.vector.broadcast.nxv8bf16.v8bf16(<8 x bfloat> %a)
+ ret <vscale x 8 x bfloat> %out
+}
+
+define <vscale x 4 x float> @broadcast_quad_f32(<4 x float> %a) {
+; CHECK-LABEL: broadcast_quad_f32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 4 x float> @llvm.vector.broadcast.nxv4f32.v4f32(<4 x float> %a)
+ ret <vscale x 4 x float> %out
+}
+
+define <vscale x 2 x double> @broadcast_quad_f64(<2 x double> %a) {
+; CHECK-LABEL: broadcast_quad_f64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 2 x double> @llvm.vector.broadcast.nxv2f64.v2f64(<2 x double> %a)
+ ret <vscale x 2 x double> %out
+}
+
+; Predicates
+
+define <vscale x 16 x i1> @broadcast_v16i1_to_nxv16i1(<16 x i8> %a) {
+; CHECK-LABEL: broadcast_v16i1_to_nxv16i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: ptrue p0.b
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.b, z0.b, #0x1
+; CHECK-NEXT: cmpne p0.b, p0/z, z0.b, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <16 x i8> %a to <16 x i1>
+ %out = call <vscale x 16 x i1> @llvm.vector.broadcast.nxv16i1.v16i1(<16 x i1> %a.legal)
+ ret <vscale x 16 x i1> %out
+}
+
+define <vscale x 16 x i1> @broadcast_v8i1_to_nxv16i1(<8 x i8> %a) {
+; CHECK-LABEL: broadcast_v8i1_to_nxv16i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: ptrue p0.b
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: and z0.b, z0.b, #0x1
+; CHECK-NEXT: cmpne p0.b, p0/z, z0.b, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <8 x i8> %a to <8 x i1>
+ %out = call <vscale x 16 x i1> @llvm.vector.broadcast.nxv16i1.v8i1(<8 x i1> %a.legal)
+ ret <vscale x 16 x i1> %out
+}
+
+define <vscale x 8 x i1> @broadcast_v8i1_double_to_nxv8i1(<8 x i8> %a) {
+; CHECK-LABEL: broadcast_v8i1_double_to_nxv8i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ushll v0.8h, v0.8b, #0
+; CHECK-NEXT: ptrue p0.h
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.h, z0.h, #0x1
+; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <8 x i8> %a to <8 x i1>
+ %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v8i1(<8 x i1> %a.legal)
+ ret <vscale x 8 x i1> %out
+}
+
+define <vscale x 8 x i1> @broadcast_v8i1_quad_to_nxv8i1(<8 x i16> %a) {
+; CHECK-LABEL: broadcast_v8i1_quad_to_nxv8i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: ptrue p0.h
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.h, z0.h, #0x1
+; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <8 x i16> %a to <8 x i1>
+ %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v8i1(<8 x i1> %a.legal)
+ ret <vscale x 8 x i1> %out
+}
+
+define <vscale x 8 x i1> @broadcast_v4i1_double_to_nxv8i1(<4 x i16> %a) {
+; CHECK-LABEL: broadcast_v4i1_double_to_nxv8i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: ptrue p0.h
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: and z0.h, z0.h, #0x1
+; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <4 x i16> %a to <4 x i1>
+ %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v4i1(<4 x i1> %a.legal)
+ ret <vscale x 8 x i1> %out
+}
+
+define <vscale x 8 x i1> @broadcast_v4i1_quad_to_nxv8i1(<4 x i32> %a) {
+; CHECK-LABEL: broadcast_v4i1_quad_to_nxv8i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: xtn v0.4h, v0.4s
+; CHECK-NEXT: ptrue p0.h
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: and z0.h, z0.h, #0x1
+; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <4 x i32> %a to <4 x i1>
+ %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v4i1(<4 x i1> %a.legal)
+ ret <vscale x 8 x i1> %out
+}
+
+define <vscale x 4 x i1> @broadcast_v4i1_double_to_nxv4i1(<4 x i16> %a) {
+; CHECK-LABEL: broadcast_v4i1_double_to_nxv4i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ushll v0.4s, v0.4h, #0
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.s, z0.s, #0x1
+; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <4 x i16> %a to <4 x i1>
+ %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v4i1(<4 x i1> %a.legal)
+ ret <vscale x 4 x i1> %out
+}
+
+define <vscale x 4 x i1> @broadcast_v4i1_quad_to_nxv4i1(<4 x i32> %a) {
+; CHECK-LABEL: broadcast_v4i1_quad_to_nxv4i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.s, z0.s, #0x1
+; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <4 x i32> %a to <4 x i1>
+ %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v4i1(<4 x i1> %a.legal)
+ ret <vscale x 4 x i1> %out
+}
+
+define <vscale x 4 x i1> @broadcast_v2i1_double_to_nxv4i1(<2 x i32> %a) {
+; CHECK-LABEL: broadcast_v2i1_double_to_nxv4i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: and z0.s, z0.s, #0x1
+; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <2 x i32> %a to <2 x i1>
+ %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v2i1(<2 x i1> %a.legal)
+ ret <vscale x 4 x i1> %out
+}
+
+define <vscale x 4 x i1> @broadcast_v2i1_quad_to_nxv4i1(<2 x i64> %a) {
+; CHECK-LABEL: broadcast_v2i1_quad_to_nxv4i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: xtn v0.2s, v0.2d
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: and z0.s, z0.s, #0x1
+; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <2 x i64> %a to <2 x i1>
+ %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v2i1(<2 x i1> %a.legal)
+ ret <vscale x 4 x i1> %out
+}
+
+define <vscale x 2 x i1> @broadcast_v2i1_double_to_nxv2i1(<2 x i32> %a) {
+; CHECK-LABEL: broadcast_v2i1_double_to_nxv2i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ushll v0.2d, v0.2s, #0
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.d, z0.d, #0x1
+; CHECK-NEXT: cmpne p0.d, p0/z, z0.d, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <2 x i32> %a to <2 x i1>
+ %out = call <vscale x 2 x i1> @llvm.vector.broadcast.nxv2i1.v2i1(<2 x i1> %a.legal)
+ ret <vscale x 2 x i1> %out
+}
+
+define <vscale x 2 x i1> @broadcast_v2i1_quad_to_nxv2i1(<2 x i64> %a) {
+; CHECK-LABEL: broadcast_v2i1_quad_to_nxv2i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.d, z0.d, #0x1
+; CHECK-NEXT: cmpne p0.d, p0/z, z0.d, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <2 x i64> %a to <2 x i1>
+ %out = call <vscale x 2 x i1> @llvm.vector.broadcast.nxv2i1.v2i1(<2 x i1> %a.legal)
+ ret <vscale x 2 x i1> %out
+}
>From 378ecf86e2ab0e44183a477e0d9dc47fea03a4e3 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Fri, 7 Aug 2026 09:28:26 +0000
Subject: [PATCH 02/19] Support splitting results
---
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 1 +
.../SelectionDAG/LegalizeVectorTypes.cpp | 46 +++++++++++++++++
.../CodeGen/AArch64/sve-vector-broadcast.ll | 51 +++++++++++++++++++
3 files changed, 98 insertions(+)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index e1b1468c4e29a..c1ebbed1ab6fc 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -954,6 +954,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
void SplitVecRes_ScalarOp(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_STEP_VECTOR(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_SETCC(SDNode *N, SDValue &Lo, SDValue &Hi);
+ void SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_VECTOR_REVERSE(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_VECTOR_SHUFFLE(ShuffleVectorSDNode *N, SDValue &Lo,
SDValue &Hi);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 48fc990700b3c..805b150f88b13 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -1449,6 +1449,9 @@ void DAGTypeLegalizer::SplitVectorResult(SDNode *N, unsigned ResNo) {
case ISD::SETCC:
SplitVecRes_SETCC(N, Lo, Hi);
break;
+ case ISD::VECTOR_BROADCAST:
+ SplitVecRes_VECTOR_BROADCAST(N, Lo, Hi);
+ break;
case ISD::VECTOR_REVERSE:
SplitVecRes_VECTOR_REVERSE(N, Lo, Hi);
break;
@@ -3485,6 +3488,49 @@ void DAGTypeLegalizer::SplitVecRes_FP_TO_XINT_SAT(SDNode *N, SDValue &Lo,
Hi = DAG.getNode(N->getOpcode(), dl, DstVTHi, SrcHi, N->getOperand(1));
}
+void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
+ SDValue &Hi) {
+ EVT VT = N->getValueType(0);
+ SDValue Src = N->getOperand(0);
+ EVT SrcVT = Src.getValueType();
+ EVT LoVT, HiVT;
+ std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(VT);
+ assert(LoVT == HiVT && "Expected equal split types");
+
+ // Simple case: The split type is same as SrcVT.
+ if (LoVT == SrcVT) {
+ Lo = Hi = Src;
+ return;
+ }
+
+ // Second case: Src is known to be wider than LoVT.
+ SDLoc DL(N);
+ if (LoVT.getVectorMinNumElements() >= SrcVT.getVectorMinNumElements()) {
+ Lo = Hi = DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Src);
+ return;
+ }
+
+ // Final case: VT is scalable and has the same minimum EC.
+ // Use smaller even/odd source vectors so their broadcasts can be
+ // reinterleaved in the original lane order for every value of vscale.
+ assert(VT.getVectorMinNumElements() == SrcVT.getVectorMinNumElements() &&
+ VT.isScalableVector());
+ SDValue SrcLo, SrcHi;
+ std::tie(SrcLo, SrcHi) = DAG.SplitVector(Src, DL);
+ EVT SrcSplitVT = SrcLo.getValueType();
+ SDValue Deinterleaved =
+ DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL,
+ DAG.getVTList(SrcSplitVT, SrcSplitVT), SrcLo, SrcHi);
+ SDValue Even =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(0));
+ SDValue Odd =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(1));
+ SDValue Interleaved = DAG.getNode(ISD::VECTOR_INTERLEAVE, DL,
+ DAG.getVTList(LoVT, HiVT), Even, Odd);
+ Lo = Interleaved.getValue(0);
+ Hi = Interleaved.getValue(1);
+}
+
void DAGTypeLegalizer::SplitVecRes_VECTOR_REVERSE(SDNode *N, SDValue &Lo,
SDValue &Hi) {
SDValue InLo, InHi;
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
index f0ef8b2c2e350..7f1a37529f912 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -45,6 +45,45 @@ define <vscale x 8 x i16> @broadcast_quad_i16(<8 x i16> %a) {
ret <vscale x 8 x i16> %out
}
+define <vscale x 16 x i8> @broadcast_quad_i16_to_wide_sve(<8 x i16> %a) {
+; CHECK-LABEL: broadcast_quad_i16_to_wide_sve:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: uzp1 z0.b, z0.b, z0.b
+; CHECK-NEXT: ret
+ %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.v8i16(<8 x i16> %a)
+ %out.legal = trunc <vscale x 16 x i16> %out to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %out.legal
+}
+
+define <vscale x 16 x i8> @broadcast_wide_i16(<8 x i16> %a.lo, <8 x i16> %a.hi) {
+; CHECK-LABEL: broadcast_wide_i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: uzp2 v2.8h, v0.8h, v1.8h
+; CHECK-NEXT: uzp1 v0.8h, v0.8h, v1.8h
+; CHECK-NEXT: mov z1.q, q2
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: zip2 z2.h, z0.h, z1.h
+; CHECK-NEXT: zip1 z0.h, z0.h, z1.h
+; CHECK-NEXT: uzp1 z0.b, z0.b, z2.b
+; CHECK-NEXT: ret
+ %a = shufflevector <8 x i16> %a.lo, <8 x i16> %a.hi,
+ <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+ %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.v16i16(<16 x i16> %a)
+ %out.legal = trunc <vscale x 16 x i16> %out to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %out.legal
+}
+
+define <vscale x 16 x i16> @broadcast_nxv8i16_to_nxv16i16(<vscale x 8 x i16> %a) {
+; CHECK-LABEL: broadcast_nxv8i16_to_nxv16i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: mov z1.d, z0.d
+; CHECK-NEXT: ret
+ %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.nxv8i16(<vscale x 8 x i16> %a)
+ ret <vscale x 16 x i16> %out
+}
+
define <vscale x 8 x i16> @broadcast_double_i16(<4 x i16> %a) {
; CHECK-LABEL: broadcast_double_i16:
; CHECK: // %bb.0:
@@ -67,6 +106,18 @@ define <vscale x 4 x i32> @broadcast_double_i16_to_double_sve(<4 x i16> %a) {
ret <vscale x 4 x i32> %out.legal
}
+define <vscale x 16 x i8> @broadcast_double_i16_to_wide_sve(<4 x i16> %a) {
+; CHECK-LABEL: broadcast_double_i16_to_wide_sve:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uzp1 z0.b, z0.b, z0.b
+; CHECK-NEXT: ret
+ %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.v4i16(<4 x i16> %a)
+ %out.legal = trunc <vscale x 16 x i16> %out to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %out.legal
+}
+
define <vscale x 4 x i32> @broadcast_quad_i32(<4 x i32> %a) {
; CHECK-LABEL: broadcast_quad_i32:
; CHECK: // %bb.0:
>From 9fe07a5a1267a2ed90dcb71ff8d32c2c37172581 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Fri, 7 Aug 2026 09:30:13 +0000
Subject: [PATCH 03/19] Support expansion to shufflevector
---
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 19 ++++++
.../Target/AArch64/AArch64ISelLowering.cpp | 3 +
llvm/test/CodeGen/AArch64/vector-broadcast.ll | 68 +++++++++++++++++++
3 files changed, 90 insertions(+)
create mode 100644 llvm/test/CodeGen/AArch64/vector-broadcast.ll
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index bd9c47806a9ca..757453f069b33 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -3690,6 +3690,25 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
case ISD::INSERT_VECTOR_ELT:
Results.push_back(ExpandINSERT_VECTOR_ELT(SDValue(Node, 0)));
break;
+ case ISD::VECTOR_BROADCAST: {
+ EVT VT = Node->getValueType(0);
+ EVT SrcVT = Node->getOperand(0).getValueType();
+ assert(VT.isFixedLengthVector() && SrcVT.isFixedLengthVector() &&
+ "Can only expand broadcasts of fixed-length vectors");
+
+ SDValue Src = Node->getOperand(0);
+ SDValue PaddedSrc = DAG.getInsertSubvector(dl, DAG.getUNDEF(VT), Src, 0);
+
+ // Create a shuffle mask that duplicates Src.
+ unsigned NumElts = VT.getVectorNumElements();
+ unsigned SrcNumElts = SrcVT.getVectorNumElements();
+ SmallVector<int, 8> Mask;
+ for (unsigned I = 0; I != NumElts; ++I)
+ Mask.push_back(I % SrcNumElts);
+ Results.push_back(
+ DAG.getVectorShuffle(VT, dl, PaddedSrc, DAG.getUNDEF(VT), Mask));
+ break;
+ }
case ISD::VECTOR_SHUFFLE: {
SmallVector<int, 32> NewMask;
ArrayRef<int> Mask = cast<ShuffleVectorSDNode>(Node)->getMask();
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 2f99a7e7db27b..62b423a3c76d6 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -740,6 +740,9 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::ROTR, VT, Expand);
}
+ for (MVT VT : MVT::fixedlen_vector_valuetypes())
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Expand);
+
// AArch64 doesn't have i32 MULH{S|U}.
setOperationAction(ISD::MULHU, MVT::i32, Expand);
setOperationAction(ISD::MULHS, MVT::i32, Expand);
diff --git a/llvm/test/CodeGen/AArch64/vector-broadcast.ll b/llvm/test/CodeGen/AArch64/vector-broadcast.ll
new file mode 100644
index 0000000000000..b5a019bd9d5bf
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/vector-broadcast.ll
@@ -0,0 +1,68 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=aarch64-linux-gnu < %s | FileCheck %s
+
+; See sve-vector-broadcast.ll for scalable types.
+
+define <16 x i8> @broadcast_v8i8_to_v16i8(<8 x i8> %vec) {
+; CHECK-LABEL: broadcast_v8i8_to_v16i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: ret
+ %result = call <16 x i8> @llvm.vector.broadcast.v16i8.v8i8(<8 x i8> %vec)
+ ret <16 x i8> %result
+}
+
+define <16 x i8> @broadcast_v4i8_to_v16i8(<4 x i16> %wide.vec) {
+; CHECK-LABEL: broadcast_v4i8_to_v16i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: uzp1 v0.16b, v0.16b, v0.16b
+; CHECK-NEXT: ret
+ %vec = trunc <4 x i16> %wide.vec to <4 x i8>
+ %result = call <16 x i8> @llvm.vector.broadcast.v16i8.v4i8(<4 x i8> %vec)
+ ret <16 x i8> %result
+}
+
+define <8 x i16> @broadcast_v4i16_to_v8i16(<4 x i16> %vec) {
+; CHECK-LABEL: broadcast_v4i16_to_v8i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: ret
+ %result = call <8 x i16> @llvm.vector.broadcast.v8i16.v4i16(<4 x i16> %vec)
+ ret <8 x i16> %result
+}
+
+define <8 x i16> @broadcast_v2i16_to_v8i16(<2 x i32> %wide.vec) {
+; CHECK-LABEL: broadcast_v2i16_to_v8i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: uzp1 v0.8h, v0.8h, v0.8h
+; CHECK-NEXT: ret
+ %vec = trunc <2 x i32> %wide.vec to <2 x i16>
+ %result = call <8 x i16> @llvm.vector.broadcast.v8i16.v2i16(<2 x i16> %vec)
+ ret <8 x i16> %result
+}
+
+define <4 x i32> @broadcast_v2i32_to_v4i32(<2 x i32> %vec) {
+; CHECK-LABEL: broadcast_v2i32_to_v4i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: ret
+ %result = call <4 x i32> @llvm.vector.broadcast.v4i32.v2i32(<2 x i32> %vec)
+ ret <4 x i32> %result
+}
+
+define <4 x float> @broadcast_v2f32_to_v4f32(<2 x float> %vec) {
+; CHECK-LABEL: broadcast_v2f32_to_v4f32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: ret
+ %result = call <4 x float> @llvm.vector.broadcast.v4f32.v2f32(<2 x float> %vec)
+ ret <4 x float> %result
+}
>From 14e30d68a4b8508cc10ef6ebd3d3744c2245d769 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Fri, 7 Aug 2026 10:19:39 +0000
Subject: [PATCH 04/19] Update LangRef
---
llvm/docs/LangRef.md | 26 ++++++++++++++++++++++++++
1 file changed, 26 insertions(+)
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index a4816d337e02b..a2b290aca1484 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -20798,6 +20798,32 @@ runtime, then the result vector is a {ref}`poison value <poisonvalues>`. The
`idx` parameter must be a vector index constant type (for most targets this
will be an integer pointer type).
+#### '`llvm.vector.broadcast`' Intrinsic
+
+##### Syntax:
+This is an overloaded intrinsic.
+
+```
+declare <8 x i32> @llvm.vector.broadcast.v8i32.v2i32(<2 x i32> %vec)
+declare <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v16i8(<16 x i8> %vec)
+```
+
+##### Overview:
+
+The '`llvm.vector.broadcast.*`' intrinsic repeatedly copies the elements of
+the source vector, in order, until the result vector is filled. For example,
+broadcasting `<A, B>` to a vector with eight elements produces
+`<A, B, A, B, A, B, A, B>`.
+This intrinsic works for both fixed and scalable vectors but the recommended way
+to express this operation for fixed-width vectors is still to use a
+`shufflevector`, as that may allow for more optimization opportunities.
+
+##### Arguments:
+
+The argument and result must be vectors with the same element type. The element
+count of the result must be a known multiple of the element count of the
+argument.
+
#### '`llvm.vector.reverse`' Intrinsic
##### Syntax:
>From e2014ba28d2b5d4668b31ff998bf26e2bd3aff44 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 10 Aug 2026 11:03:21 +0000
Subject: [PATCH 05/19] Support operand widening
---
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 1 +
.../SelectionDAG/LegalizeVectorTypes.cpp | 25 ++++++++++++++++++
.../sve-vector-broadcast-unsupported.ll | 9 -------
.../CodeGen/AArch64/sve-vector-broadcast.ll | 26 +++++++++++++++++++
4 files changed, 52 insertions(+), 9 deletions(-)
delete mode 100644 llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index c1ebbed1ab6fc..53845c83a42d9 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -1103,6 +1103,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
bool WidenVectorOperand(SDNode *N, unsigned OpNo);
SDValue WidenVecOp_BITCAST(SDNode *N);
SDValue WidenVecOp_CONCAT_VECTORS(SDNode *N);
+ SDValue WidenVecOp_VECTOR_BROADCAST(SDNode *N);
SDValue WidenVecOp_EXTEND(SDNode *N);
SDValue WidenVecOp_CMP(SDNode *N);
SDValue WidenVecOp_EXTRACT_VECTOR_ELT(SDNode *N);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 805b150f88b13..e2545e23cd26b 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -7853,6 +7853,9 @@ bool DAGTypeLegalizer::WidenVectorOperand(SDNode *N, unsigned OpNo) {
Res = WidenVecOp_FAKE_USE(N);
break;
case ISD::CONCAT_VECTORS: Res = WidenVecOp_CONCAT_VECTORS(N); break;
+ case ISD::VECTOR_BROADCAST:
+ Res = WidenVecOp_VECTOR_BROADCAST(N);
+ break;
case ISD::INSERT_SUBVECTOR: Res = WidenVecOp_INSERT_SUBVECTOR(N); break;
case ISD::EXTRACT_SUBVECTOR: Res = WidenVecOp_EXTRACT_SUBVECTOR(N); break;
case ISD::EXTRACT_VECTOR_ELT: Res = WidenVecOp_EXTRACT_VECTOR_ELT(N); break;
@@ -8307,6 +8310,28 @@ SDValue DAGTypeLegalizer::WidenVecOp_CONCAT_VECTORS(SDNode *N) {
return DAG.getBuildVector(VT, dl, Ops);
}
+SDValue DAGTypeLegalizer::WidenVecOp_VECTOR_BROADCAST(SDNode *N) {
+ SDLoc DL(N);
+ EVT VT = N->getValueType(0);
+ SDValue Src = N->getOperand(0);
+ EVT SrcVT = Src.getValueType();
+ EVT WidenVT = TLI.getTypeToTransformTo(*DAG.getContext(), SrcVT);
+ assert(WidenVT.getVectorElementCount().isKnownMultipleOf(
+ SrcVT.getVectorElementCount()) &&
+ "Cannot widen VECTOR_BROADCAST operand to an ElementCount that's not "
+ "a multiple of the input ElementCount.");
+ unsigned NumConcat =
+ WidenVT.getVectorMinNumElements() / SrcVT.getVectorMinNumElements();
+
+ // Repeat the original source because the extra lanes of its widened value
+ // are unspecified.
+ SmallVector<SDValue, 8> Ops(NumConcat, Src);
+ SDValue WidenedSrc = DAG.getNode(ISD::CONCAT_VECTORS, DL, WidenVT, Ops);
+ if (VT == WidenVT)
+ return WidenedSrc;
+ return DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, WidenedSrc);
+}
+
SDValue DAGTypeLegalizer::WidenVecOp_INSERT_SUBVECTOR(SDNode *N) {
EVT VT = N->getValueType(0);
SDValue SubVec = N->getOperand(1);
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
deleted file mode 100644
index 49152c702505a..0000000000000
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
+++ /dev/null
@@ -1,9 +0,0 @@
-; RUN: not --crash llc -mtriple=aarch64-linux-gnu -mattr=+sve < %s 2>&1 | FileCheck %s
-
-; CHECK: LLVM ERROR: Do not know how to widen this operator's operand!
-
-; TODO: Support broadcasts from 32-bit vec
-define <vscale x 8 x half> @broadcast_single_f16(<2 x half> %a) {
- %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<2 x half> %a)
- ret <vscale x 8 x half> %out
-}
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
index 7f1a37529f912..7ba31a9c77464 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -138,6 +138,8 @@ define <vscale x 2 x i64> @broadcast_quad_i64(<2 x i64> %a) {
ret <vscale x 2 x i64> %out
}
+; FP / BFP types
+
define <vscale x 8 x half> @broadcast_quad_f16(<8 x half> %a) {
; CHECK-LABEL: broadcast_quad_f16:
; CHECK: // %bb.0:
@@ -168,6 +170,30 @@ define <vscale x 4 x half> @broadcast_double_f16_to_double_sve(<4 x half> %a) {
ret <vscale x 4 x half> %out
}
+define <vscale x 8 x half> @broadcast_2f16_to_nxv8f16(<4 x half> %a) {
+; CHECK-LABEL: broadcast_2f16_to_nxv8f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2s, v0.s[0]
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
+ %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<2 x half> %a.legal)
+ ret <vscale x 8 x half> %out
+}
+
+define <vscale x 4 x half> @broadcast_2f16_to_nxv4f16(<4 x half> %a) {
+; CHECK-LABEL: broadcast_2f16_to_nxv4f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2s, v0.s[0]
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
+ %out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<2 x half> %a.legal)
+ ret <vscale x 4 x half> %out
+}
+
define <vscale x 8 x bfloat> @broadcast_quad_bf16(<8 x bfloat> %a) #0 {
; CHECK-LABEL: broadcast_quad_bf16:
; CHECK: // %bb.0:
>From 3a3d9f5d77cf9d97ee83e124d8c6d718e4495dda Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 10 Aug 2026 12:03:39 +0000
Subject: [PATCH 06/19] Fix AArch64 ISel for unpacked types
Using DUP instructions only really works for packed types, as it will
not introduce the spacing between elements that is required for unpacked
types like nxv2f16.
For those, we'll first broadcast to their corresponding packed type, and
then extract the lo lanes into an unpacked type, effectively introducing
the required spacing.
---
.../Target/AArch64/AArch64ISelLowering.cpp | 22 ++++++++
llvm/lib/Target/AArch64/AArch64ISelLowering.h | 1 +
llvm/lib/Target/AArch64/SVEInstrFormats.td | 42 ++++-----------
.../CodeGen/AArch64/sve-vector-broadcast.ll | 51 +++++++++++++++++++
4 files changed, 85 insertions(+), 31 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 62b423a3c76d6..4511380d17f2f 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -1993,6 +1993,12 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::VECTOR_SPLICE_RIGHT, VT, Custom);
}
+ // Broadcasts to unpacked SVE type require explicit unpacking to add spacing
+ // between elements.
+ for (auto VT : {MVT::nxv2f16, MVT::nxv4f16, MVT::nxv2f32, MVT::nxv2bf16,
+ MVT::nxv4bf16})
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Custom);
+
if (Subtarget->hasSVEB16B16() &&
Subtarget->isNonStreamingSVEorSME2Available()) {
// Note: Use SVE for bfloat16 operations when +sve-b16b16 is available.
@@ -8883,6 +8889,8 @@ SDValue AArch64TargetLowering::LowerOperation(SDValue Op,
return LowerEXTEND_VECTOR_INREG(Op, DAG);
case ISD::ZERO_EXTEND_VECTOR_INREG:
return LowerZERO_EXTEND_VECTOR_INREG(Op, DAG);
+ case ISD::VECTOR_BROADCAST:
+ return LowerVECTOR_BROADCAST(Op, DAG);
case ISD::VECTOR_SHUFFLE:
return LowerVECTOR_SHUFFLE(Op, DAG);
case ISD::SPLAT_VECTOR:
@@ -17829,6 +17837,20 @@ SDValue AArch64TargetLowering::LowerEXTRACT_SUBVECTOR(SDValue Op,
return SDValue();
}
+SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
+ SelectionDAG &DAG) const {
+ SDLoc DL(Op);
+ EVT VT = Op.getValueType();
+ assert(isUnpackedType(VT, DAG) && "Expected an unpacked vector type!");
+
+ // Broadcast into a packed container before extracting the low lanes, which
+ // places the result elements at the spacing required by the unpacked type.
+ EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
+ SDValue Broadcast =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, Op.getOperand(0));
+ return DAG.getExtractSubvector(DL, VT, Broadcast, 0);
+}
+
SDValue AArch64TargetLowering::LowerINSERT_SUBVECTOR(SDValue Op,
SelectionDAG &DAG) const {
assert(Op.getValueType().isScalableVector() &&
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.h b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
index b68e06dede580..99c78316b7367 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.h
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
@@ -765,6 +765,7 @@ class AArch64TargetLowering : public TargetLowering {
SDValue LowerBUILD_VECTOR(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerEXTEND_VECTOR_INREG(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerZERO_EXTEND_VECTOR_INREG(SDValue Op, SelectionDAG &DAG) const;
+ SDValue LowerVECTOR_BROADCAST(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerVECTOR_SHUFFLE(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerSPLAT_VECTOR(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerDUPQLane(SDValue Op, SelectionDAG &DAG) const;
diff --git a/llvm/lib/Target/AArch64/SVEInstrFormats.td b/llvm/lib/Target/AArch64/SVEInstrFormats.td
index 0528eaa854f3e..92213bfca0f73 100644
--- a/llvm/lib/Target/AArch64/SVEInstrFormats.td
+++ b/llvm/lib/Target/AArch64/SVEInstrFormats.td
@@ -1553,21 +1553,12 @@ multiclass sve_int_perm_dup_i<string asm> {
// Duplicate an extracted vector element across a vector.
- foreach VT = [nxv16i8] in {
- def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.Packed ZPR:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) ZPR:$vec, sve_elm_idx_extdup_b:$index)>;
- def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.ZSub V128:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_b:$index)>;
- def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_b:$index)>;
-
- // Broadcast a whole 128-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
- (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
- // Broadcast a whole 64-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
- (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
- }
+ def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (nxv16i8 ZPR:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) ZPR:$vec, sve_elm_idx_extdup_b:$index)>;
+ def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (v16i8 V128:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_b:$index)>;
+ def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (v8i8 V64:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_b:$index)>;
foreach VT = [nxv8i16, nxv2f16, nxv4f16, nxv8f16, nxv2bf16, nxv4bf16, nxv8bf16] in {
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.Packed ZPR:$vec), sve_elm_idx_extdup_h:$index)))),
@@ -1576,13 +1567,6 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _H) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_h:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_h:$index)))),
(!cast<Instruction>(NAME # _H) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_h:$index)>;
-
- // Broadcast a whole 128-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
- (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
- // Broadcast a whole 64-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
- (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
foreach VT = [nxv4i32, nxv2f32, nxv4f32 ] in {
@@ -1592,13 +1576,6 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _S) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_s:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_s:$index)))),
(!cast<Instruction>(NAME # _S) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_s:$index)>;
-
- // Broadcast a whole 128-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
- (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
- // Broadcast a whole 64-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
- (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
foreach VT = [nxv2i64, nxv2f64] in {
@@ -1608,11 +1585,14 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_d:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_d:$index)))),
(!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_d:$index)>;
+ }
- // Broadcast a whole 128-bit vector
+ // Broadcast whole fixed-length vectors to packed SVE types.
+ // See LowerVECTOR_BROADCAST for handling of legal unpacked types.
+ foreach VT = [nxv16i8, nxv8i16, nxv8f16, nxv8bf16,
+ nxv4i32, nxv4f32, nxv2i64, nxv2f64] in {
def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
(!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
- // Broadcast a whole 64-bit vector
def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
(!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
index 7ba31a9c77464..34fbc4638fe4f 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -165,6 +165,7 @@ define <vscale x 4 x half> @broadcast_double_f16_to_double_sve(<4 x half> %a) {
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: ret
%out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<4 x half> %a)
ret <vscale x 4 x half> %out
@@ -188,12 +189,40 @@ define <vscale x 4 x half> @broadcast_2f16_to_nxv4f16(<4 x half> %a) {
; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
; CHECK-NEXT: dup v0.2s, v0.s[0]
; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: ret
%a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
%out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<2 x half> %a.legal)
ret <vscale x 4 x half> %out
}
+define <vscale x 8 x half> @broadcast_2f16_to_nxv8f16_lo(<4 x half> %a) {
+; CHECK-LABEL: broadcast_2f16_to_nxv8f16_lo:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2s, v0.s[0]
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
+ %out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<2 x half> %a.legal)
+ %out.in.lo = call <vscale x 8 x half> @llvm.vector.insert.nxv8f16.nxv4f16(<vscale x 8 x half> poison, <vscale x 4 x half> %out, i64 0)
+ ret <vscale x 8 x half> %out.in.lo
+}
+
+define <vscale x 2 x half> @broadcast_2f16_to_nxv2f16(<4 x half> %a) {
+; CHECK-LABEL: broadcast_2f16_to_nxv2f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2s, v0.s[0]
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uunpklo z0.s, z0.h
+; CHECK-NEXT: uunpklo z0.d, z0.s
+; CHECK-NEXT: ret
+ %a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
+ %out = call <vscale x 2 x half> @llvm.vector.broadcast.nxv2f16(<2 x half> %a.legal)
+ ret <vscale x 2 x half> %out
+}
+
define <vscale x 8 x bfloat> @broadcast_quad_bf16(<8 x bfloat> %a) #0 {
; CHECK-LABEL: broadcast_quad_bf16:
; CHECK: // %bb.0:
@@ -214,6 +243,28 @@ define <vscale x 4 x float> @broadcast_quad_f32(<4 x float> %a) {
ret <vscale x 4 x float> %out
}
+define <vscale x 2 x float> @broadcast_double_f32_to_nxv2f32(<2 x float> %a) {
+; CHECK-LABEL: broadcast_double_f32_to_nxv2f32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uunpklo z0.d, z0.s
+; CHECK-NEXT: ret
+ %out = call <vscale x 2 x float> @llvm.vector.broadcast.nxv2f32.v2f32(<2 x float> %a)
+ ret <vscale x 2 x float> %out
+}
+
+define <vscale x 4 x bfloat> @broadcast_double_bf16_to_nxv4bf16(<4 x bfloat> %a) #0 {
+; CHECK-LABEL: broadcast_double_bf16_to_nxv4bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uunpklo z0.s, z0.h
+; CHECK-NEXT: ret
+ %out = call <vscale x 4 x bfloat> @llvm.vector.broadcast.nxv4bf16.v4bf16(<4 x bfloat> %a)
+ ret <vscale x 4 x bfloat> %out
+}
+
define <vscale x 2 x double> @broadcast_quad_f64(<2 x double> %a) {
; CHECK-LABEL: broadcast_quad_f64:
; CHECK: // %bb.0:
>From 0e157987ba33cc7a9630accbc369e53d9580101c Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 10 Aug 2026 12:33:35 +0000
Subject: [PATCH 07/19] Set Expand as default action for ISD::VECTOR_BROADCAST
---
llvm/lib/CodeGen/TargetLoweringBase.cpp | 4 ++++
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 8 +++++---
2 files changed, 9 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/CodeGen/TargetLoweringBase.cpp b/llvm/lib/CodeGen/TargetLoweringBase.cpp
index f852f7551ce16..5b486c0600db9 100644
--- a/llvm/lib/CodeGen/TargetLoweringBase.cpp
+++ b/llvm/lib/CodeGen/TargetLoweringBase.cpp
@@ -772,6 +772,10 @@ void TargetLoweringBase::initActions() {
llvm::fill(RegClassForVT, nullptr);
llvm::fill(TargetDAGCombineArray, 0);
+ // Targets must explicitly opt into legal VECTOR_BROADCAST nodes.
+ for (MVT VT : MVT::vector_valuetypes())
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Expand);
+
// Let extending atomic loads be unsupported by default.
for (MVT ValVT : MVT::all_valuetypes())
for (MVT MemVT : MVT::all_valuetypes())
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 4511380d17f2f..2dbd0669fb631 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -740,9 +740,6 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::ROTR, VT, Expand);
}
- for (MVT VT : MVT::fixedlen_vector_valuetypes())
- setOperationAction(ISD::VECTOR_BROADCAST, VT, Expand);
-
// AArch64 doesn't have i32 MULH{S|U}.
setOperationAction(ISD::MULHU, MVT::i32, Expand);
setOperationAction(ISD::MULHS, MVT::i32, Expand);
@@ -1993,6 +1990,11 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::VECTOR_SPLICE_RIGHT, VT, Custom);
}
+ // Direct patterns exist for broadcasts to packed SVE types.
+ for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32, MVT::nxv2i64,
+ MVT::nxv8f16, MVT::nxv4f32, MVT::nxv2f64, MVT::nxv8bf16})
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Legal);
+
// Broadcasts to unpacked SVE type require explicit unpacking to add spacing
// between elements.
for (auto VT : {MVT::nxv2f16, MVT::nxv4f16, MVT::nxv2f32, MVT::nxv2bf16,
>From af4e603537a09e9f553358c81fc710faf6560cd0 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Thu, 20 Aug 2026 13:57:57 +0000
Subject: [PATCH 08/19] Allow input min EC to be wider than output min EC with
vscale_range
This relaxes the requirements for vector.broadcast, allowing e.g.
<vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v4i64(<4 x i64> %a)
or
<vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %a)
when vscale is known > 1.
Langref and verifier are updated to reflect the new requirements.
Isel is now updated to properly support legal sources wider than NEON,
which can happen when the minimum vscale value is known > 1.
---
llvm/docs/LangRef.md | 8 ++-
.../lib/CodeGen/SelectionDAG/SelectionDAG.cpp | 12 ----
llvm/lib/IR/Verifier.cpp | 31 +++++++++
.../Target/AArch64/AArch64ISelLowering.cpp | 63 ++++++++++++++++++-
.../CodeGen/AArch64/sve-vector-broadcast.ll | 51 +++++++++++++++
.../vector-broadcast-intrinsic-invalid.ll | 37 +++++++++++
.../Verifier/vector-broadcast-intrinsic.ll | 16 +++++
7 files changed, 201 insertions(+), 17 deletions(-)
create mode 100644 llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
create mode 100644 llvm/test/Verifier/vector-broadcast-intrinsic.ll
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index a2b290aca1484..35b9fb10a4fad 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -20820,9 +20820,11 @@ to express this operation for fixed-width vectors is still to use a
##### Arguments:
-The argument and result must be vectors with the same element type. The element
-count of the result must be a known multiple of the element count of the
-argument.
+The argument and result must be vectors with the same element type. A scalable
+argument cannot be broadcast to a fixed-width result. For every possible value
+of `vscale`, the element count of the result must be a multiple of the element
+count of the argument. A `vscale_range` attribute may be used to establish this
+for a scalable result and fixed-width argument.
#### '`llvm.vector.reverse`' Intrinsic
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
index a2fef38e9d4ca..0a3eb65965826 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
@@ -9185,18 +9185,6 @@ SDValue SelectionDAG::getNode(unsigned Opcode, const SDLoc &DL, EVT VT,
}
break;
}
- case ISD::VECTOR_BROADCAST: {
- [[maybe_unused]] EVT InputVT = N1.getValueType();
- assert(InputVT.isVector() && VT.isVector() &&
- "Expected the input and output of the VECTOR_BROADCAST node to be "
- "vectors!");
- assert(VT.getVectorElementCount().hasKnownScalarFactor(
- InputVT.getVectorElementCount()) &&
- "Expected the element count of the output of the VECTOR_BROADCAST "
- "node to be a positive integer multiple of the element count of the "
- "source operand!");
- break;
- }
case ISD::BITCAST:
// Fold bit_convert nodes from a type to themselves.
if (N1.getValueType() == VT)
diff --git a/llvm/lib/IR/Verifier.cpp b/llvm/lib/IR/Verifier.cpp
index 1440b95896474..6c381d71745da 100644
--- a/llvm/lib/IR/Verifier.cpp
+++ b/llvm/lib/IR/Verifier.cpp
@@ -7057,6 +7057,37 @@ void Verifier::visitIntrinsicCall(Intrinsic::ID ID, CallBase &Call) {
}
break;
}
+ case Intrinsic::vector_broadcast: {
+ auto *ResultTy = cast<VectorType>(Call.getType());
+ auto *ArgTy = cast<VectorType>(Call.getArgOperand(0)->getType());
+ ElementCount ResultEC = ResultTy->getElementCount();
+ ElementCount InputEC = ArgTy->getElementCount();
+
+ Check(ResultTy->getElementType() == ArgTy->getElementType(),
+ "vector_broadcast argument and result must have the same element "
+ "type.",
+ &Call);
+
+ if (InputEC.isScalable() && ResultEC.isFixed()) {
+ CheckFailed("vector_broadcast cannot broadcast a scalable vector to a "
+ "fixed-width vector.",
+ &Call);
+ break;
+ }
+
+ uint64_t MinResultElements = ResultEC.getKnownMinValue();
+ if (ResultEC.isScalable() && InputEC.isFixed()) {
+ Attribute Attr =
+ Call.getFunction()->getFnAttribute(Attribute::VScaleRange);
+ if (Attr.isValid())
+ MinResultElements *= Attr.getVScaleRangeMin();
+ }
+ Check(MinResultElements % InputEC.getKnownMinValue() == 0,
+ "vector_broadcast result element count must be a multiple of the "
+ "argument element count for all possible values of vscale.",
+ &Call);
+ break;
+ }
case Intrinsic::vector_insert: {
Value *Vec = Call.getArgOperand(0);
Value *SubVec = Call.getArgOperand(1);
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 2dbd0669fb631..dccfc9ac821e4 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -1990,10 +1990,17 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::VECTOR_SPLICE_RIGHT, VT, Custom);
}
- // Direct patterns exist for broadcasts to packed SVE types.
+ // Direct patterns exist for broadcasts to packed SVE types whose sources
+ // fit in a NEON register. Custom lowering handles wider fixed sources.
for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32, MVT::nxv2i64,
MVT::nxv8f16, MVT::nxv4f32, MVT::nxv2f64, MVT::nxv8bf16})
- setOperationAction(ISD::VECTOR_BROADCAST, VT, Legal);
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Custom);
+
+ // Custom legalise unpacked types to avoid promoting fixed-length sources
+ // beyond the width supported by NEON.
+ for (auto VT : {MVT::nxv2i8, MVT::nxv2i16, MVT::nxv2i32, MVT::nxv4i8,
+ MVT::nxv4i16, MVT::nxv8i8})
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Custom);
// Broadcasts to unpacked SVE type require explicit unpacking to add spacing
// between elements.
@@ -17843,6 +17850,38 @@ SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
SelectionDAG &DAG) const {
SDLoc DL(Op);
EVT VT = Op.getValueType();
+ assert(VT.isScalableVector() &&
+ "Custom lowering expected for scalable vectors only.");
+
+ if (isPackedVectorType(VT, DAG)) {
+ SDValue Src = Op.getOperand(0);
+ EVT SrcVT = Src.getValueType();
+ if (ElementCount::isKnownGE(VT.getVectorElementCount(),
+ SrcVT.getVectorElementCount()))
+ return Op;
+
+ // We are broadcasting to a packed scalable vector for a wider-than-NEON
+ // vector. Deinterleave smaller source vectors until we get to a quad.
+ // This is similar to SplitVecRes_VECTOR_BROADCAST, but we are past type
+ // legalisation so it cannot be reused.
+ // TODO: tbl might be cheaper? Or or repeated dup z.q[idx] followed by zip1
+ // z.q, but that requires +F64MM+NS
+ assert(SrcVT.isFixedLengthVector() &&
+ SrcVT.getFixedSizeInBits() > AArch64::SVEBitsPerBlock &&
+ "Expected a fixed source for a vscale-dependent broadcast");
+ auto [SrcLo, SrcHi] = DAG.SplitVector(Src, DL);
+ EVT SplitVT = SrcLo.getValueType();
+ SDValue Deinterleaved =
+ DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL,
+ DAG.getVTList(SplitVT, SplitVT), SrcLo, SrcHi);
+ SDValue Even = DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved);
+ SDValue Odd =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
+ SDValue Interleaved = DAG.getNode(ISD::VECTOR_INTERLEAVE, DL,
+ DAG.getVTList(VT, VT), Even, Odd);
+ return Interleaved.getValue(0);
+ }
+
assert(isUnpackedType(VT, DAG) && "Expected an unpacked vector type!");
// Broadcast into a packed container before extracting the low lanes, which
@@ -32838,6 +32877,26 @@ void AArch64TargetLowering::ReplaceNodeResults(
case ISD::BITCAST:
ReplaceBITCASTResults(N, Results, DAG);
return;
+ case ISD::VECTOR_BROADCAST: {
+ EVT VT = N->getValueType(0);
+ EVT PromotedVT = getTypeToTransformTo(*DAG.getContext(), VT);
+ EVT PromotedInVT = N->getOperand(0).getValueType().changeVectorElementType(
+ *DAG.getContext(), PromotedVT.getVectorElementType());
+ if (!PromotedInVT.isFixedLengthVector() ||
+ PromotedInVT.getFixedSizeInBits() <= AArch64::SVEBitsPerBlock)
+ return;
+
+ // Avoid promoting the input vector if that creates a vector wider than
+ // 128bits. Instead, keep the element type and broadcast to its packed type.
+ EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
+ SDLoc DL(N);
+ SDValue Broadcast =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, N->getOperand(0));
+ SDValue Promoted =
+ DAG.getNode(ISD::ANY_EXTEND_VECTOR_INREG, DL, PromotedVT, Broadcast);
+ Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, Promoted));
+ return;
+ }
case ISD::VECREDUCE_ADD:
case ISD::VECREDUCE_SMAX:
case ISD::VECREDUCE_SMIN:
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
index 34fbc4638fe4f..87d0783158bba 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -138,6 +138,57 @@ define <vscale x 2 x i64> @broadcast_quad_i64(<2 x i64> %a) {
ret <vscale x 2 x i64> %out
}
+; vscale_range tests
+
+define <vscale x 2 x i64> @broadcast_v4i32_to_nxv2i32(<4 x i32> %a) vscale_range(2,8) {
+; CHECK-LABEL: broadcast_v4i32_to_nxv2i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: uunpklo z0.d, z0.s
+; CHECK-NEXT: ret
+ %out = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %a)
+ %out.legal = zext <vscale x 2 x i32> %out to <vscale x 2 x i64>
+ ret <vscale x 2 x i64> %out.legal
+}
+
+; wider-than-NEON fixed-length source
+define <vscale x 2 x i64> @broadcast_v4i64_to_nxv2i64(<vscale x 2 x i64> %a.legal) vscale_range(2,8) {
+; CHECK-LABEL: broadcast_v4i64_to_nxv2i64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: movprfx z1, z0
+; CHECK-NEXT: ext z1.b, z1.b, z0.b, #16
+; CHECK-NEXT: uzp2 v2.2d, v0.2d, v1.2d
+; CHECK-NEXT: uzp1 v0.2d, v0.2d, v1.2d
+; CHECK-NEXT: mov z1.q, q2
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: zip1 z0.d, z0.d, z1.d
+; CHECK-NEXT: ret
+ %a = call <4 x i64> @llvm.vector.extract.v4i64.nxv2i64(<vscale x 2 x i64> %a.legal, i64 0)
+ %r = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v4i64(<4 x i64> %a)
+ ret <vscale x 2 x i64> %r
+}
+
+; wider-than-NEON fixed-length source and wide destination
+define <vscale x 4 x i32> @broadcast_v4i64_to_nxv4i64(<vscale x 2 x i64> %a.legal) vscale_range(2,8) {
+; CHECK-LABEL: broadcast_v4i64_to_nxv4i64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: movprfx z1, z0
+; CHECK-NEXT: ext z1.b, z1.b, z0.b, #16
+; CHECK-NEXT: uzp2 v2.2d, v0.2d, v1.2d
+; CHECK-NEXT: uzp1 v0.2d, v0.2d, v1.2d
+; CHECK-NEXT: mov z1.q, q2
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: zip2 z2.d, z0.d, z1.d
+; CHECK-NEXT: zip1 z0.d, z0.d, z1.d
+; CHECK-NEXT: uzp1 z0.s, z0.s, z2.s
+; CHECK-NEXT: ret
+ %a = call <4 x i64> @llvm.vector.extract.v4i64.nxv2i64(<vscale x 2 x i64> %a.legal, i64 0)
+ %r = call <vscale x 4 x i64> @llvm.vector.broadcast.nxv4i64.v4i64(<4 x i64> %a)
+ %r.legal = trunc <vscale x 4 x i64> %r to <vscale x 4 x i32>
+ ret <vscale x 4 x i32> %r.legal
+}
+
; FP / BFP types
define <vscale x 8 x half> @broadcast_quad_f16(<8 x half> %a) {
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll b/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
new file mode 100644
index 0000000000000..ae7be4203316e
--- /dev/null
+++ b/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
@@ -0,0 +1,37 @@
+; RUN: not opt -passes=verify -disable-output < %s 2>&1 | FileCheck %s
+
+; CHECK: vector_broadcast argument and result must have the same element type.
+define <8 x i32> @mismatched_element_types(<2 x i64> %vec) {
+ %result = call <8 x i32> @llvm.vector.broadcast.v8i32.v2i64(<2 x i64> %vec)
+ ret <8 x i32> %result
+}
+
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+define <8 x i32> @non_multiple_fixed(<3 x i32> %vec) {
+ %result = call <8 x i32> @llvm.vector.broadcast.v8i32.v3i32(<3 x i32> %vec)
+ ret <8 x i32> %result
+}
+
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+define <vscale x 8 x i32> @non_multiple_scalable(<vscale x 3 x i32> %vec) {
+ %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.nxv3i32(<vscale x 3 x i32> %vec)
+ ret <vscale x 8 x i32> %result
+}
+
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+define <vscale x 2 x i32> @fixed_to_scalable_without_vscale_range(<4 x i32> %vec) {
+ %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
+ ret <vscale x 2 x i32> %result
+}
+
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+define <vscale x 2 x i32> @fixed_to_scalable_without_sufficient_vscale_range(<8 x i32> %vec) vscale_range(2,8) {
+ %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v8i32(<8 x i32> %vec)
+ ret <vscale x 2 x i32> %result
+}
+
+; CHECK: vector_broadcast cannot broadcast a scalable vector to a fixed-width vector.
+define <8 x i32> @scalable_to_fixed(<vscale x 2 x i32> %vec) {
+ %result = call <8 x i32> @llvm.vector.broadcast.v8i32.nxv2i32(<vscale x 2 x i32> %vec)
+ ret <8 x i32> %result
+}
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic.ll b/llvm/test/Verifier/vector-broadcast-intrinsic.ll
new file mode 100644
index 0000000000000..b9813e67d009d
--- /dev/null
+++ b/llvm/test/Verifier/vector-broadcast-intrinsic.ll
@@ -0,0 +1,16 @@
+; RUN: opt -passes=verify -disable-output < %s
+
+define <8 x i32> @fixed_to_fixed(<2 x i32> %vec) {
+ %result = call <8 x i32> @llvm.vector.broadcast.v8i32.v2i32(<2 x i32> %vec)
+ ret <8 x i32> %result
+}
+
+define <vscale x 8 x i32> @scalable_to_scalable(<vscale x 2 x i32> %vec) {
+ %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.nxv2i32(<vscale x 2 x i32> %vec)
+ ret <vscale x 8 x i32> %result
+}
+
+define <vscale x 2 x i32> @fixed_to_scalable_with_vscale_range(<4 x i32> %vec) vscale_range(2, 8) {
+ %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
+ ret <vscale x 2 x i32> %result
+}
>From d4b0ff1e5ee3c58dcacb3f31e38baa08194b96ac Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Fri, 21 Aug 2026 14:41:25 +0000
Subject: [PATCH 09/19] Do not require vscale_range
If the runtime EC of the output is smaller than the input, then the
result in poison.
---
llvm/docs/LangRef.md | 7 +--
llvm/include/llvm/CodeGen/ISDOpcodes.h | 3 +-
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 1 +
.../SelectionDAG/LegalizeVectorTypes.cpp | 42 +++++++++-----
llvm/lib/IR/Verifier.cpp | 19 +++----
.../CodeGen/AArch64/sve-vector-broadcast.ll | 56 +++++++++++++++++++
.../vector-broadcast-intrinsic-invalid.ll | 16 +-----
.../Verifier/vector-broadcast-intrinsic.ll | 10 ++++
8 files changed, 110 insertions(+), 44 deletions(-)
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index 35b9fb10a4fad..4f82592b27683 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -20821,10 +20821,9 @@ to express this operation for fixed-width vectors is still to use a
##### Arguments:
The argument and result must be vectors with the same element type. A scalable
-argument cannot be broadcast to a fixed-width result. For every possible value
-of `vscale`, the element count of the result must be a multiple of the element
-count of the argument. A `vscale_range` attribute may be used to establish this
-for a scalable result and fixed-width argument.
+argument cannot be broadcast to a fixed-width result. At runtime, the element
+count of the result must be a multiple of the element count of the argument.
+Otherwise, the results is a {ref}`poison value <poisonvalues>`.
#### '`llvm.vector.reverse`' Intrinsic
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index fa215c4ce262f..e6f2e5769d4ec 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -639,7 +639,8 @@ enum NodeType {
/// VECTOR_BROADCAST(SRC_SUBVEC)
/// Duplicate a vector in a larger vector. The element count of the result
- /// type is expected to be a multiple of the input vector's.
+ /// type is expected to be a multiple of the input vector's at runtime. If it
+ /// is not, the result is poison.
VECTOR_BROADCAST,
/// VECTOR_REVERSE(VECTOR) - Returns a vector, of the same type as VECTOR,
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index 53845c83a42d9..4783110c9ebb7 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -979,6 +979,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue SplitVecOp_UnaryOp(SDNode *N);
SDValue SplitVecOp_TruncateHelper(SDNode *N);
SDValue SplitVecOp_VECTOR_COMPRESS(SDNode *N, unsigned OpNo);
+ SDValue SplitVecOp_VECTOR_BROADCAST(SDNode *N);
SDValue SplitVecOp_BITCAST(SDNode *N);
SDValue SplitVecOp_INSERT_SUBVECTOR(SDNode *N, unsigned OpNo);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index e2545e23cd26b..2d11283cdb010 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -3488,6 +3488,19 @@ void DAGTypeLegalizer::SplitVecRes_FP_TO_XINT_SAT(SDNode *N, SDValue &Lo,
Hi = DAG.getNode(N->getOpcode(), dl, DstVTHi, SrcHi, N->getOperand(1));
}
+static SDValue buildSplitVectorBroadcast(SelectionDAG &DAG, SDLoc DL, EVT VT,
+ SDValue SrcLo, SDValue SrcHi) {
+ EVT SrcVT = SrcLo.getValueType();
+ SDValue Deinterleaved = DAG.getNode(
+ ISD::VECTOR_DEINTERLEAVE, DL, DAG.getVTList(SrcVT, SrcVT), SrcLo, SrcHi);
+ SDValue Even =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(0));
+ SDValue Odd =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
+ return DAG.getNode(ISD::VECTOR_INTERLEAVE, DL, DAG.getVTList(VT, VT), Even,
+ Odd);
+}
+
void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
SDValue &Hi) {
EVT VT = N->getValueType(0);
@@ -3503,30 +3516,21 @@ void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
return;
}
- // Second case: Src is known to be wider than LoVT.
+ // Second case: the split result is known to be at least as wide as Src.
SDLoc DL(N);
if (LoVT.getVectorMinNumElements() >= SrcVT.getVectorMinNumElements()) {
Lo = Hi = DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Src);
return;
}
- // Final case: VT is scalable and has the same minimum EC.
+ // Final case: VT is scalable and Src is a wider fixed-length vector.
// Use smaller even/odd source vectors so their broadcasts can be
// reinterleaved in the original lane order for every value of vscale.
- assert(VT.getVectorMinNumElements() == SrcVT.getVectorMinNumElements() &&
- VT.isScalableVector());
+ assert(VT.isScalableVector() && SrcVT.isFixedLengthVector() &&
+ "Expected a fixed source wider than the split scalable result");
SDValue SrcLo, SrcHi;
std::tie(SrcLo, SrcHi) = DAG.SplitVector(Src, DL);
- EVT SrcSplitVT = SrcLo.getValueType();
- SDValue Deinterleaved =
- DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL,
- DAG.getVTList(SrcSplitVT, SrcSplitVT), SrcLo, SrcHi);
- SDValue Even =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(0));
- SDValue Odd =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(1));
- SDValue Interleaved = DAG.getNode(ISD::VECTOR_INTERLEAVE, DL,
- DAG.getVTList(LoVT, HiVT), Even, Odd);
+ SDValue Interleaved = buildSplitVectorBroadcast(DAG, DL, LoVT, SrcLo, SrcHi);
Lo = Interleaved.getValue(0);
Hi = Interleaved.getValue(1);
}
@@ -3848,6 +3852,9 @@ bool DAGTypeLegalizer::SplitVectorOperand(SDNode *N, unsigned OpNo) {
case ISD::INSERT_SUBVECTOR: Res = SplitVecOp_INSERT_SUBVECTOR(N, OpNo); break;
case ISD::EXTRACT_VECTOR_ELT:Res = SplitVecOp_EXTRACT_VECTOR_ELT(N); break;
case ISD::CONCAT_VECTORS: Res = SplitVecOp_CONCAT_VECTORS(N); break;
+ case ISD::VECTOR_BROADCAST:
+ Res = SplitVecOp_VECTOR_BROADCAST(N);
+ break;
case ISD::VECTOR_FIND_LAST_ACTIVE:
Res = SplitVecOp_VECTOR_FIND_LAST_ACTIVE(N);
break;
@@ -4027,6 +4034,13 @@ bool DAGTypeLegalizer::SplitVectorOperand(SDNode *N, unsigned OpNo) {
return false;
}
+SDValue DAGTypeLegalizer::SplitVecOp_VECTOR_BROADCAST(SDNode *N) {
+ SDLoc DL(N);
+ SDValue SrcLo, SrcHi;
+ GetSplitVector(N->getOperand(0), SrcLo, SrcHi);
+ return buildSplitVectorBroadcast(DAG, DL, N->getValueType(0), SrcLo, SrcHi);
+}
+
SDValue DAGTypeLegalizer::SplitVecOp_VECTOR_FIND_LAST_ACTIVE(SDNode *N) {
SDLoc DL(N);
diff --git a/llvm/lib/IR/Verifier.cpp b/llvm/lib/IR/Verifier.cpp
index 6c381d71745da..af830c69e01ec 100644
--- a/llvm/lib/IR/Verifier.cpp
+++ b/llvm/lib/IR/Verifier.cpp
@@ -7075,17 +7075,14 @@ void Verifier::visitIntrinsicCall(Intrinsic::ID ID, CallBase &Call) {
break;
}
- uint64_t MinResultElements = ResultEC.getKnownMinValue();
- if (ResultEC.isScalable() && InputEC.isFixed()) {
- Attribute Attr =
- Call.getFunction()->getFnAttribute(Attribute::VScaleRange);
- if (Attr.isValid())
- MinResultElements *= Attr.getVScaleRangeMin();
- }
- Check(MinResultElements % InputEC.getKnownMinValue() == 0,
- "vector_broadcast result element count must be a multiple of the "
- "argument element count for all possible values of vscale.",
- &Call);
+ // We can only compare element counts when the types are both scalable or
+ // non-scalable.
+ if (ResultEC.isScalable() == InputEC.isScalable()) {
+ Check(ResultEC.isKnownMultipleOf(InputEC),
+ "vector_broadcast result element count must be a multiple of the "
+ "argument element count.",
+ &Call);
+ }
break;
}
case Intrinsic::vector_insert: {
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
index 87d0783158bba..f4273b90ce3a1 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -189,6 +189,62 @@ define <vscale x 4 x i32> @broadcast_v4i64_to_nxv4i64(<vscale x 2 x i64> %a.lega
ret <vscale x 4 x i32> %r.legal
}
+define <vscale x 2 x i64> @broadcast_v4i32_to_nxv2i32_no_vscale_range(<4 x i32> %a) {
+; CHECK-LABEL: broadcast_v4i32_to_nxv2i32_no_vscale_range:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: uunpklo z0.d, z0.s
+; CHECK-NEXT: ret
+ %out = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %a)
+ %out.legal = zext <vscale x 2 x i32> %out to <vscale x 2 x i64>
+ ret <vscale x 2 x i64> %out.legal
+}
+
+define <vscale x 2 x i64> @broadcast_v4i64_to_nxv2i64_no_vscale_range(ptr %ptr) {
+; CHECK-LABEL: broadcast_v4i64_to_nxv2i64_no_vscale_range:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ldp q1, q0, [x0]
+; CHECK-NEXT: uzp2 v2.2d, v1.2d, v0.2d
+; CHECK-NEXT: uzp1 v0.2d, v1.2d, v0.2d
+; CHECK-NEXT: mov z1.q, q2
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: zip1 z0.d, z0.d, z1.d
+; CHECK-NEXT: ret
+ %a = load <4 x i64>, ptr %ptr, align 8
+ %out = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v4i64(<4 x i64> %a)
+ ret <vscale x 2 x i64> %out
+}
+
+define <vscale x 4 x i32> @broadcast_v8i64_to_nxv4i64_no_vscale_range(ptr %ptr) {
+; CHECK-LABEL: broadcast_v8i64_to_nxv4i64_no_vscale_range:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ldp q1, q0, [x0]
+; CHECK-NEXT: ldp q3, q2, [x0, #32]
+; CHECK-NEXT: uzp2 v5.2d, v1.2d, v0.2d
+; CHECK-NEXT: uzp1 v0.2d, v1.2d, v0.2d
+; CHECK-NEXT: uzp2 v4.2d, v3.2d, v2.2d
+; CHECK-NEXT: uzp1 v2.2d, v3.2d, v2.2d
+; CHECK-NEXT: uzp2 v1.2d, v5.2d, v4.2d
+; CHECK-NEXT: uzp1 v3.2d, v5.2d, v4.2d
+; CHECK-NEXT: uzp2 v4.2d, v0.2d, v2.2d
+; CHECK-NEXT: uzp1 v0.2d, v0.2d, v2.2d
+; CHECK-NEXT: mov z1.q, q1
+; CHECK-NEXT: mov z2.q, q3
+; CHECK-NEXT: mov z3.q, q4
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: zip1 z1.d, z2.d, z1.d
+; CHECK-NEXT: zip1 z0.d, z0.d, z3.d
+; CHECK-NEXT: zip2 z2.d, z0.d, z1.d
+; CHECK-NEXT: zip1 z0.d, z0.d, z1.d
+; CHECK-NEXT: uzp1 z0.s, z0.s, z2.s
+; CHECK-NEXT: ret
+ %a = load <8 x i64>, ptr %ptr, align 8
+ %out = call <vscale x 4 x i64> @llvm.vector.broadcast.nxv4i64.v8i64(<8 x i64> %a)
+ %out.legal = trunc <vscale x 4 x i64> %out to <vscale x 4 x i32>
+ ret <vscale x 4 x i32> %out.legal
+}
+
; FP / BFP types
define <vscale x 8 x half> @broadcast_quad_f16(<8 x half> %a) {
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll b/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
index ae7be4203316e..f62c0b53bebe0 100644
--- a/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
+++ b/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
@@ -6,30 +6,18 @@ define <8 x i32> @mismatched_element_types(<2 x i64> %vec) {
ret <8 x i32> %result
}
-; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count.
define <8 x i32> @non_multiple_fixed(<3 x i32> %vec) {
%result = call <8 x i32> @llvm.vector.broadcast.v8i32.v3i32(<3 x i32> %vec)
ret <8 x i32> %result
}
-; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count.
define <vscale x 8 x i32> @non_multiple_scalable(<vscale x 3 x i32> %vec) {
%result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.nxv3i32(<vscale x 3 x i32> %vec)
ret <vscale x 8 x i32> %result
}
-; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
-define <vscale x 2 x i32> @fixed_to_scalable_without_vscale_range(<4 x i32> %vec) {
- %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
- ret <vscale x 2 x i32> %result
-}
-
-; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
-define <vscale x 2 x i32> @fixed_to_scalable_without_sufficient_vscale_range(<8 x i32> %vec) vscale_range(2,8) {
- %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v8i32(<8 x i32> %vec)
- ret <vscale x 2 x i32> %result
-}
-
; CHECK: vector_broadcast cannot broadcast a scalable vector to a fixed-width vector.
define <8 x i32> @scalable_to_fixed(<vscale x 2 x i32> %vec) {
%result = call <8 x i32> @llvm.vector.broadcast.v8i32.nxv2i32(<vscale x 2 x i32> %vec)
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic.ll b/llvm/test/Verifier/vector-broadcast-intrinsic.ll
index b9813e67d009d..1f7434ec116c6 100644
--- a/llvm/test/Verifier/vector-broadcast-intrinsic.ll
+++ b/llvm/test/Verifier/vector-broadcast-intrinsic.ll
@@ -14,3 +14,13 @@ define <vscale x 2 x i32> @fixed_to_scalable_with_vscale_range(<4 x i32> %vec) v
%result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
ret <vscale x 2 x i32> %result
}
+
+define <vscale x 2 x i32> @fixed_to_scalable_without_vscale_range(<4 x i32> %vec) {
+ %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
+ ret <vscale x 2 x i32> %result
+}
+
+define <vscale x 2 x i32> @fixed_to_scalable_without_sufficient_vscale_range(<8 x i32> %vec) vscale_range(2, 8) {
+ %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v8i32(<8 x i32> %vec)
+ ret <vscale x 2 x i32> %result
+}
>From 03d14d2bf777950f6d82ca9fe6ad00c3e13198d4 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 24 Aug 2026 10:02:19 +0000
Subject: [PATCH 10/19] Reduce indent level and comment about legal operation
---
.../Target/AArch64/AArch64ISelLowering.cpp | 77 ++++++++++---------
1 file changed, 41 insertions(+), 36 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index dccfc9ac821e4..441957245a5d0 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -17853,43 +17853,48 @@ SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
assert(VT.isScalableVector() &&
"Custom lowering expected for scalable vectors only.");
- if (isPackedVectorType(VT, DAG)) {
- SDValue Src = Op.getOperand(0);
- EVT SrcVT = Src.getValueType();
- if (ElementCount::isKnownGE(VT.getVectorElementCount(),
- SrcVT.getVectorElementCount()))
- return Op;
+ if (isUnpackedType(VT, DAG)) { // Broadcast into a packed container before
+ // extracting the low lanes, which
+ // places the result elements at the spacing required by the unpacked type.
+ EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
+ SDValue Broadcast =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, Op.getOperand(0));
+ return DAG.getExtractSubvector(DL, VT, Broadcast, 0);
+ }
- // We are broadcasting to a packed scalable vector for a wider-than-NEON
- // vector. Deinterleave smaller source vectors until we get to a quad.
- // This is similar to SplitVecRes_VECTOR_BROADCAST, but we are past type
- // legalisation so it cannot be reused.
- // TODO: tbl might be cheaper? Or or repeated dup z.q[idx] followed by zip1
- // z.q, but that requires +F64MM+NS
- assert(SrcVT.isFixedLengthVector() &&
- SrcVT.getFixedSizeInBits() > AArch64::SVEBitsPerBlock &&
- "Expected a fixed source for a vscale-dependent broadcast");
- auto [SrcLo, SrcHi] = DAG.SplitVector(Src, DL);
- EVT SplitVT = SrcLo.getValueType();
- SDValue Deinterleaved =
- DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL,
- DAG.getVTList(SplitVT, SplitVT), SrcLo, SrcHi);
- SDValue Even = DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved);
- SDValue Odd =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
- SDValue Interleaved = DAG.getNode(ISD::VECTOR_INTERLEAVE, DL,
- DAG.getVTList(VT, VT), Even, Odd);
- return Interleaved.getValue(0);
- }
-
- assert(isUnpackedType(VT, DAG) && "Expected an unpacked vector type!");
-
- // Broadcast into a packed container before extracting the low lanes, which
- // places the result elements at the spacing required by the unpacked type.
- EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
- SDValue Broadcast =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, Op.getOperand(0));
- return DAG.getExtractSubvector(DL, VT, Broadcast, 0);
+ assert(isPackedVectorType(VT, DAG) && "Expected a packed vector type!");
+ SDValue Src = Op.getOperand(0);
+ EVT SrcVT = Src.getValueType();
+
+ // We are broadcasting a NEON-sized vector to a packed (legal) SVE type.
+ // This is a legal operation and there are patterns for it.
+ if (ElementCount::isKnownGE(VT.getVectorElementCount(),
+ SrcVT.getVectorElementCount())) {
+ assert(SrcVT.getFixedSizeInBits() <= AArch64::SVEBitsPerBlock &&
+ "Expected NEON-sized source");
+ return Op;
+ }
+
+ // We are broadcasting to a packed scalable vector for a wider-than-NEON
+ // vector. Deinterleave smaller source vectors until we get to a quad.
+ // This is similar to SplitVecRes_VECTOR_BROADCAST, but we are past type
+ // legalisation so it cannot be reused.
+ // TODO: tbl might be cheaper? Or repeated dup z.q[idx] followed by zip1
+ // z.q, but that requires +F64MM+NS
+ assert(SrcVT.isFixedLengthVector() &&
+ SrcVT.getFixedSizeInBits() > AArch64::SVEBitsPerBlock &&
+ "Expected a fixed source for a vscale-dependent broadcast");
+ auto [SrcLo, SrcHi] = DAG.SplitVector(Src, DL);
+ EVT SplitVT = SrcLo.getValueType();
+ SDValue Deinterleaved =
+ DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL, DAG.getVTList(SplitVT, SplitVT),
+ SrcLo, SrcHi);
+ SDValue Even = DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved);
+ SDValue Odd =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
+ SDValue Interleaved =
+ DAG.getNode(ISD::VECTOR_INTERLEAVE, DL, DAG.getVTList(VT, VT), Even, Odd);
+ return Interleaved.getValue(0);
}
SDValue AArch64TargetLowering::LowerINSERT_SUBVECTOR(SDValue Op,
>From 42de3e11704f8768cb701546d7910d4976708559 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 24 Aug 2026 10:31:31 +0000
Subject: [PATCH 11/19] Tidy up buildSplitVectorBroadcast
This is used in two cases:
- Split the source operand, but keeping the output VT as is
- Split the output VT as well as the source VT
---
.../SelectionDAG/LegalizeVectorTypes.cpp | 29 ++++++++++++-------
1 file changed, 19 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 2d11283cdb010..d08b53bf82903 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -3488,8 +3488,14 @@ void DAGTypeLegalizer::SplitVecRes_FP_TO_XINT_SAT(SDNode *N, SDValue &Lo,
Hi = DAG.getNode(N->getOpcode(), dl, DstVTHi, SrcHi, N->getOperand(1));
}
-static SDValue buildSplitVectorBroadcast(SelectionDAG &DAG, SDLoc DL, EVT VT,
- SDValue SrcLo, SDValue SrcHi) {
+/// For two Lo/Hi source halves, perform a broadcast for each to VT and
+/// re-interleave the elements so that the result is equivalent to:
+/// Dst: DoubleWidthVT = VECTOR_BROADCAST(VECTOR_CONCAT(SrcLo, SrcHi))
+/// DstLo,DstHi: VT,VT = split(Dst)
+static std::pair<SDValue, SDValue> buildSplitVectorBroadcast(SelectionDAG &DAG,
+ SDLoc DL, EVT VT,
+ SDValue SrcLo,
+ SDValue SrcHi) {
EVT SrcVT = SrcLo.getValueType();
SDValue Deinterleaved = DAG.getNode(
ISD::VECTOR_DEINTERLEAVE, DL, DAG.getVTList(SrcVT, SrcVT), SrcLo, SrcHi);
@@ -3497,8 +3503,9 @@ static SDValue buildSplitVectorBroadcast(SelectionDAG &DAG, SDLoc DL, EVT VT,
DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(0));
SDValue Odd =
DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
- return DAG.getNode(ISD::VECTOR_INTERLEAVE, DL, DAG.getVTList(VT, VT), Even,
- Odd);
+ SDValue Interleaved =
+ DAG.getNode(ISD::VECTOR_INTERLEAVE, DL, DAG.getVTList(VT, VT), Even, Odd);
+ return std::make_pair(Interleaved.getValue(0), Interleaved.getValue(1));
}
void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
@@ -3528,11 +3535,8 @@ void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
// reinterleaved in the original lane order for every value of vscale.
assert(VT.isScalableVector() && SrcVT.isFixedLengthVector() &&
"Expected a fixed source wider than the split scalable result");
- SDValue SrcLo, SrcHi;
- std::tie(SrcLo, SrcHi) = DAG.SplitVector(Src, DL);
- SDValue Interleaved = buildSplitVectorBroadcast(DAG, DL, LoVT, SrcLo, SrcHi);
- Lo = Interleaved.getValue(0);
- Hi = Interleaved.getValue(1);
+ auto [SrcLo, SrcHi] = DAG.SplitVector(Src, DL);
+ std::tie(Lo, Hi) = buildSplitVectorBroadcast(DAG, DL, LoVT, SrcLo, SrcHi);
}
void DAGTypeLegalizer::SplitVecRes_VECTOR_REVERSE(SDNode *N, SDValue &Lo,
@@ -4038,7 +4042,12 @@ SDValue DAGTypeLegalizer::SplitVecOp_VECTOR_BROADCAST(SDNode *N) {
SDLoc DL(N);
SDValue SrcLo, SrcHi;
GetSplitVector(N->getOperand(0), SrcLo, SrcHi);
- return buildSplitVectorBroadcast(DAG, DL, N->getValueType(0), SrcLo, SrcHi);
+
+ // We are only splitting SrcVT, not the destiniation type, which remains VT.
+ // buildSplitVectorBroadcast produces (VT,VT) so only Lo is needed.
+ EVT VT = N->getValueType(0);
+ auto [Lo, _] = buildSplitVectorBroadcast(DAG, DL, VT, SrcLo, SrcHi);
+ return Lo;
}
SDValue DAGTypeLegalizer::SplitVecOp_VECTOR_FIND_LAST_ACTIVE(SDNode *N) {
>From cd0b2b0738501f0397206d493d66a9ce87a22e04 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 24 Aug 2026 10:50:12 +0000
Subject: [PATCH 12/19] Fix comments placement
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 8 ++++----
1 file changed, 4 insertions(+), 4 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 441957245a5d0..31ff0e6af7cf0 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -17853,8 +17853,8 @@ SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
assert(VT.isScalableVector() &&
"Custom lowering expected for scalable vectors only.");
- if (isUnpackedType(VT, DAG)) { // Broadcast into a packed container before
- // extracting the low lanes, which
+ if (isUnpackedType(VT, DAG)) {
+ // Broadcast into a packed container before extracting the low lanes, which
// places the result elements at the spacing required by the unpacked type.
EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
SDValue Broadcast =
@@ -17866,10 +17866,10 @@ SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
SDValue Src = Op.getOperand(0);
EVT SrcVT = Src.getValueType();
- // We are broadcasting a NEON-sized vector to a packed (legal) SVE type.
- // This is a legal operation and there are patterns for it.
if (ElementCount::isKnownGE(VT.getVectorElementCount(),
SrcVT.getVectorElementCount())) {
+ // We are broadcasting a NEON-sized vector to a packed (legal) SVE type.
+ // This is a legal operation and there are patterns for it.
assert(SrcVT.getFixedSizeInBits() <= AArch64::SVEBitsPerBlock &&
"Expected NEON-sized source");
return Op;
>From 4e95d5fd5761bdfc9011a30bedbcd7f55221db1b Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Thu, 27 Aug 2026 15:56:14 +0000
Subject: [PATCH 13/19] Restrict intrinsic to fixed->scalable with same min EC
This simplifies a lot of the legalisation code. The codegen regressions
for 64b -> scalable 128b can be addressed with a DAG combiner in a
follow-up.
---
llvm/docs/LangRef.md | 20 +-
llvm/include/llvm/CodeGen/ISDOpcodes.h | 5 +-
llvm/include/llvm/IR/Intrinsics.td | 3 +-
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 19 --
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 2 -
.../SelectionDAG/LegalizeVectorTypes.cpp | 82 ++----
llvm/lib/CodeGen/TargetLoweringBase.cpp | 4 -
llvm/lib/IR/Verifier.cpp | 27 +-
.../Target/AArch64/AArch64ISelLowering.cpp | 87 ++----
.../CodeGen/AArch64/sve-vector-broadcast.ll | 248 ++++--------------
llvm/test/CodeGen/AArch64/vector-broadcast.ll | 68 -----
.../vector-broadcast-intrinsic-invalid.ll | 26 +-
.../Verifier/vector-broadcast-intrinsic.ll | 26 +-
13 files changed, 127 insertions(+), 490 deletions(-)
delete mode 100644 llvm/test/CodeGen/AArch64/vector-broadcast.ll
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index 4f82592b27683..bfa60f75cd47d 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -20804,26 +20804,22 @@ will be an integer pointer type).
This is an overloaded intrinsic.
```
-declare <8 x i32> @llvm.vector.broadcast.v8i32.v2i32(<2 x i32> %vec)
declare <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v16i8(<16 x i8> %vec)
```
##### Overview:
-The '`llvm.vector.broadcast.*`' intrinsic repeatedly copies the elements of
-the source vector, in order, until the result vector is filled. For example,
-broadcasting `<A, B>` to a vector with eight elements produces
-`<A, B, A, B, A, B, A, B>`.
-This intrinsic works for both fixed and scalable vectors but the recommended way
-to express this operation for fixed-width vectors is still to use a
-`shufflevector`, as that may allow for more optimization opportunities.
+The '`llvm.vector.broadcast.*`' intrinsic repeatedly copies the elements of the
+source fixed-length vector, in order, until the result scalable vector is
+filled. For example, broadcasting `<A, B>` produces a scalable vector containing
+`vscale` copies of `<A, B>`.
##### Arguments:
-The argument and result must be vectors with the same element type. A scalable
-argument cannot be broadcast to a fixed-width result. At runtime, the element
-count of the result must be a multiple of the element count of the argument.
-Otherwise, the results is a {ref}`poison value <poisonvalues>`.
+The argument must be a fixed-length vector and the result must be a scalable
+vector with the same element type and minimum element count. In other words,
+the result type is formed by adding `vscale x` in front of the argument type's
+element count.
#### '`llvm.vector.reverse`' Intrinsic
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index e6f2e5769d4ec..f9fc070cd643a 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -638,9 +638,8 @@ enum NodeType {
VECTOR_INTERLEAVE,
/// VECTOR_BROADCAST(SRC_SUBVEC)
- /// Duplicate a vector in a larger vector. The element count of the result
- /// type is expected to be a multiple of the input vector's at runtime. If it
- /// is not, the result is poison.
+ /// Duplicate a fixed-length vector in a scalable vector with the same minimum
+ /// element count.
VECTOR_BROADCAST,
/// VECTOR_REVERSE(VECTOR) - Returns a vector, of the same type as VECTOR,
diff --git a/llvm/include/llvm/IR/Intrinsics.td b/llvm/include/llvm/IR/Intrinsics.td
index 36499fcb560d8..ba1e2c5b6170e 100644
--- a/llvm/include/llvm/IR/Intrinsics.td
+++ b/llvm/include/llvm/IR/Intrinsics.td
@@ -2817,8 +2817,7 @@ foreach n = 2...8 in {
}
// vector_broadcast( SrcVector )
-// Broadcast a vector in a larger one.
-// This is the equivalent of a vector splat for vector types.
+// Broadcast a fixed-length vector to its "vscale x" equivalent.
def int_vector_broadcast : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
[llvm_anyvector_ty],
[IntrNoMem, IntrSpeculatable]>;
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index 757453f069b33..bd9c47806a9ca 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -3690,25 +3690,6 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
case ISD::INSERT_VECTOR_ELT:
Results.push_back(ExpandINSERT_VECTOR_ELT(SDValue(Node, 0)));
break;
- case ISD::VECTOR_BROADCAST: {
- EVT VT = Node->getValueType(0);
- EVT SrcVT = Node->getOperand(0).getValueType();
- assert(VT.isFixedLengthVector() && SrcVT.isFixedLengthVector() &&
- "Can only expand broadcasts of fixed-length vectors");
-
- SDValue Src = Node->getOperand(0);
- SDValue PaddedSrc = DAG.getInsertSubvector(dl, DAG.getUNDEF(VT), Src, 0);
-
- // Create a shuffle mask that duplicates Src.
- unsigned NumElts = VT.getVectorNumElements();
- unsigned SrcNumElts = SrcVT.getVectorNumElements();
- SmallVector<int, 8> Mask;
- for (unsigned I = 0; I != NumElts; ++I)
- Mask.push_back(I % SrcNumElts);
- Results.push_back(
- DAG.getVectorShuffle(VT, dl, PaddedSrc, DAG.getUNDEF(VT), Mask));
- break;
- }
case ISD::VECTOR_SHUFFLE: {
SmallVector<int, 32> NewMask;
ArrayRef<int> Mask = cast<ShuffleVectorSDNode>(Node)->getMask();
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index 4783110c9ebb7..cd7f4a733220f 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -979,8 +979,6 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue SplitVecOp_UnaryOp(SDNode *N);
SDValue SplitVecOp_TruncateHelper(SDNode *N);
SDValue SplitVecOp_VECTOR_COMPRESS(SDNode *N, unsigned OpNo);
- SDValue SplitVecOp_VECTOR_BROADCAST(SDNode *N);
-
SDValue SplitVecOp_BITCAST(SDNode *N);
SDValue SplitVecOp_INSERT_SUBVECTOR(SDNode *N, unsigned OpNo);
SDValue SplitVecOp_EXTRACT_SUBVECTOR(SDNode *N);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index d08b53bf82903..3644e2ac87930 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -3488,55 +3488,30 @@ void DAGTypeLegalizer::SplitVecRes_FP_TO_XINT_SAT(SDNode *N, SDValue &Lo,
Hi = DAG.getNode(N->getOpcode(), dl, DstVTHi, SrcHi, N->getOperand(1));
}
-/// For two Lo/Hi source halves, perform a broadcast for each to VT and
-/// re-interleave the elements so that the result is equivalent to:
-/// Dst: DoubleWidthVT = VECTOR_BROADCAST(VECTOR_CONCAT(SrcLo, SrcHi))
-/// DstLo,DstHi: VT,VT = split(Dst)
-static std::pair<SDValue, SDValue> buildSplitVectorBroadcast(SelectionDAG &DAG,
- SDLoc DL, EVT VT,
- SDValue SrcLo,
- SDValue SrcHi) {
- EVT SrcVT = SrcLo.getValueType();
- SDValue Deinterleaved = DAG.getNode(
- ISD::VECTOR_DEINTERLEAVE, DL, DAG.getVTList(SrcVT, SrcVT), SrcLo, SrcHi);
- SDValue Even =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(0));
- SDValue Odd =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
- SDValue Interleaved =
- DAG.getNode(ISD::VECTOR_INTERLEAVE, DL, DAG.getVTList(VT, VT), Even, Odd);
- return std::make_pair(Interleaved.getValue(0), Interleaved.getValue(1));
-}
-
void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
SDValue &Hi) {
EVT VT = N->getValueType(0);
SDValue Src = N->getOperand(0);
- EVT SrcVT = Src.getValueType();
EVT LoVT, HiVT;
std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(VT);
assert(LoVT == HiVT && "Expected equal split types");
- // Simple case: The split type is same as SrcVT.
- if (LoVT == SrcVT) {
- Lo = Hi = Src;
- return;
- }
-
- // Second case: the split result is known to be at least as wide as Src.
- SDLoc DL(N);
- if (LoVT.getVectorMinNumElements() >= SrcVT.getVectorMinNumElements()) {
- Lo = Hi = DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Src);
- return;
- }
-
- // Final case: VT is scalable and Src is a wider fixed-length vector.
// Use smaller even/odd source vectors so their broadcasts can be
// reinterleaved in the original lane order for every value of vscale.
- assert(VT.isScalableVector() && SrcVT.isFixedLengthVector() &&
- "Expected a fixed source wider than the split scalable result");
+ SDLoc DL(N);
auto [SrcLo, SrcHi] = DAG.SplitVector(Src, DL);
- std::tie(Lo, Hi) = buildSplitVectorBroadcast(DAG, DL, LoVT, SrcLo, SrcHi);
+ EVT SplitSrcVT = SrcLo.getValueType();
+ SDValue Deinterleaved =
+ DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL,
+ DAG.getVTList(SplitSrcVT, SplitSrcVT), SrcLo, SrcHi);
+ SDValue Even =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(0));
+ SDValue Odd =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(1));
+ SDValue Interleaved = DAG.getNode(ISD::VECTOR_INTERLEAVE, DL,
+ DAG.getVTList(LoVT, LoVT), Even, Odd);
+ Lo = Interleaved.getValue(0);
+ Hi = Interleaved.getValue(1);
}
void DAGTypeLegalizer::SplitVecRes_VECTOR_REVERSE(SDNode *N, SDValue &Lo,
@@ -3856,8 +3831,6 @@ bool DAGTypeLegalizer::SplitVectorOperand(SDNode *N, unsigned OpNo) {
case ISD::INSERT_SUBVECTOR: Res = SplitVecOp_INSERT_SUBVECTOR(N, OpNo); break;
case ISD::EXTRACT_VECTOR_ELT:Res = SplitVecOp_EXTRACT_VECTOR_ELT(N); break;
case ISD::CONCAT_VECTORS: Res = SplitVecOp_CONCAT_VECTORS(N); break;
- case ISD::VECTOR_BROADCAST:
- Res = SplitVecOp_VECTOR_BROADCAST(N);
break;
case ISD::VECTOR_FIND_LAST_ACTIVE:
Res = SplitVecOp_VECTOR_FIND_LAST_ACTIVE(N);
@@ -4038,18 +4011,6 @@ bool DAGTypeLegalizer::SplitVectorOperand(SDNode *N, unsigned OpNo) {
return false;
}
-SDValue DAGTypeLegalizer::SplitVecOp_VECTOR_BROADCAST(SDNode *N) {
- SDLoc DL(N);
- SDValue SrcLo, SrcHi;
- GetSplitVector(N->getOperand(0), SrcLo, SrcHi);
-
- // We are only splitting SrcVT, not the destiniation type, which remains VT.
- // buildSplitVectorBroadcast produces (VT,VT) so only Lo is needed.
- EVT VT = N->getValueType(0);
- auto [Lo, _] = buildSplitVectorBroadcast(DAG, DL, VT, SrcLo, SrcHi);
- return Lo;
-}
-
SDValue DAGTypeLegalizer::SplitVecOp_VECTOR_FIND_LAST_ACTIVE(SDNode *N) {
SDLoc DL(N);
@@ -8338,21 +8299,24 @@ SDValue DAGTypeLegalizer::WidenVecOp_VECTOR_BROADCAST(SDNode *N) {
EVT VT = N->getValueType(0);
SDValue Src = N->getOperand(0);
EVT SrcVT = Src.getValueType();
- EVT WidenVT = TLI.getTypeToTransformTo(*DAG.getContext(), SrcVT);
- assert(WidenVT.getVectorElementCount().isKnownMultipleOf(
+ EVT WidennedSrcVT = TLI.getTypeToTransformTo(*DAG.getContext(), SrcVT);
+ assert(WidennedSrcVT.getVectorElementCount().isKnownMultipleOf(
SrcVT.getVectorElementCount()) &&
"Cannot widen VECTOR_BROADCAST operand to an ElementCount that's not "
"a multiple of the input ElementCount.");
unsigned NumConcat =
- WidenVT.getVectorMinNumElements() / SrcVT.getVectorMinNumElements();
+ WidennedSrcVT.getVectorMinNumElements() / SrcVT.getVectorMinNumElements();
// Repeat the original source because the extra lanes of its widened value
// are unspecified.
SmallVector<SDValue, 8> Ops(NumConcat, Src);
- SDValue WidenedSrc = DAG.getNode(ISD::CONCAT_VECTORS, DL, WidenVT, Ops);
- if (VT == WidenVT)
- return WidenedSrc;
- return DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, WidenedSrc);
+ SDValue WidenedSrc = DAG.getNode(ISD::CONCAT_VECTORS, DL, WidennedSrcVT, Ops);
+ EVT WidenedVT = VT.changeVectorElementCount(
+ *DAG.getContext(),
+ ElementCount::getScalable(WidennedSrcVT.getVectorMinNumElements()));
+ SDValue Widened =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, WidenedVT, WidenedSrc);
+ return DAG.getExtractSubvector(DL, VT, Widened, 0);
}
SDValue DAGTypeLegalizer::WidenVecOp_INSERT_SUBVECTOR(SDNode *N) {
diff --git a/llvm/lib/CodeGen/TargetLoweringBase.cpp b/llvm/lib/CodeGen/TargetLoweringBase.cpp
index 5b486c0600db9..f852f7551ce16 100644
--- a/llvm/lib/CodeGen/TargetLoweringBase.cpp
+++ b/llvm/lib/CodeGen/TargetLoweringBase.cpp
@@ -772,10 +772,6 @@ void TargetLoweringBase::initActions() {
llvm::fill(RegClassForVT, nullptr);
llvm::fill(TargetDAGCombineArray, 0);
- // Targets must explicitly opt into legal VECTOR_BROADCAST nodes.
- for (MVT VT : MVT::vector_valuetypes())
- setOperationAction(ISD::VECTOR_BROADCAST, VT, Expand);
-
// Let extending atomic loads be unsupported by default.
for (MVT ValVT : MVT::all_valuetypes())
for (MVT MemVT : MVT::all_valuetypes())
diff --git a/llvm/lib/IR/Verifier.cpp b/llvm/lib/IR/Verifier.cpp
index af830c69e01ec..c1766f7b8f858 100644
--- a/llvm/lib/IR/Verifier.cpp
+++ b/llvm/lib/IR/Verifier.cpp
@@ -7060,29 +7060,20 @@ void Verifier::visitIntrinsicCall(Intrinsic::ID ID, CallBase &Call) {
case Intrinsic::vector_broadcast: {
auto *ResultTy = cast<VectorType>(Call.getType());
auto *ArgTy = cast<VectorType>(Call.getArgOperand(0)->getType());
- ElementCount ResultEC = ResultTy->getElementCount();
- ElementCount InputEC = ArgTy->getElementCount();
Check(ResultTy->getElementType() == ArgTy->getElementType(),
"vector_broadcast argument and result must have the same element "
"type.",
&Call);
-
- if (InputEC.isScalable() && ResultEC.isFixed()) {
- CheckFailed("vector_broadcast cannot broadcast a scalable vector to a "
- "fixed-width vector.",
- &Call);
- break;
- }
-
- // We can only compare element counts when the types are both scalable or
- // non-scalable.
- if (ResultEC.isScalable() == InputEC.isScalable()) {
- Check(ResultEC.isKnownMultipleOf(InputEC),
- "vector_broadcast result element count must be a multiple of the "
- "argument element count.",
- &Call);
- }
+ Check(ArgTy->getElementCount().isFixed(),
+ "vector_broadcast argument must be a fixed-length vector.", &Call);
+ Check(ResultTy->getElementCount().isScalable(),
+ "vector_broadcast result must be a scalable vector.", &Call);
+ Check(ArgTy->getElementCount().getKnownMinValue() ==
+ ResultTy->getElementCount().getKnownMinValue(),
+ "vector_broadcast argument and result must have the same minimum "
+ "element count.",
+ &Call);
break;
}
case Intrinsic::vector_insert: {
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 31ff0e6af7cf0..4a7b27cfc98b7 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -1990,17 +1990,10 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::VECTOR_SPLICE_RIGHT, VT, Custom);
}
- // Direct patterns exist for broadcasts to packed SVE types whose sources
- // fit in a NEON register. Custom lowering handles wider fixed sources.
+ // Direct patterns exist for broadcasts to packed SVE types.
for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32, MVT::nxv2i64,
MVT::nxv8f16, MVT::nxv4f32, MVT::nxv2f64, MVT::nxv8bf16})
- setOperationAction(ISD::VECTOR_BROADCAST, VT, Custom);
-
- // Custom legalise unpacked types to avoid promoting fixed-length sources
- // beyond the width supported by NEON.
- for (auto VT : {MVT::nxv2i8, MVT::nxv2i16, MVT::nxv2i32, MVT::nxv4i8,
- MVT::nxv4i16, MVT::nxv8i8})
- setOperationAction(ISD::VECTOR_BROADCAST, VT, Custom);
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Legal);
// Broadcasts to unpacked SVE type require explicit unpacking to add spacing
// between elements.
@@ -17850,51 +17843,23 @@ SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
SelectionDAG &DAG) const {
SDLoc DL(Op);
EVT VT = Op.getValueType();
- assert(VT.isScalableVector() &&
- "Custom lowering expected for scalable vectors only.");
+ assert(isUnpackedType(VT, DAG) && "Expected an unpacked vector type!");
- if (isUnpackedType(VT, DAG)) {
- // Broadcast into a packed container before extracting the low lanes, which
- // places the result elements at the spacing required by the unpacked type.
- EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
- SDValue Broadcast =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, Op.getOperand(0));
- return DAG.getExtractSubvector(DL, VT, Broadcast, 0);
- }
-
- assert(isPackedVectorType(VT, DAG) && "Expected a packed vector type!");
+ // Broadcast into a packed container before extracting the low lanes, which
+ // places the result elements at the spacing required by the unpacked type.
+ EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
SDValue Src = Op.getOperand(0);
EVT SrcVT = Src.getValueType();
-
- if (ElementCount::isKnownGE(VT.getVectorElementCount(),
- SrcVT.getVectorElementCount())) {
- // We are broadcasting a NEON-sized vector to a packed (legal) SVE type.
- // This is a legal operation and there are patterns for it.
- assert(SrcVT.getFixedSizeInBits() <= AArch64::SVEBitsPerBlock &&
- "Expected NEON-sized source");
- return Op;
- }
-
- // We are broadcasting to a packed scalable vector for a wider-than-NEON
- // vector. Deinterleave smaller source vectors until we get to a quad.
- // This is similar to SplitVecRes_VECTOR_BROADCAST, but we are past type
- // legalisation so it cannot be reused.
- // TODO: tbl might be cheaper? Or repeated dup z.q[idx] followed by zip1
- // z.q, but that requires +F64MM+NS
- assert(SrcVT.isFixedLengthVector() &&
- SrcVT.getFixedSizeInBits() > AArch64::SVEBitsPerBlock &&
- "Expected a fixed source for a vscale-dependent broadcast");
- auto [SrcLo, SrcHi] = DAG.SplitVector(Src, DL);
- EVT SplitVT = SrcLo.getValueType();
- SDValue Deinterleaved =
- DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL, DAG.getVTList(SplitVT, SplitVT),
- SrcLo, SrcHi);
- SDValue Even = DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved);
- SDValue Odd =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
- SDValue Interleaved =
- DAG.getNode(ISD::VECTOR_INTERLEAVE, DL, DAG.getVTList(VT, VT), Even, Odd);
- return Interleaved.getValue(0);
+ unsigned NumConcat =
+ PackedVT.getVectorMinNumElements() / SrcVT.getVectorMinNumElements();
+ SmallVector<SDValue, 8> Ops(NumConcat, Src);
+ EVT PackedSrcVT = SrcVT.changeVectorElementCount(
+ *DAG.getContext(),
+ ElementCount::getFixed(PackedVT.getVectorMinNumElements()));
+ SDValue PackedSrc = DAG.getNode(ISD::CONCAT_VECTORS, DL, PackedSrcVT, Ops);
+ SDValue Broadcast =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, PackedSrc);
+ return DAG.getExtractSubvector(DL, VT, Broadcast, 0);
}
SDValue AArch64TargetLowering::LowerINSERT_SUBVECTOR(SDValue Op,
@@ -32882,26 +32847,6 @@ void AArch64TargetLowering::ReplaceNodeResults(
case ISD::BITCAST:
ReplaceBITCASTResults(N, Results, DAG);
return;
- case ISD::VECTOR_BROADCAST: {
- EVT VT = N->getValueType(0);
- EVT PromotedVT = getTypeToTransformTo(*DAG.getContext(), VT);
- EVT PromotedInVT = N->getOperand(0).getValueType().changeVectorElementType(
- *DAG.getContext(), PromotedVT.getVectorElementType());
- if (!PromotedInVT.isFixedLengthVector() ||
- PromotedInVT.getFixedSizeInBits() <= AArch64::SVEBitsPerBlock)
- return;
-
- // Avoid promoting the input vector if that creates a vector wider than
- // 128bits. Instead, keep the element type and broadcast to its packed type.
- EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
- SDLoc DL(N);
- SDValue Broadcast =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, N->getOperand(0));
- SDValue Promoted =
- DAG.getNode(ISD::ANY_EXTEND_VECTOR_INREG, DL, PromotedVT, Broadcast);
- Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, Promoted));
- return;
- }
case ISD::VECREDUCE_ADD:
case ISD::VECREDUCE_SMAX:
case ISD::VECREDUCE_SMIN:
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
index f4273b90ce3a1..aeb8dd3d691f7 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -16,9 +16,11 @@ define <vscale x 16 x i8> @broadcast_double_i8(<8 x i8> %a) {
; CHECK-LABEL: broadcast_double_i8:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
-; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
+; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v8i8(<8 x i8> %a)
+ %tmp = shufflevector <8 x i8> %a, <8 x i8> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+ %out = call <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v16i8(<16 x i8> %tmp)
ret <vscale x 16 x i8> %out
}
@@ -45,18 +47,6 @@ define <vscale x 8 x i16> @broadcast_quad_i16(<8 x i16> %a) {
ret <vscale x 8 x i16> %out
}
-define <vscale x 16 x i8> @broadcast_quad_i16_to_wide_sve(<8 x i16> %a) {
-; CHECK-LABEL: broadcast_quad_i16_to_wide_sve:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
-; CHECK-NEXT: mov z0.q, q0
-; CHECK-NEXT: uzp1 z0.b, z0.b, z0.b
-; CHECK-NEXT: ret
- %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.v8i16(<8 x i16> %a)
- %out.legal = trunc <vscale x 16 x i16> %out to <vscale x 16 x i8>
- ret <vscale x 16 x i8> %out.legal
-}
-
define <vscale x 16 x i8> @broadcast_wide_i16(<8 x i16> %a.lo, <8 x i16> %a.hi) {
; CHECK-LABEL: broadcast_wide_i16:
; CHECK: // %bb.0:
@@ -75,22 +65,15 @@ define <vscale x 16 x i8> @broadcast_wide_i16(<8 x i16> %a.lo, <8 x i16> %a.hi)
ret <vscale x 16 x i8> %out.legal
}
-define <vscale x 16 x i16> @broadcast_nxv8i16_to_nxv16i16(<vscale x 8 x i16> %a) {
-; CHECK-LABEL: broadcast_nxv8i16_to_nxv16i16:
-; CHECK: // %bb.0:
-; CHECK-NEXT: mov z1.d, z0.d
-; CHECK-NEXT: ret
- %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.nxv8i16(<vscale x 8 x i16> %a)
- ret <vscale x 16 x i16> %out
-}
-
define <vscale x 8 x i16> @broadcast_double_i16(<4 x i16> %a) {
; CHECK-LABEL: broadcast_double_i16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
-; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
+; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 8 x i16> @llvm.vector.broadcast.nxv8i16.v4i16(<4 x i16> %a)
+ %tmp = shufflevector <4 x i16> %a, <4 x i16> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
+ %out = call <vscale x 8 x i16> @llvm.vector.broadcast.nxv8i16.v8i16(<8 x i16> %tmp)
ret <vscale x 8 x i16> %out
}
@@ -106,18 +89,6 @@ define <vscale x 4 x i32> @broadcast_double_i16_to_double_sve(<4 x i16> %a) {
ret <vscale x 4 x i32> %out.legal
}
-define <vscale x 16 x i8> @broadcast_double_i16_to_wide_sve(<4 x i16> %a) {
-; CHECK-LABEL: broadcast_double_i16_to_wide_sve:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
-; CHECK-NEXT: mov z0.d, d0
-; CHECK-NEXT: uzp1 z0.b, z0.b, z0.b
-; CHECK-NEXT: ret
- %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.v4i16(<4 x i16> %a)
- %out.legal = trunc <vscale x 16 x i16> %out to <vscale x 16 x i8>
- ret <vscale x 16 x i8> %out.legal
-}
-
define <vscale x 4 x i32> @broadcast_quad_i32(<4 x i32> %a) {
; CHECK-LABEL: broadcast_quad_i32:
; CHECK: // %bb.0:
@@ -128,47 +99,42 @@ define <vscale x 4 x i32> @broadcast_quad_i32(<4 x i32> %a) {
ret <vscale x 4 x i32> %out
}
-define <vscale x 2 x i64> @broadcast_quad_i64(<2 x i64> %a) {
-; CHECK-LABEL: broadcast_quad_i64:
+define <vscale x 4 x i32> @broadcast_double_i32(<2 x i32> %a) {
+; CHECK-LABEL: broadcast_double_i32:
; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v2i64(<2 x i64> %a)
- ret <vscale x 2 x i64> %out
+ %tmp = shufflevector <2 x i32> %a, <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 1>
+ %out = call <vscale x 4 x i32> @llvm.vector.broadcast.nxv4i32.v4i32(<4 x i32> %tmp)
+ ret <vscale x 4 x i32> %out
}
-; vscale_range tests
-
-define <vscale x 2 x i64> @broadcast_v4i32_to_nxv2i32(<4 x i32> %a) vscale_range(2,8) {
-; CHECK-LABEL: broadcast_v4i32_to_nxv2i32:
+define <vscale x 2 x i64> @broadcast_quad_i64(<2 x i64> %a) {
+; CHECK-LABEL: broadcast_quad_i64:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: mov z0.q, q0
-; CHECK-NEXT: uunpklo z0.d, z0.s
; CHECK-NEXT: ret
- %out = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %a)
- %out.legal = zext <vscale x 2 x i32> %out to <vscale x 2 x i64>
- ret <vscale x 2 x i64> %out.legal
+ %out = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v2i64(<2 x i64> %a)
+ ret <vscale x 2 x i64> %out
}
-; wider-than-NEON fixed-length source
-define <vscale x 2 x i64> @broadcast_v4i64_to_nxv2i64(<vscale x 2 x i64> %a.legal) vscale_range(2,8) {
-; CHECK-LABEL: broadcast_v4i64_to_nxv2i64:
+define <vscale x 2 x i64> @broadcast_double_i64(<1 x i64> %a) {
+; CHECK-LABEL: broadcast_double_i64:
; CHECK: // %bb.0:
-; CHECK-NEXT: movprfx z1, z0
-; CHECK-NEXT: ext z1.b, z1.b, z0.b, #16
-; CHECK-NEXT: uzp2 v2.2d, v0.2d, v1.2d
-; CHECK-NEXT: uzp1 v0.2d, v0.2d, v1.2d
-; CHECK-NEXT: mov z1.q, q2
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
; CHECK-NEXT: mov z0.q, q0
-; CHECK-NEXT: zip1 z0.d, z0.d, z1.d
; CHECK-NEXT: ret
- %a = call <4 x i64> @llvm.vector.extract.v4i64.nxv2i64(<vscale x 2 x i64> %a.legal, i64 0)
- %r = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v4i64(<4 x i64> %a)
- ret <vscale x 2 x i64> %r
+ %tmp = shufflevector <1 x i64> %a, <1 x i64> poison, <2 x i32> zeroinitializer
+ %out = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v2i64(<2 x i64> %tmp)
+ ret <vscale x 2 x i64> %out
}
+; vscale_range tests
+
; wider-than-NEON fixed-length source and wide destination
define <vscale x 4 x i32> @broadcast_v4i64_to_nxv4i64(<vscale x 2 x i64> %a.legal) vscale_range(2,8) {
; CHECK-LABEL: broadcast_v4i64_to_nxv4i64:
@@ -189,62 +155,6 @@ define <vscale x 4 x i32> @broadcast_v4i64_to_nxv4i64(<vscale x 2 x i64> %a.lega
ret <vscale x 4 x i32> %r.legal
}
-define <vscale x 2 x i64> @broadcast_v4i32_to_nxv2i32_no_vscale_range(<4 x i32> %a) {
-; CHECK-LABEL: broadcast_v4i32_to_nxv2i32_no_vscale_range:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
-; CHECK-NEXT: mov z0.q, q0
-; CHECK-NEXT: uunpklo z0.d, z0.s
-; CHECK-NEXT: ret
- %out = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %a)
- %out.legal = zext <vscale x 2 x i32> %out to <vscale x 2 x i64>
- ret <vscale x 2 x i64> %out.legal
-}
-
-define <vscale x 2 x i64> @broadcast_v4i64_to_nxv2i64_no_vscale_range(ptr %ptr) {
-; CHECK-LABEL: broadcast_v4i64_to_nxv2i64_no_vscale_range:
-; CHECK: // %bb.0:
-; CHECK-NEXT: ldp q1, q0, [x0]
-; CHECK-NEXT: uzp2 v2.2d, v1.2d, v0.2d
-; CHECK-NEXT: uzp1 v0.2d, v1.2d, v0.2d
-; CHECK-NEXT: mov z1.q, q2
-; CHECK-NEXT: mov z0.q, q0
-; CHECK-NEXT: zip1 z0.d, z0.d, z1.d
-; CHECK-NEXT: ret
- %a = load <4 x i64>, ptr %ptr, align 8
- %out = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v4i64(<4 x i64> %a)
- ret <vscale x 2 x i64> %out
-}
-
-define <vscale x 4 x i32> @broadcast_v8i64_to_nxv4i64_no_vscale_range(ptr %ptr) {
-; CHECK-LABEL: broadcast_v8i64_to_nxv4i64_no_vscale_range:
-; CHECK: // %bb.0:
-; CHECK-NEXT: ldp q1, q0, [x0]
-; CHECK-NEXT: ldp q3, q2, [x0, #32]
-; CHECK-NEXT: uzp2 v5.2d, v1.2d, v0.2d
-; CHECK-NEXT: uzp1 v0.2d, v1.2d, v0.2d
-; CHECK-NEXT: uzp2 v4.2d, v3.2d, v2.2d
-; CHECK-NEXT: uzp1 v2.2d, v3.2d, v2.2d
-; CHECK-NEXT: uzp2 v1.2d, v5.2d, v4.2d
-; CHECK-NEXT: uzp1 v3.2d, v5.2d, v4.2d
-; CHECK-NEXT: uzp2 v4.2d, v0.2d, v2.2d
-; CHECK-NEXT: uzp1 v0.2d, v0.2d, v2.2d
-; CHECK-NEXT: mov z1.q, q1
-; CHECK-NEXT: mov z2.q, q3
-; CHECK-NEXT: mov z3.q, q4
-; CHECK-NEXT: mov z0.q, q0
-; CHECK-NEXT: zip1 z1.d, z2.d, z1.d
-; CHECK-NEXT: zip1 z0.d, z0.d, z3.d
-; CHECK-NEXT: zip2 z2.d, z0.d, z1.d
-; CHECK-NEXT: zip1 z0.d, z0.d, z1.d
-; CHECK-NEXT: uzp1 z0.s, z0.s, z2.s
-; CHECK-NEXT: ret
- %a = load <8 x i64>, ptr %ptr, align 8
- %out = call <vscale x 4 x i64> @llvm.vector.broadcast.nxv4i64.v8i64(<8 x i64> %a)
- %out.legal = trunc <vscale x 4 x i64> %out to <vscale x 4 x i32>
- ret <vscale x 4 x i32> %out.legal
-}
-
; FP / BFP types
define <vscale x 8 x half> @broadcast_quad_f16(<8 x half> %a) {
@@ -261,9 +171,11 @@ define <vscale x 8 x half> @broadcast_double_f16(<4 x half> %a) {
; CHECK-LABEL: broadcast_double_f16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
-; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
+; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<4 x half> %a)
+ %tmp = shufflevector <4 x half> %a, <4 x half> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
+ %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<8 x half> %tmp)
ret <vscale x 8 x half> %out
}
@@ -271,57 +183,21 @@ define <vscale x 4 x half> @broadcast_double_f16_to_double_sve(<4 x half> %a) {
; CHECK-LABEL: broadcast_double_f16_to_double_sve:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
-; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
+; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: ret
%out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<4 x half> %a)
ret <vscale x 4 x half> %out
}
-define <vscale x 8 x half> @broadcast_2f16_to_nxv8f16(<4 x half> %a) {
-; CHECK-LABEL: broadcast_2f16_to_nxv8f16:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
-; CHECK-NEXT: dup v0.2s, v0.s[0]
-; CHECK-NEXT: mov z0.d, d0
-; CHECK-NEXT: ret
- %a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
- %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<2 x half> %a.legal)
- ret <vscale x 8 x half> %out
-}
-
-define <vscale x 4 x half> @broadcast_2f16_to_nxv4f16(<4 x half> %a) {
-; CHECK-LABEL: broadcast_2f16_to_nxv4f16:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
-; CHECK-NEXT: dup v0.2s, v0.s[0]
-; CHECK-NEXT: mov z0.d, d0
-; CHECK-NEXT: uunpklo z0.s, z0.h
-; CHECK-NEXT: ret
- %a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
- %out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<2 x half> %a.legal)
- ret <vscale x 4 x half> %out
-}
-
-define <vscale x 8 x half> @broadcast_2f16_to_nxv8f16_lo(<4 x half> %a) {
-; CHECK-LABEL: broadcast_2f16_to_nxv8f16_lo:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
-; CHECK-NEXT: dup v0.2s, v0.s[0]
-; CHECK-NEXT: mov z0.d, d0
-; CHECK-NEXT: ret
- %a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
- %out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<2 x half> %a.legal)
- %out.in.lo = call <vscale x 8 x half> @llvm.vector.insert.nxv8f16.nxv4f16(<vscale x 8 x half> poison, <vscale x 4 x half> %out, i64 0)
- ret <vscale x 8 x half> %out.in.lo
-}
-
define <vscale x 2 x half> @broadcast_2f16_to_nxv2f16(<4 x half> %a) {
; CHECK-LABEL: broadcast_2f16_to_nxv2f16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
; CHECK-NEXT: dup v0.2s, v0.s[0]
-; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
+; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: uunpklo z0.d, z0.s
; CHECK-NEXT: ret
@@ -354,7 +230,8 @@ define <vscale x 2 x float> @broadcast_double_f32_to_nxv2f32(<2 x float> %a) {
; CHECK-LABEL: broadcast_double_f32_to_nxv2f32:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
-; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
+; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: uunpklo z0.d, z0.s
; CHECK-NEXT: ret
%out = call <vscale x 2 x float> @llvm.vector.broadcast.nxv2f32.v2f32(<2 x float> %a)
@@ -365,7 +242,8 @@ define <vscale x 4 x bfloat> @broadcast_double_bf16_to_nxv4bf16(<4 x bfloat> %a)
; CHECK-LABEL: broadcast_double_bf16_to_nxv4bf16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
-; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
+; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: ret
%out = call <vscale x 4 x bfloat> @llvm.vector.broadcast.nxv4bf16.v4bf16(<4 x bfloat> %a)
@@ -403,12 +281,14 @@ define <vscale x 16 x i1> @broadcast_v8i1_to_nxv16i1(<8 x i8> %a) {
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: ptrue p0.b
-; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
+; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: and z0.b, z0.b, #0x1
; CHECK-NEXT: cmpne p0.b, p0/z, z0.b, #0
; CHECK-NEXT: ret
%a.legal = trunc <8 x i8> %a to <8 x i1>
- %out = call <vscale x 16 x i1> @llvm.vector.broadcast.nxv16i1.v8i1(<8 x i1> %a.legal)
+ %tmp = shufflevector <8 x i1> %a.legal, <8 x i1> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+ %out = call <vscale x 16 x i1> @llvm.vector.broadcast.nxv16i1.v16i1(<16 x i1> %tmp)
ret <vscale x 16 x i1> %out
}
@@ -445,26 +325,14 @@ define <vscale x 8 x i1> @broadcast_v4i1_double_to_nxv8i1(<4 x i16> %a) {
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: ptrue p0.h
-; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
+; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: and z0.h, z0.h, #0x1
; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
; CHECK-NEXT: ret
%a.legal = trunc <4 x i16> %a to <4 x i1>
- %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v4i1(<4 x i1> %a.legal)
- ret <vscale x 8 x i1> %out
-}
-
-define <vscale x 8 x i1> @broadcast_v4i1_quad_to_nxv8i1(<4 x i32> %a) {
-; CHECK-LABEL: broadcast_v4i1_quad_to_nxv8i1:
-; CHECK: // %bb.0:
-; CHECK-NEXT: xtn v0.4h, v0.4s
-; CHECK-NEXT: ptrue p0.h
-; CHECK-NEXT: mov z0.d, d0
-; CHECK-NEXT: and z0.h, z0.h, #0x1
-; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
-; CHECK-NEXT: ret
- %a.legal = trunc <4 x i32> %a to <4 x i1>
- %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v4i1(<4 x i1> %a.legal)
+ %tmp = shufflevector <4 x i1> %a.legal, <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
+ %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v8i1(<8 x i1> %tmp)
ret <vscale x 8 x i1> %out
}
@@ -501,26 +369,14 @@ define <vscale x 4 x i1> @broadcast_v2i1_double_to_nxv4i1(<2 x i32> %a) {
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: ptrue p0.s
-; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: mov v0.d[1], v0.d[0]
+; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: and z0.s, z0.s, #0x1
; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
; CHECK-NEXT: ret
%a.legal = trunc <2 x i32> %a to <2 x i1>
- %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v2i1(<2 x i1> %a.legal)
- ret <vscale x 4 x i1> %out
-}
-
-define <vscale x 4 x i1> @broadcast_v2i1_quad_to_nxv4i1(<2 x i64> %a) {
-; CHECK-LABEL: broadcast_v2i1_quad_to_nxv4i1:
-; CHECK: // %bb.0:
-; CHECK-NEXT: xtn v0.2s, v0.2d
-; CHECK-NEXT: ptrue p0.s
-; CHECK-NEXT: mov z0.d, d0
-; CHECK-NEXT: and z0.s, z0.s, #0x1
-; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
-; CHECK-NEXT: ret
- %a.legal = trunc <2 x i64> %a to <2 x i1>
- %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v2i1(<2 x i1> %a.legal)
+ %tmp = shufflevector <2 x i1> %a.legal, <2 x i1> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 1>
+ %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v4i1(<4 x i1> %tmp)
ret <vscale x 4 x i1> %out
}
diff --git a/llvm/test/CodeGen/AArch64/vector-broadcast.ll b/llvm/test/CodeGen/AArch64/vector-broadcast.ll
deleted file mode 100644
index b5a019bd9d5bf..0000000000000
--- a/llvm/test/CodeGen/AArch64/vector-broadcast.ll
+++ /dev/null
@@ -1,68 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -mtriple=aarch64-linux-gnu < %s | FileCheck %s
-
-; See sve-vector-broadcast.ll for scalable types.
-
-define <16 x i8> @broadcast_v8i8_to_v16i8(<8 x i8> %vec) {
-; CHECK-LABEL: broadcast_v8i8_to_v16i8:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
-; CHECK-NEXT: dup v0.2d, v0.d[0]
-; CHECK-NEXT: ret
- %result = call <16 x i8> @llvm.vector.broadcast.v16i8.v8i8(<8 x i8> %vec)
- ret <16 x i8> %result
-}
-
-define <16 x i8> @broadcast_v4i8_to_v16i8(<4 x i16> %wide.vec) {
-; CHECK-LABEL: broadcast_v4i8_to_v16i8:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
-; CHECK-NEXT: dup v0.2d, v0.d[0]
-; CHECK-NEXT: uzp1 v0.16b, v0.16b, v0.16b
-; CHECK-NEXT: ret
- %vec = trunc <4 x i16> %wide.vec to <4 x i8>
- %result = call <16 x i8> @llvm.vector.broadcast.v16i8.v4i8(<4 x i8> %vec)
- ret <16 x i8> %result
-}
-
-define <8 x i16> @broadcast_v4i16_to_v8i16(<4 x i16> %vec) {
-; CHECK-LABEL: broadcast_v4i16_to_v8i16:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
-; CHECK-NEXT: dup v0.2d, v0.d[0]
-; CHECK-NEXT: ret
- %result = call <8 x i16> @llvm.vector.broadcast.v8i16.v4i16(<4 x i16> %vec)
- ret <8 x i16> %result
-}
-
-define <8 x i16> @broadcast_v2i16_to_v8i16(<2 x i32> %wide.vec) {
-; CHECK-LABEL: broadcast_v2i16_to_v8i16:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
-; CHECK-NEXT: dup v0.2d, v0.d[0]
-; CHECK-NEXT: uzp1 v0.8h, v0.8h, v0.8h
-; CHECK-NEXT: ret
- %vec = trunc <2 x i32> %wide.vec to <2 x i16>
- %result = call <8 x i16> @llvm.vector.broadcast.v8i16.v2i16(<2 x i16> %vec)
- ret <8 x i16> %result
-}
-
-define <4 x i32> @broadcast_v2i32_to_v4i32(<2 x i32> %vec) {
-; CHECK-LABEL: broadcast_v2i32_to_v4i32:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
-; CHECK-NEXT: dup v0.2d, v0.d[0]
-; CHECK-NEXT: ret
- %result = call <4 x i32> @llvm.vector.broadcast.v4i32.v2i32(<2 x i32> %vec)
- ret <4 x i32> %result
-}
-
-define <4 x float> @broadcast_v2f32_to_v4f32(<2 x float> %vec) {
-; CHECK-LABEL: broadcast_v2f32_to_v4f32:
-; CHECK: // %bb.0:
-; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
-; CHECK-NEXT: dup v0.2d, v0.d[0]
-; CHECK-NEXT: ret
- %result = call <4 x float> @llvm.vector.broadcast.v4f32.v2f32(<2 x float> %vec)
- ret <4 x float> %result
-}
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll b/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
index f62c0b53bebe0..df3c9bfc81564 100644
--- a/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
+++ b/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
@@ -1,25 +1,25 @@
; RUN: not opt -passes=verify -disable-output < %s 2>&1 | FileCheck %s
; CHECK: vector_broadcast argument and result must have the same element type.
-define <8 x i32> @mismatched_element_types(<2 x i64> %vec) {
- %result = call <8 x i32> @llvm.vector.broadcast.v8i32.v2i64(<2 x i64> %vec)
- ret <8 x i32> %result
+define <vscale x 8 x i32> @mismatched_element_types(<8 x i64> %vec) {
+ %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.v8i64(<8 x i64> %vec)
+ ret <vscale x 8 x i32> %result
}
-; CHECK: vector_broadcast result element count must be a multiple of the argument element count.
-define <8 x i32> @non_multiple_fixed(<3 x i32> %vec) {
- %result = call <8 x i32> @llvm.vector.broadcast.v8i32.v3i32(<3 x i32> %vec)
+; CHECK: vector_broadcast result must be a scalable vector.
+define <8 x i32> @fixed_to_fixed(<8 x i32> %vec) {
+ %result = call <8 x i32> @llvm.vector.broadcast.v8i32(<8 x i32> %vec)
ret <8 x i32> %result
}
-; CHECK: vector_broadcast result element count must be a multiple of the argument element count.
-define <vscale x 8 x i32> @non_multiple_scalable(<vscale x 3 x i32> %vec) {
- %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.nxv3i32(<vscale x 3 x i32> %vec)
+; CHECK: vector_broadcast argument must be a fixed-length vector.
+define <vscale x 8 x i32> @scalable_to_scalable(<vscale x 8 x i32> %vec) {
+ %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32(<vscale x 8 x i32> %vec)
ret <vscale x 8 x i32> %result
}
-; CHECK: vector_broadcast cannot broadcast a scalable vector to a fixed-width vector.
-define <8 x i32> @scalable_to_fixed(<vscale x 2 x i32> %vec) {
- %result = call <8 x i32> @llvm.vector.broadcast.v8i32.nxv2i32(<vscale x 2 x i32> %vec)
- ret <8 x i32> %result
+; CHECK: vector_broadcast argument and result must have the same minimum element count.
+define <vscale x 8 x i32> @mismatched_minimum_element_count(<4 x i32> %vec) {
+ %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.v4i32(<4 x i32> %vec)
+ ret <vscale x 8 x i32> %result
}
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic.ll b/llvm/test/Verifier/vector-broadcast-intrinsic.ll
index 1f7434ec116c6..8487f2fff7994 100644
--- a/llvm/test/Verifier/vector-broadcast-intrinsic.ll
+++ b/llvm/test/Verifier/vector-broadcast-intrinsic.ll
@@ -1,26 +1,6 @@
; RUN: opt -passes=verify -disable-output < %s
-define <8 x i32> @fixed_to_fixed(<2 x i32> %vec) {
- %result = call <8 x i32> @llvm.vector.broadcast.v8i32.v2i32(<2 x i32> %vec)
- ret <8 x i32> %result
-}
-
-define <vscale x 8 x i32> @scalable_to_scalable(<vscale x 2 x i32> %vec) {
- %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.nxv2i32(<vscale x 2 x i32> %vec)
- ret <vscale x 8 x i32> %result
-}
-
-define <vscale x 2 x i32> @fixed_to_scalable_with_vscale_range(<4 x i32> %vec) vscale_range(2, 8) {
- %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
- ret <vscale x 2 x i32> %result
-}
-
-define <vscale x 2 x i32> @fixed_to_scalable_without_vscale_range(<4 x i32> %vec) {
- %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
- ret <vscale x 2 x i32> %result
-}
-
-define <vscale x 2 x i32> @fixed_to_scalable_without_sufficient_vscale_range(<8 x i32> %vec) vscale_range(2, 8) {
- %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v8i32(<8 x i32> %vec)
- ret <vscale x 2 x i32> %result
+define <vscale x 4 x i32> @fixed_to_scalable(<4 x i32> %vec) {
+ %result = call <vscale x 4 x i32> @llvm.vector.broadcast.nxv4i32.v4i32(<4 x i32> %vec)
+ ret <vscale x 4 x i32> %result
}
>From 676479cfebdfc907df956f90d7d54051906495b7 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Tue, 1 Sep 2026 15:56:53 +0000
Subject: [PATCH 14/19] Rename vector.broadcast -> vector.repeat
---
llvm/docs/LangRef.md | 8 +-
llvm/include/llvm/CodeGen/ISDOpcodes.h | 8 +-
llvm/include/llvm/IR/Intrinsics.td | 6 +-
.../include/llvm/Target/TargetSelectionDAG.td | 2 +-
.../SelectionDAG/LegalizeIntegerTypes.cpp | 14 +-
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 8 +-
.../SelectionDAG/LegalizeVectorTypes.cpp | 23 ++-
.../SelectionDAG/SelectionDAGBuilder.cpp | 4 +-
.../SelectionDAG/SelectionDAGDumper.cpp | 2 +-
llvm/lib/IR/Verifier.cpp | 10 +-
.../Target/AArch64/AArch64ISelLowering.cpp | 21 +-
llvm/lib/Target/AArch64/AArch64ISelLowering.h | 2 +-
llvm/lib/Target/AArch64/SVEInstrFormats.td | 6 +-
...ctor-broadcast.ll => sve-vector-repeat.ll} | 186 +++++++++---------
.../vector-broadcast-intrinsic-invalid.ll | 25 ---
.../vector-repeat-intrinsic-invalid.ll | 25 +++
...ntrinsic.ll => vector-repeat-intrinsic.ll} | 2 +-
17 files changed, 175 insertions(+), 177 deletions(-)
rename llvm/test/CodeGen/AArch64/{sve-vector-broadcast.ll => sve-vector-repeat.ll} (61%)
delete mode 100644 llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
create mode 100644 llvm/test/Verifier/vector-repeat-intrinsic-invalid.ll
rename llvm/test/Verifier/{vector-broadcast-intrinsic.ll => vector-repeat-intrinsic.ll} (62%)
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index bfa60f75cd47d..970f85e1a0d6b 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -20798,20 +20798,20 @@ runtime, then the result vector is a {ref}`poison value <poisonvalues>`. The
`idx` parameter must be a vector index constant type (for most targets this
will be an integer pointer type).
-#### '`llvm.vector.broadcast`' Intrinsic
+#### '`llvm.vector.repeat`' Intrinsic
##### Syntax:
This is an overloaded intrinsic.
```
-declare <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v16i8(<16 x i8> %vec)
+declare <vscale x 16 x i8> @llvm.vector.repeat.nxv16i8.v16i8(<16 x i8> %vec)
```
##### Overview:
-The '`llvm.vector.broadcast.*`' intrinsic repeatedly copies the elements of the
+The '`llvm.vector.repeat.*`' intrinsic repeatedly copies the elements of the
source fixed-length vector, in order, until the result scalable vector is
-filled. For example, broadcasting `<A, B>` produces a scalable vector containing
+filled. For example, repeating `<A, B>` produces a scalable vector containing
`vscale` copies of `<A, B>`.
##### Arguments:
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index f9fc070cd643a..953201a26a3e7 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -637,10 +637,10 @@ enum NodeType {
/// Result[J] = EXTRACT_SUBVECTOR(Interleaved, J * getVectorMinNumElements())
VECTOR_INTERLEAVE,
- /// VECTOR_BROADCAST(SRC_SUBVEC)
- /// Duplicate a fixed-length vector in a scalable vector with the same minimum
- /// element count.
- VECTOR_BROADCAST,
+ /// VECTOR_REPEAT(FIXED_LENGTH_VECTOR)
+ /// Repeatedly copies the elements of the source fixed-length vector to fill a
+ /// scalable vector with the same minimum element count.
+ VECTOR_REPEAT,
/// VECTOR_REVERSE(VECTOR) - Returns a vector, of the same type as VECTOR,
/// whose elements are shuffled using the following algorithm:
diff --git a/llvm/include/llvm/IR/Intrinsics.td b/llvm/include/llvm/IR/Intrinsics.td
index ba1e2c5b6170e..cf6b6ca40d2e0 100644
--- a/llvm/include/llvm/IR/Intrinsics.td
+++ b/llvm/include/llvm/IR/Intrinsics.td
@@ -2816,9 +2816,9 @@ foreach n = 2...8 in {
[IntrNoMem, IntrSpeculatable]>;
}
-// vector_broadcast( SrcVector )
-// Broadcast a fixed-length vector to its "vscale x" equivalent.
-def int_vector_broadcast : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
+// vector_repeat( SrcVector )
+// Repeat a fixed-length vector to fill its "vscale x" equivalent.
+def int_vector_repeat : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
[llvm_anyvector_ty],
[IntrNoMem, IntrSpeculatable]>;
diff --git a/llvm/include/llvm/Target/TargetSelectionDAG.td b/llvm/include/llvm/Target/TargetSelectionDAG.td
index cab55a327fa53..22fe30fea4fea 100644
--- a/llvm/include/llvm/Target/TargetSelectionDAG.td
+++ b/llvm/include/llvm/Target/TargetSelectionDAG.td
@@ -936,7 +936,7 @@ def vector_insert_subvec : SDNode<"ISD::INSERT_SUBVECTOR",
def extract_subvector : SDNode<"ISD::EXTRACT_SUBVECTOR", SDTSubVecExtract, []>;
def insert_subvector : SDNode<"ISD::INSERT_SUBVECTOR", SDTSubVecInsert, []>;
-def vector_broadcast : SDNode<"ISD::VECTOR_BROADCAST",
+def vector_repeat : SDNode<"ISD::VECTOR_REPEAT",
SDTypeProfile<1, 1, [SDTCisVec<1>, SDTCisVec<0>]>,
[]>;
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
index 5509199255560..794cb1039f556 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
@@ -127,8 +127,8 @@ void DAGTypeLegalizer::PromoteIntegerResult(SDNode *N, unsigned ResNo) {
case ISD::VECTOR_SPLICE_RIGHT:
Res = PromoteIntRes_VECTOR_SPLICE(N);
break;
- case ISD::VECTOR_BROADCAST:
- Res = PromoteIntRes_VECTOR_BROADCAST(N);
+ case ISD::VECTOR_REPEAT:
+ Res = PromoteIntRes_VECTOR_REPEAT(N);
break;
case ISD::VECTOR_INTERLEAVE:
case ISD::VECTOR_DEINTERLEAVE:
@@ -2160,8 +2160,8 @@ bool DAGTypeLegalizer::PromoteIntegerOperand(SDNode *N, unsigned OpNo) {
case ISD::PARTIAL_REDUCE_SUMLA:
Res = PromoteIntOp_PARTIAL_REDUCE_MLA(N);
break;
- case ISD::VECTOR_BROADCAST:
- Res = PromoteIntOp_VECTOR_BROADCAST(N);
+ case ISD::VECTOR_REPEAT:
+ Res = PromoteIntOp_VECTOR_REPEAT(N);
break;
case ISD::LOOP_DEPENDENCE_RAW_MASK:
case ISD::LOOP_DEPENDENCE_WAR_MASK:
@@ -3024,14 +3024,14 @@ SDValue DAGTypeLegalizer::PromoteIntOp_LOOP_DEPENDENCE_MASK(SDNode *N) {
return SDValue(DAG.UpdateNodeOperands(N, NewOps), 0);
}
-SDValue DAGTypeLegalizer::PromoteIntOp_VECTOR_BROADCAST(SDNode *N) {
+SDValue DAGTypeLegalizer::PromoteIntOp_VECTOR_REPEAT(SDNode *N) {
SDLoc DL(N);
SDValue Src = GetPromotedInteger(N->getOperand(0));
EVT SrcVT = Src.getValueType();
EVT OrigVT = N->getValueType(0);
EVT NewVT = EVT::getVectorVT(*DAG.getContext(), SrcVT.getVectorElementType(),
OrigVT.getVectorElementCount());
- SDValue Res = DAG.getNode(ISD::VECTOR_BROADCAST, DL, NewVT, Src);
+ SDValue Res = DAG.getNode(ISD::VECTOR_REPEAT, DL, NewVT, Src);
return DAG.getNode(ISD::TRUNCATE, DL, OrigVT, Res);
}
@@ -6123,7 +6123,7 @@ SDValue DAGTypeLegalizer::PromoteIntRes_VECTOR_SPLICE(SDNode *N) {
return DAG.getNode(N->getOpcode(), dl, OutVT, V0, V1, N->getOperand(2));
}
-SDValue DAGTypeLegalizer::PromoteIntRes_VECTOR_BROADCAST(SDNode *N) {
+SDValue DAGTypeLegalizer::PromoteIntRes_VECTOR_REPEAT(SDNode *N) {
SDLoc DL(N);
EVT OutVT = N->getValueType(0);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index cd7f4a733220f..0b0ac2d387665 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -286,7 +286,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue PromoteIntRes_VECTOR_REVERSE(SDNode *N);
SDValue PromoteIntRes_VECTOR_SHUFFLE(SDNode *N);
SDValue PromoteIntRes_VECTOR_SPLICE(SDNode *N);
- SDValue PromoteIntRes_VECTOR_BROADCAST(SDNode *N);
+ SDValue PromoteIntRes_VECTOR_REPEAT(SDNode *N);
SDValue PromoteIntRes_VECTOR_INTERLEAVE_DEINTERLEAVE(SDNode *N);
SDValue PromoteIntRes_BUILD_VECTOR(SDNode *N);
SDValue PromoteIntRes_ScalarOp(SDNode *N);
@@ -420,7 +420,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue PromoteIntOp_GET_ACTIVE_LANE_MASK(SDNode *N);
SDValue PromoteIntOp_VECTOR_MATCH(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_PARTIAL_REDUCE_MLA(SDNode *N);
- SDValue PromoteIntOp_VECTOR_BROADCAST(SDNode *N);
+ SDValue PromoteIntOp_VECTOR_REPEAT(SDNode *N);
SDValue PromoteIntOp_LOOP_DEPENDENCE_MASK(SDNode *N);
SDValue PromoteIntOp_MaskedBinOp(SDNode *N, unsigned OpNo);
@@ -954,7 +954,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
void SplitVecRes_ScalarOp(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_STEP_VECTOR(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_SETCC(SDNode *N, SDValue &Lo, SDValue &Hi);
- void SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo, SDValue &Hi);
+ void SplitVecRes_VECTOR_REPEAT(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_VECTOR_REVERSE(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_VECTOR_SHUFFLE(ShuffleVectorSDNode *N, SDValue &Lo,
SDValue &Hi);
@@ -1102,7 +1102,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
bool WidenVectorOperand(SDNode *N, unsigned OpNo);
SDValue WidenVecOp_BITCAST(SDNode *N);
SDValue WidenVecOp_CONCAT_VECTORS(SDNode *N);
- SDValue WidenVecOp_VECTOR_BROADCAST(SDNode *N);
+ SDValue WidenVecOp_VECTOR_REPEAT(SDNode *N);
SDValue WidenVecOp_EXTEND(SDNode *N);
SDValue WidenVecOp_CMP(SDNode *N);
SDValue WidenVecOp_EXTRACT_VECTOR_ELT(SDNode *N);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 3644e2ac87930..9e5746ddf3c9e 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -1449,8 +1449,8 @@ void DAGTypeLegalizer::SplitVectorResult(SDNode *N, unsigned ResNo) {
case ISD::SETCC:
SplitVecRes_SETCC(N, Lo, Hi);
break;
- case ISD::VECTOR_BROADCAST:
- SplitVecRes_VECTOR_BROADCAST(N, Lo, Hi);
+ case ISD::VECTOR_REPEAT:
+ SplitVecRes_VECTOR_REPEAT(N, Lo, Hi);
break;
case ISD::VECTOR_REVERSE:
SplitVecRes_VECTOR_REVERSE(N, Lo, Hi);
@@ -3488,8 +3488,8 @@ void DAGTypeLegalizer::SplitVecRes_FP_TO_XINT_SAT(SDNode *N, SDValue &Lo,
Hi = DAG.getNode(N->getOpcode(), dl, DstVTHi, SrcHi, N->getOperand(1));
}
-void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
- SDValue &Hi) {
+void DAGTypeLegalizer::SplitVecRes_VECTOR_REPEAT(SDNode *N, SDValue &Lo,
+ SDValue &Hi) {
EVT VT = N->getValueType(0);
SDValue Src = N->getOperand(0);
EVT LoVT, HiVT;
@@ -3505,9 +3505,9 @@ void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL,
DAG.getVTList(SplitSrcVT, SplitSrcVT), SrcLo, SrcHi);
SDValue Even =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(0));
+ DAG.getNode(ISD::VECTOR_REPEAT, DL, LoVT, Deinterleaved.getValue(0));
SDValue Odd =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(1));
+ DAG.getNode(ISD::VECTOR_REPEAT, DL, LoVT, Deinterleaved.getValue(1));
SDValue Interleaved = DAG.getNode(ISD::VECTOR_INTERLEAVE, DL,
DAG.getVTList(LoVT, LoVT), Even, Odd);
Lo = Interleaved.getValue(0);
@@ -7837,8 +7837,8 @@ bool DAGTypeLegalizer::WidenVectorOperand(SDNode *N, unsigned OpNo) {
Res = WidenVecOp_FAKE_USE(N);
break;
case ISD::CONCAT_VECTORS: Res = WidenVecOp_CONCAT_VECTORS(N); break;
- case ISD::VECTOR_BROADCAST:
- Res = WidenVecOp_VECTOR_BROADCAST(N);
+ case ISD::VECTOR_REPEAT:
+ Res = WidenVecOp_VECTOR_REPEAT(N);
break;
case ISD::INSERT_SUBVECTOR: Res = WidenVecOp_INSERT_SUBVECTOR(N); break;
case ISD::EXTRACT_SUBVECTOR: Res = WidenVecOp_EXTRACT_SUBVECTOR(N); break;
@@ -8294,7 +8294,7 @@ SDValue DAGTypeLegalizer::WidenVecOp_CONCAT_VECTORS(SDNode *N) {
return DAG.getBuildVector(VT, dl, Ops);
}
-SDValue DAGTypeLegalizer::WidenVecOp_VECTOR_BROADCAST(SDNode *N) {
+SDValue DAGTypeLegalizer::WidenVecOp_VECTOR_REPEAT(SDNode *N) {
SDLoc DL(N);
EVT VT = N->getValueType(0);
SDValue Src = N->getOperand(0);
@@ -8302,7 +8302,7 @@ SDValue DAGTypeLegalizer::WidenVecOp_VECTOR_BROADCAST(SDNode *N) {
EVT WidennedSrcVT = TLI.getTypeToTransformTo(*DAG.getContext(), SrcVT);
assert(WidennedSrcVT.getVectorElementCount().isKnownMultipleOf(
SrcVT.getVectorElementCount()) &&
- "Cannot widen VECTOR_BROADCAST operand to an ElementCount that's not "
+ "Cannot widen VECTOR_REPEAT operand to an ElementCount that's not "
"a multiple of the input ElementCount.");
unsigned NumConcat =
WidennedSrcVT.getVectorMinNumElements() / SrcVT.getVectorMinNumElements();
@@ -8314,8 +8314,7 @@ SDValue DAGTypeLegalizer::WidenVecOp_VECTOR_BROADCAST(SDNode *N) {
EVT WidenedVT = VT.changeVectorElementCount(
*DAG.getContext(),
ElementCount::getScalable(WidennedSrcVT.getVectorMinNumElements()));
- SDValue Widened =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, WidenedVT, WidenedSrc);
+ SDValue Widened = DAG.getNode(ISD::VECTOR_REPEAT, DL, WidenedVT, WidenedSrc);
return DAG.getExtractSubvector(DL, VT, Widened, 0);
}
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
index aa4ec35b1955a..ffe1ed7ec45e0 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
@@ -8649,10 +8649,10 @@ void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
case Intrinsic::vector_deinterleave8:
visitVectorDeinterleave(I, 8);
return;
- case Intrinsic::vector_broadcast: {
+ case Intrinsic::vector_repeat: {
SDValue Vec = getValue(I.getOperand(0));
EVT ResultVT = TLI.getValueType(DAG.getDataLayout(), I.getType());
- setValue(&I, DAG.getNode(ISD::VECTOR_BROADCAST, sdl, ResultVT, Vec));
+ setValue(&I, DAG.getNode(ISD::VECTOR_REPEAT, sdl, ResultVT, Vec));
return;
}
case Intrinsic::experimental_vector_compress:
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
index 2819a2dd78ad4..8a92f5cc6d863 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
@@ -360,7 +360,7 @@ std::string SDNode::getOperationName(const SelectionDAG *G) const {
case ISD::EXTRACT_SUBVECTOR: return "extract_subvector";
case ISD::VECTOR_DEINTERLEAVE: return "vector_deinterleave";
case ISD::VECTOR_INTERLEAVE: return "vector_interleave";
- case ISD::VECTOR_BROADCAST: return "vector_broadcast";
+ case ISD::VECTOR_REPEAT: return "vector_repeat";
case ISD::SCALAR_TO_VECTOR: return "scalar_to_vector";
case ISD::VECTOR_SHUFFLE: return "vector_shuffle";
case ISD::VECTOR_SPLICE_LEFT: return "vector_splice_left";
diff --git a/llvm/lib/IR/Verifier.cpp b/llvm/lib/IR/Verifier.cpp
index c1766f7b8f858..6abccff8d8432 100644
--- a/llvm/lib/IR/Verifier.cpp
+++ b/llvm/lib/IR/Verifier.cpp
@@ -7057,21 +7057,21 @@ void Verifier::visitIntrinsicCall(Intrinsic::ID ID, CallBase &Call) {
}
break;
}
- case Intrinsic::vector_broadcast: {
+ case Intrinsic::vector_repeat: {
auto *ResultTy = cast<VectorType>(Call.getType());
auto *ArgTy = cast<VectorType>(Call.getArgOperand(0)->getType());
Check(ResultTy->getElementType() == ArgTy->getElementType(),
- "vector_broadcast argument and result must have the same element "
+ "vector_repeat argument and result must have the same element "
"type.",
&Call);
Check(ArgTy->getElementCount().isFixed(),
- "vector_broadcast argument must be a fixed-length vector.", &Call);
+ "vector_repeat argument must be a fixed-length vector.", &Call);
Check(ResultTy->getElementCount().isScalable(),
- "vector_broadcast result must be a scalable vector.", &Call);
+ "vector_repeat result must be a scalable vector.", &Call);
Check(ArgTy->getElementCount().getKnownMinValue() ==
ResultTy->getElementCount().getKnownMinValue(),
- "vector_broadcast argument and result must have the same minimum "
+ "vector_repeat argument and result must have the same minimum "
"element count.",
&Call);
break;
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 4a7b27cfc98b7..d8b79f69f40ad 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -1990,16 +1990,16 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::VECTOR_SPLICE_RIGHT, VT, Custom);
}
- // Direct patterns exist for broadcasts to packed SVE types.
+ // Direct patterns exist for quad broadcasts.
for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32, MVT::nxv2i64,
MVT::nxv8f16, MVT::nxv4f32, MVT::nxv2f64, MVT::nxv8bf16})
- setOperationAction(ISD::VECTOR_BROADCAST, VT, Legal);
+ setOperationAction(ISD::VECTOR_REPEAT, VT, Legal);
- // Broadcasts to unpacked SVE type require explicit unpacking to add spacing
- // between elements.
+ // VECTOR_REPEAT to unpacked SVE types require explicit unpacking to add
+ // spacing between elements.
for (auto VT : {MVT::nxv2f16, MVT::nxv4f16, MVT::nxv2f32, MVT::nxv2bf16,
MVT::nxv4bf16})
- setOperationAction(ISD::VECTOR_BROADCAST, VT, Custom);
+ setOperationAction(ISD::VECTOR_REPEAT, VT, Custom);
if (Subtarget->hasSVEB16B16() &&
Subtarget->isNonStreamingSVEorSME2Available()) {
@@ -8891,8 +8891,8 @@ SDValue AArch64TargetLowering::LowerOperation(SDValue Op,
return LowerEXTEND_VECTOR_INREG(Op, DAG);
case ISD::ZERO_EXTEND_VECTOR_INREG:
return LowerZERO_EXTEND_VECTOR_INREG(Op, DAG);
- case ISD::VECTOR_BROADCAST:
- return LowerVECTOR_BROADCAST(Op, DAG);
+ case ISD::VECTOR_REPEAT:
+ return LowerVECTOR_REPEAT(Op, DAG);
case ISD::VECTOR_SHUFFLE:
return LowerVECTOR_SHUFFLE(Op, DAG);
case ISD::SPLAT_VECTOR:
@@ -17839,8 +17839,8 @@ SDValue AArch64TargetLowering::LowerEXTRACT_SUBVECTOR(SDValue Op,
return SDValue();
}
-SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
- SelectionDAG &DAG) const {
+SDValue AArch64TargetLowering::LowerVECTOR_REPEAT(SDValue Op,
+ SelectionDAG &DAG) const {
SDLoc DL(Op);
EVT VT = Op.getValueType();
assert(isUnpackedType(VT, DAG) && "Expected an unpacked vector type!");
@@ -17857,8 +17857,7 @@ SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
*DAG.getContext(),
ElementCount::getFixed(PackedVT.getVectorMinNumElements()));
SDValue PackedSrc = DAG.getNode(ISD::CONCAT_VECTORS, DL, PackedSrcVT, Ops);
- SDValue Broadcast =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, PackedSrc);
+ SDValue Broadcast = DAG.getNode(ISD::VECTOR_REPEAT, DL, PackedVT, PackedSrc);
return DAG.getExtractSubvector(DL, VT, Broadcast, 0);
}
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.h b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
index 99c78316b7367..e1ac77334d2c7 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.h
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
@@ -765,7 +765,7 @@ class AArch64TargetLowering : public TargetLowering {
SDValue LowerBUILD_VECTOR(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerEXTEND_VECTOR_INREG(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerZERO_EXTEND_VECTOR_INREG(SDValue Op, SelectionDAG &DAG) const;
- SDValue LowerVECTOR_BROADCAST(SDValue Op, SelectionDAG &DAG) const;
+ SDValue LowerVECTOR_REPEAT(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerVECTOR_SHUFFLE(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerSPLAT_VECTOR(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerDUPQLane(SDValue Op, SelectionDAG &DAG) const;
diff --git a/llvm/lib/Target/AArch64/SVEInstrFormats.td b/llvm/lib/Target/AArch64/SVEInstrFormats.td
index 92213bfca0f73..f2e3bfb3ac866 100644
--- a/llvm/lib/Target/AArch64/SVEInstrFormats.td
+++ b/llvm/lib/Target/AArch64/SVEInstrFormats.td
@@ -1588,12 +1588,12 @@ multiclass sve_int_perm_dup_i<string asm> {
}
// Broadcast whole fixed-length vectors to packed SVE types.
- // See LowerVECTOR_BROADCAST for handling of legal unpacked types.
+ // See LowerVECTOR_REPEAT for handling of legal unpacked types.
foreach VT = [nxv16i8, nxv8i16, nxv8f16, nxv8bf16,
nxv4i32, nxv4f32, nxv2i64, nxv2f64] in {
- def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
+ def : Pat<(VT (vector_repeat (SVEType<VT>.ZSub V128:$vec))),
(!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
- def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
+ def : Pat<(VT (vector_repeat (SVEType<VT>.DSub V64:$vec))),
(!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-repeat.ll
similarity index 61%
rename from llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
rename to llvm/test/CodeGen/AArch64/sve-vector-repeat.ll
index aeb8dd3d691f7..f3ca78ab2368b 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-repeat.ll
@@ -2,53 +2,53 @@
; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve < %s | FileCheck %s
-define <vscale x 16 x i8> @broadcast_quad_i8(<16 x i8> %a) {
-; CHECK-LABEL: broadcast_quad_i8:
+define <vscale x 16 x i8> @repeat_quad_i8(<16 x i8> %a) {
+; CHECK-LABEL: repeat_quad_i8:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v16i8(<16 x i8> %a)
+ %out = call <vscale x 16 x i8> @llvm.vector.repeat.nxv16i8.v16i8(<16 x i8> %a)
ret <vscale x 16 x i8> %out
}
-define <vscale x 16 x i8> @broadcast_double_i8(<8 x i8> %a) {
-; CHECK-LABEL: broadcast_double_i8:
+define <vscale x 16 x i8> @repeat_double_i8(<8 x i8> %a) {
+; CHECK-LABEL: repeat_double_i8:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: mov v0.d[1], v0.d[0]
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
%tmp = shufflevector <8 x i8> %a, <8 x i8> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
- %out = call <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v16i8(<16 x i8> %tmp)
+ %out = call <vscale x 16 x i8> @llvm.vector.repeat.nxv16i8.v16i8(<16 x i8> %tmp)
ret <vscale x 16 x i8> %out
}
-define <vscale x 8 x i16> @broadcast_double_i8_to_double_sve(<8 x i8> %a) {
-; CHECK-LABEL: broadcast_double_i8_to_double_sve:
+define <vscale x 8 x i16> @repeat_double_i8_to_double_sve(<8 x i8> %a) {
+; CHECK-LABEL: repeat_double_i8_to_double_sve:
; CHECK: // %bb.0:
; CHECK-NEXT: ushll v0.8h, v0.8b, #0
; CHECK-NEXT: ptrue p0.h
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: sxtb z0.h, p0/m, z0.h
; CHECK-NEXT: ret
- %out = call <vscale x 8 x i8> @llvm.vector.broadcast.nxv8i8.v8i8(<8 x i8> %a)
+ %out = call <vscale x 8 x i8> @llvm.vector.repeat.nxv8i8.v8i8(<8 x i8> %a)
%out.legal = sext <vscale x 8 x i8> %out to <vscale x 8 x i16>
ret <vscale x 8 x i16> %out.legal
}
-define <vscale x 8 x i16> @broadcast_quad_i16(<8 x i16> %a) {
-; CHECK-LABEL: broadcast_quad_i16:
+define <vscale x 8 x i16> @repeat_quad_i16(<8 x i16> %a) {
+; CHECK-LABEL: repeat_quad_i16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 8 x i16> @llvm.vector.broadcast.nxv8i16.v8i16(<8 x i16> %a)
+ %out = call <vscale x 8 x i16> @llvm.vector.repeat.nxv8i16.v8i16(<8 x i16> %a)
ret <vscale x 8 x i16> %out
}
-define <vscale x 16 x i8> @broadcast_wide_i16(<8 x i16> %a.lo, <8 x i16> %a.hi) {
-; CHECK-LABEL: broadcast_wide_i16:
+define <vscale x 16 x i8> @repeat_wide_i16(<8 x i16> %a.lo, <8 x i16> %a.hi) {
+; CHECK-LABEL: repeat_wide_i16:
; CHECK: // %bb.0:
; CHECK-NEXT: uzp2 v2.8h, v0.8h, v1.8h
; CHECK-NEXT: uzp1 v0.8h, v0.8h, v1.8h
@@ -60,84 +60,84 @@ define <vscale x 16 x i8> @broadcast_wide_i16(<8 x i16> %a.lo, <8 x i16> %a.hi)
; CHECK-NEXT: ret
%a = shufflevector <8 x i16> %a.lo, <8 x i16> %a.hi,
<16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
- %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.v16i16(<16 x i16> %a)
+ %out = call <vscale x 16 x i16> @llvm.vector.repeat.nxv16i16.v16i16(<16 x i16> %a)
%out.legal = trunc <vscale x 16 x i16> %out to <vscale x 16 x i8>
ret <vscale x 16 x i8> %out.legal
}
-define <vscale x 8 x i16> @broadcast_double_i16(<4 x i16> %a) {
-; CHECK-LABEL: broadcast_double_i16:
+define <vscale x 8 x i16> @repeat_double_i16(<4 x i16> %a) {
+; CHECK-LABEL: repeat_double_i16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: mov v0.d[1], v0.d[0]
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
%tmp = shufflevector <4 x i16> %a, <4 x i16> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
- %out = call <vscale x 8 x i16> @llvm.vector.broadcast.nxv8i16.v8i16(<8 x i16> %tmp)
+ %out = call <vscale x 8 x i16> @llvm.vector.repeat.nxv8i16.v8i16(<8 x i16> %tmp)
ret <vscale x 8 x i16> %out
}
-define <vscale x 4 x i32> @broadcast_double_i16_to_double_sve(<4 x i16> %a) {
-; CHECK-LABEL: broadcast_double_i16_to_double_sve:
+define <vscale x 4 x i32> @repeat_double_i16_to_double_sve(<4 x i16> %a) {
+; CHECK-LABEL: repeat_double_i16_to_double_sve:
; CHECK: // %bb.0:
; CHECK-NEXT: ushll v0.4s, v0.4h, #0
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: and z0.s, z0.s, #0xffff
; CHECK-NEXT: ret
- %out = call <vscale x 4 x i16> @llvm.vector.broadcast.nxv4i16.v4i16(<4 x i16> %a)
+ %out = call <vscale x 4 x i16> @llvm.vector.repeat.nxv4i16.v4i16(<4 x i16> %a)
%out.legal = zext <vscale x 4 x i16> %out to <vscale x 4 x i32>
ret <vscale x 4 x i32> %out.legal
}
-define <vscale x 4 x i32> @broadcast_quad_i32(<4 x i32> %a) {
-; CHECK-LABEL: broadcast_quad_i32:
+define <vscale x 4 x i32> @repeat_quad_i32(<4 x i32> %a) {
+; CHECK-LABEL: repeat_quad_i32:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 4 x i32> @llvm.vector.broadcast.nxv4i32.v4i32(<4 x i32> %a)
+ %out = call <vscale x 4 x i32> @llvm.vector.repeat.nxv4i32.v4i32(<4 x i32> %a)
ret <vscale x 4 x i32> %out
}
-define <vscale x 4 x i32> @broadcast_double_i32(<2 x i32> %a) {
-; CHECK-LABEL: broadcast_double_i32:
+define <vscale x 4 x i32> @repeat_double_i32(<2 x i32> %a) {
+; CHECK-LABEL: repeat_double_i32:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: mov v0.d[1], v0.d[0]
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
%tmp = shufflevector <2 x i32> %a, <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 1>
- %out = call <vscale x 4 x i32> @llvm.vector.broadcast.nxv4i32.v4i32(<4 x i32> %tmp)
+ %out = call <vscale x 4 x i32> @llvm.vector.repeat.nxv4i32.v4i32(<4 x i32> %tmp)
ret <vscale x 4 x i32> %out
}
-define <vscale x 2 x i64> @broadcast_quad_i64(<2 x i64> %a) {
-; CHECK-LABEL: broadcast_quad_i64:
+define <vscale x 2 x i64> @repeat_quad_i64(<2 x i64> %a) {
+; CHECK-LABEL: repeat_quad_i64:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v2i64(<2 x i64> %a)
+ %out = call <vscale x 2 x i64> @llvm.vector.repeat.nxv2i64.v2i64(<2 x i64> %a)
ret <vscale x 2 x i64> %out
}
-define <vscale x 2 x i64> @broadcast_double_i64(<1 x i64> %a) {
-; CHECK-LABEL: broadcast_double_i64:
+define <vscale x 2 x i64> @repeat_double_i64(<1 x i64> %a) {
+; CHECK-LABEL: repeat_double_i64:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
; CHECK-NEXT: dup v0.2d, v0.d[0]
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
%tmp = shufflevector <1 x i64> %a, <1 x i64> poison, <2 x i32> zeroinitializer
- %out = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v2i64(<2 x i64> %tmp)
+ %out = call <vscale x 2 x i64> @llvm.vector.repeat.nxv2i64.v2i64(<2 x i64> %tmp)
ret <vscale x 2 x i64> %out
}
; vscale_range tests
; wider-than-NEON fixed-length source and wide destination
-define <vscale x 4 x i32> @broadcast_v4i64_to_nxv4i64(<vscale x 2 x i64> %a.legal) vscale_range(2,8) {
-; CHECK-LABEL: broadcast_v4i64_to_nxv4i64:
+define <vscale x 4 x i32> @repeat_v4i64_to_nxv4i64(<vscale x 2 x i64> %a.legal) vscale_range(2,8) {
+; CHECK-LABEL: repeat_v4i64_to_nxv4i64:
; CHECK: // %bb.0:
; CHECK-NEXT: movprfx z1, z0
; CHECK-NEXT: ext z1.b, z1.b, z0.b, #16
@@ -150,49 +150,49 @@ define <vscale x 4 x i32> @broadcast_v4i64_to_nxv4i64(<vscale x 2 x i64> %a.lega
; CHECK-NEXT: uzp1 z0.s, z0.s, z2.s
; CHECK-NEXT: ret
%a = call <4 x i64> @llvm.vector.extract.v4i64.nxv2i64(<vscale x 2 x i64> %a.legal, i64 0)
- %r = call <vscale x 4 x i64> @llvm.vector.broadcast.nxv4i64.v4i64(<4 x i64> %a)
+ %r = call <vscale x 4 x i64> @llvm.vector.repeat.nxv4i64.v4i64(<4 x i64> %a)
%r.legal = trunc <vscale x 4 x i64> %r to <vscale x 4 x i32>
ret <vscale x 4 x i32> %r.legal
}
; FP / BFP types
-define <vscale x 8 x half> @broadcast_quad_f16(<8 x half> %a) {
-; CHECK-LABEL: broadcast_quad_f16:
+define <vscale x 8 x half> @repeat_quad_f16(<8 x half> %a) {
+; CHECK-LABEL: repeat_quad_f16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<8 x half> %a)
+ %out = call <vscale x 8 x half> @llvm.vector.repeat.nxv8f16(<8 x half> %a)
ret <vscale x 8 x half> %out
}
-define <vscale x 8 x half> @broadcast_double_f16(<4 x half> %a) {
-; CHECK-LABEL: broadcast_double_f16:
+define <vscale x 8 x half> @repeat_double_f16(<4 x half> %a) {
+; CHECK-LABEL: repeat_double_f16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: mov v0.d[1], v0.d[0]
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
%tmp = shufflevector <4 x half> %a, <4 x half> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
- %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<8 x half> %tmp)
+ %out = call <vscale x 8 x half> @llvm.vector.repeat.nxv8f16(<8 x half> %tmp)
ret <vscale x 8 x half> %out
}
-define <vscale x 4 x half> @broadcast_double_f16_to_double_sve(<4 x half> %a) {
-; CHECK-LABEL: broadcast_double_f16_to_double_sve:
+define <vscale x 4 x half> @repeat_double_f16_to_double_sve(<4 x half> %a) {
+; CHECK-LABEL: repeat_double_f16_to_double_sve:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: mov v0.d[1], v0.d[0]
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: ret
- %out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<4 x half> %a)
+ %out = call <vscale x 4 x half> @llvm.vector.repeat.nxv4f16(<4 x half> %a)
ret <vscale x 4 x half> %out
}
-define <vscale x 2 x half> @broadcast_2f16_to_nxv2f16(<4 x half> %a) {
-; CHECK-LABEL: broadcast_2f16_to_nxv2f16:
+define <vscale x 2 x half> @repeat_2f16_to_nxv2f16(<4 x half> %a) {
+; CHECK-LABEL: repeat_2f16_to_nxv2f16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
; CHECK-NEXT: dup v0.2s, v0.s[0]
@@ -202,68 +202,68 @@ define <vscale x 2 x half> @broadcast_2f16_to_nxv2f16(<4 x half> %a) {
; CHECK-NEXT: uunpklo z0.d, z0.s
; CHECK-NEXT: ret
%a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
- %out = call <vscale x 2 x half> @llvm.vector.broadcast.nxv2f16(<2 x half> %a.legal)
+ %out = call <vscale x 2 x half> @llvm.vector.repeat.nxv2f16(<2 x half> %a.legal)
ret <vscale x 2 x half> %out
}
-define <vscale x 8 x bfloat> @broadcast_quad_bf16(<8 x bfloat> %a) #0 {
-; CHECK-LABEL: broadcast_quad_bf16:
+define <vscale x 8 x bfloat> @repeat_quad_bf16(<8 x bfloat> %a) #0 {
+; CHECK-LABEL: repeat_quad_bf16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 8 x bfloat> @llvm.vector.broadcast.nxv8bf16.v8bf16(<8 x bfloat> %a)
+ %out = call <vscale x 8 x bfloat> @llvm.vector.repeat.nxv8bf16.v8bf16(<8 x bfloat> %a)
ret <vscale x 8 x bfloat> %out
}
-define <vscale x 4 x float> @broadcast_quad_f32(<4 x float> %a) {
-; CHECK-LABEL: broadcast_quad_f32:
+define <vscale x 4 x float> @repeat_quad_f32(<4 x float> %a) {
+; CHECK-LABEL: repeat_quad_f32:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 4 x float> @llvm.vector.broadcast.nxv4f32.v4f32(<4 x float> %a)
+ %out = call <vscale x 4 x float> @llvm.vector.repeat.nxv4f32.v4f32(<4 x float> %a)
ret <vscale x 4 x float> %out
}
-define <vscale x 2 x float> @broadcast_double_f32_to_nxv2f32(<2 x float> %a) {
-; CHECK-LABEL: broadcast_double_f32_to_nxv2f32:
+define <vscale x 2 x float> @repeat_double_f32_to_nxv2f32(<2 x float> %a) {
+; CHECK-LABEL: repeat_double_f32_to_nxv2f32:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: mov v0.d[1], v0.d[0]
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: uunpklo z0.d, z0.s
; CHECK-NEXT: ret
- %out = call <vscale x 2 x float> @llvm.vector.broadcast.nxv2f32.v2f32(<2 x float> %a)
+ %out = call <vscale x 2 x float> @llvm.vector.repeat.nxv2f32.v2f32(<2 x float> %a)
ret <vscale x 2 x float> %out
}
-define <vscale x 4 x bfloat> @broadcast_double_bf16_to_nxv4bf16(<4 x bfloat> %a) #0 {
-; CHECK-LABEL: broadcast_double_bf16_to_nxv4bf16:
+define <vscale x 4 x bfloat> @repeat_double_bf16_to_nxv4bf16(<4 x bfloat> %a) #0 {
+; CHECK-LABEL: repeat_double_bf16_to_nxv4bf16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: mov v0.d[1], v0.d[0]
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: ret
- %out = call <vscale x 4 x bfloat> @llvm.vector.broadcast.nxv4bf16.v4bf16(<4 x bfloat> %a)
+ %out = call <vscale x 4 x bfloat> @llvm.vector.repeat.nxv4bf16.v4bf16(<4 x bfloat> %a)
ret <vscale x 4 x bfloat> %out
}
-define <vscale x 2 x double> @broadcast_quad_f64(<2 x double> %a) {
-; CHECK-LABEL: broadcast_quad_f64:
+define <vscale x 2 x double> @repeat_quad_f64(<2 x double> %a) {
+; CHECK-LABEL: repeat_quad_f64:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: mov z0.q, q0
; CHECK-NEXT: ret
- %out = call <vscale x 2 x double> @llvm.vector.broadcast.nxv2f64.v2f64(<2 x double> %a)
+ %out = call <vscale x 2 x double> @llvm.vector.repeat.nxv2f64.v2f64(<2 x double> %a)
ret <vscale x 2 x double> %out
}
; Predicates
-define <vscale x 16 x i1> @broadcast_v16i1_to_nxv16i1(<16 x i8> %a) {
-; CHECK-LABEL: broadcast_v16i1_to_nxv16i1:
+define <vscale x 16 x i1> @repeat_v16i1_to_nxv16i1(<16 x i8> %a) {
+; CHECK-LABEL: repeat_v16i1_to_nxv16i1:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: ptrue p0.b
@@ -272,12 +272,12 @@ define <vscale x 16 x i1> @broadcast_v16i1_to_nxv16i1(<16 x i8> %a) {
; CHECK-NEXT: cmpne p0.b, p0/z, z0.b, #0
; CHECK-NEXT: ret
%a.legal = trunc <16 x i8> %a to <16 x i1>
- %out = call <vscale x 16 x i1> @llvm.vector.broadcast.nxv16i1.v16i1(<16 x i1> %a.legal)
+ %out = call <vscale x 16 x i1> @llvm.vector.repeat.nxv16i1.v16i1(<16 x i1> %a.legal)
ret <vscale x 16 x i1> %out
}
-define <vscale x 16 x i1> @broadcast_v8i1_to_nxv16i1(<8 x i8> %a) {
-; CHECK-LABEL: broadcast_v8i1_to_nxv16i1:
+define <vscale x 16 x i1> @repeat_v8i1_to_nxv16i1(<8 x i8> %a) {
+; CHECK-LABEL: repeat_v8i1_to_nxv16i1:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: ptrue p0.b
@@ -288,12 +288,12 @@ define <vscale x 16 x i1> @broadcast_v8i1_to_nxv16i1(<8 x i8> %a) {
; CHECK-NEXT: ret
%a.legal = trunc <8 x i8> %a to <8 x i1>
%tmp = shufflevector <8 x i1> %a.legal, <8 x i1> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
- %out = call <vscale x 16 x i1> @llvm.vector.broadcast.nxv16i1.v16i1(<16 x i1> %tmp)
+ %out = call <vscale x 16 x i1> @llvm.vector.repeat.nxv16i1.v16i1(<16 x i1> %tmp)
ret <vscale x 16 x i1> %out
}
-define <vscale x 8 x i1> @broadcast_v8i1_double_to_nxv8i1(<8 x i8> %a) {
-; CHECK-LABEL: broadcast_v8i1_double_to_nxv8i1:
+define <vscale x 8 x i1> @repeat_v8i1_double_to_nxv8i1(<8 x i8> %a) {
+; CHECK-LABEL: repeat_v8i1_double_to_nxv8i1:
; CHECK: // %bb.0:
; CHECK-NEXT: ushll v0.8h, v0.8b, #0
; CHECK-NEXT: ptrue p0.h
@@ -302,12 +302,12 @@ define <vscale x 8 x i1> @broadcast_v8i1_double_to_nxv8i1(<8 x i8> %a) {
; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
; CHECK-NEXT: ret
%a.legal = trunc <8 x i8> %a to <8 x i1>
- %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v8i1(<8 x i1> %a.legal)
+ %out = call <vscale x 8 x i1> @llvm.vector.repeat.nxv8i1.v8i1(<8 x i1> %a.legal)
ret <vscale x 8 x i1> %out
}
-define <vscale x 8 x i1> @broadcast_v8i1_quad_to_nxv8i1(<8 x i16> %a) {
-; CHECK-LABEL: broadcast_v8i1_quad_to_nxv8i1:
+define <vscale x 8 x i1> @repeat_v8i1_quad_to_nxv8i1(<8 x i16> %a) {
+; CHECK-LABEL: repeat_v8i1_quad_to_nxv8i1:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: ptrue p0.h
@@ -316,12 +316,12 @@ define <vscale x 8 x i1> @broadcast_v8i1_quad_to_nxv8i1(<8 x i16> %a) {
; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
; CHECK-NEXT: ret
%a.legal = trunc <8 x i16> %a to <8 x i1>
- %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v8i1(<8 x i1> %a.legal)
+ %out = call <vscale x 8 x i1> @llvm.vector.repeat.nxv8i1.v8i1(<8 x i1> %a.legal)
ret <vscale x 8 x i1> %out
}
-define <vscale x 8 x i1> @broadcast_v4i1_double_to_nxv8i1(<4 x i16> %a) {
-; CHECK-LABEL: broadcast_v4i1_double_to_nxv8i1:
+define <vscale x 8 x i1> @repeat_v4i1_double_to_nxv8i1(<4 x i16> %a) {
+; CHECK-LABEL: repeat_v4i1_double_to_nxv8i1:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: ptrue p0.h
@@ -332,12 +332,12 @@ define <vscale x 8 x i1> @broadcast_v4i1_double_to_nxv8i1(<4 x i16> %a) {
; CHECK-NEXT: ret
%a.legal = trunc <4 x i16> %a to <4 x i1>
%tmp = shufflevector <4 x i1> %a.legal, <4 x i1> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 0, i32 1, i32 2, i32 3>
- %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v8i1(<8 x i1> %tmp)
+ %out = call <vscale x 8 x i1> @llvm.vector.repeat.nxv8i1.v8i1(<8 x i1> %tmp)
ret <vscale x 8 x i1> %out
}
-define <vscale x 4 x i1> @broadcast_v4i1_double_to_nxv4i1(<4 x i16> %a) {
-; CHECK-LABEL: broadcast_v4i1_double_to_nxv4i1:
+define <vscale x 4 x i1> @repeat_v4i1_double_to_nxv4i1(<4 x i16> %a) {
+; CHECK-LABEL: repeat_v4i1_double_to_nxv4i1:
; CHECK: // %bb.0:
; CHECK-NEXT: ushll v0.4s, v0.4h, #0
; CHECK-NEXT: ptrue p0.s
@@ -346,12 +346,12 @@ define <vscale x 4 x i1> @broadcast_v4i1_double_to_nxv4i1(<4 x i16> %a) {
; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
; CHECK-NEXT: ret
%a.legal = trunc <4 x i16> %a to <4 x i1>
- %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v4i1(<4 x i1> %a.legal)
+ %out = call <vscale x 4 x i1> @llvm.vector.repeat.nxv4i1.v4i1(<4 x i1> %a.legal)
ret <vscale x 4 x i1> %out
}
-define <vscale x 4 x i1> @broadcast_v4i1_quad_to_nxv4i1(<4 x i32> %a) {
-; CHECK-LABEL: broadcast_v4i1_quad_to_nxv4i1:
+define <vscale x 4 x i1> @repeat_v4i1_quad_to_nxv4i1(<4 x i32> %a) {
+; CHECK-LABEL: repeat_v4i1_quad_to_nxv4i1:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: ptrue p0.s
@@ -360,12 +360,12 @@ define <vscale x 4 x i1> @broadcast_v4i1_quad_to_nxv4i1(<4 x i32> %a) {
; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
; CHECK-NEXT: ret
%a.legal = trunc <4 x i32> %a to <4 x i1>
- %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v4i1(<4 x i1> %a.legal)
+ %out = call <vscale x 4 x i1> @llvm.vector.repeat.nxv4i1.v4i1(<4 x i1> %a.legal)
ret <vscale x 4 x i1> %out
}
-define <vscale x 4 x i1> @broadcast_v2i1_double_to_nxv4i1(<2 x i32> %a) {
-; CHECK-LABEL: broadcast_v2i1_double_to_nxv4i1:
+define <vscale x 4 x i1> @repeat_v2i1_double_to_nxv4i1(<2 x i32> %a) {
+; CHECK-LABEL: repeat_v2i1_double_to_nxv4i1:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: ptrue p0.s
@@ -376,12 +376,12 @@ define <vscale x 4 x i1> @broadcast_v2i1_double_to_nxv4i1(<2 x i32> %a) {
; CHECK-NEXT: ret
%a.legal = trunc <2 x i32> %a to <2 x i1>
%tmp = shufflevector <2 x i1> %a.legal, <2 x i1> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 1>
- %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v4i1(<4 x i1> %tmp)
+ %out = call <vscale x 4 x i1> @llvm.vector.repeat.nxv4i1.v4i1(<4 x i1> %tmp)
ret <vscale x 4 x i1> %out
}
-define <vscale x 2 x i1> @broadcast_v2i1_double_to_nxv2i1(<2 x i32> %a) {
-; CHECK-LABEL: broadcast_v2i1_double_to_nxv2i1:
+define <vscale x 2 x i1> @repeat_v2i1_double_to_nxv2i1(<2 x i32> %a) {
+; CHECK-LABEL: repeat_v2i1_double_to_nxv2i1:
; CHECK: // %bb.0:
; CHECK-NEXT: ushll v0.2d, v0.2s, #0
; CHECK-NEXT: ptrue p0.d
@@ -390,12 +390,12 @@ define <vscale x 2 x i1> @broadcast_v2i1_double_to_nxv2i1(<2 x i32> %a) {
; CHECK-NEXT: cmpne p0.d, p0/z, z0.d, #0
; CHECK-NEXT: ret
%a.legal = trunc <2 x i32> %a to <2 x i1>
- %out = call <vscale x 2 x i1> @llvm.vector.broadcast.nxv2i1.v2i1(<2 x i1> %a.legal)
+ %out = call <vscale x 2 x i1> @llvm.vector.repeat.nxv2i1.v2i1(<2 x i1> %a.legal)
ret <vscale x 2 x i1> %out
}
-define <vscale x 2 x i1> @broadcast_v2i1_quad_to_nxv2i1(<2 x i64> %a) {
-; CHECK-LABEL: broadcast_v2i1_quad_to_nxv2i1:
+define <vscale x 2 x i1> @repeat_v2i1_quad_to_nxv2i1(<2 x i64> %a) {
+; CHECK-LABEL: repeat_v2i1_quad_to_nxv2i1:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
; CHECK-NEXT: ptrue p0.d
@@ -404,6 +404,6 @@ define <vscale x 2 x i1> @broadcast_v2i1_quad_to_nxv2i1(<2 x i64> %a) {
; CHECK-NEXT: cmpne p0.d, p0/z, z0.d, #0
; CHECK-NEXT: ret
%a.legal = trunc <2 x i64> %a to <2 x i1>
- %out = call <vscale x 2 x i1> @llvm.vector.broadcast.nxv2i1.v2i1(<2 x i1> %a.legal)
+ %out = call <vscale x 2 x i1> @llvm.vector.repeat.nxv2i1.v2i1(<2 x i1> %a.legal)
ret <vscale x 2 x i1> %out
}
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll b/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
deleted file mode 100644
index df3c9bfc81564..0000000000000
--- a/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
+++ /dev/null
@@ -1,25 +0,0 @@
-; RUN: not opt -passes=verify -disable-output < %s 2>&1 | FileCheck %s
-
-; CHECK: vector_broadcast argument and result must have the same element type.
-define <vscale x 8 x i32> @mismatched_element_types(<8 x i64> %vec) {
- %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.v8i64(<8 x i64> %vec)
- ret <vscale x 8 x i32> %result
-}
-
-; CHECK: vector_broadcast result must be a scalable vector.
-define <8 x i32> @fixed_to_fixed(<8 x i32> %vec) {
- %result = call <8 x i32> @llvm.vector.broadcast.v8i32(<8 x i32> %vec)
- ret <8 x i32> %result
-}
-
-; CHECK: vector_broadcast argument must be a fixed-length vector.
-define <vscale x 8 x i32> @scalable_to_scalable(<vscale x 8 x i32> %vec) {
- %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32(<vscale x 8 x i32> %vec)
- ret <vscale x 8 x i32> %result
-}
-
-; CHECK: vector_broadcast argument and result must have the same minimum element count.
-define <vscale x 8 x i32> @mismatched_minimum_element_count(<4 x i32> %vec) {
- %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.v4i32(<4 x i32> %vec)
- ret <vscale x 8 x i32> %result
-}
diff --git a/llvm/test/Verifier/vector-repeat-intrinsic-invalid.ll b/llvm/test/Verifier/vector-repeat-intrinsic-invalid.ll
new file mode 100644
index 0000000000000..da2c261556cbb
--- /dev/null
+++ b/llvm/test/Verifier/vector-repeat-intrinsic-invalid.ll
@@ -0,0 +1,25 @@
+; RUN: not opt -passes=verify -disable-output < %s 2>&1 | FileCheck %s
+
+; CHECK: vector_repeat argument and result must have the same element type.
+define <vscale x 8 x i32> @mismatched_element_types(<8 x i64> %vec) {
+ %result = call <vscale x 8 x i32> @llvm.vector.repeat.nxv8i32.v8i64(<8 x i64> %vec)
+ ret <vscale x 8 x i32> %result
+}
+
+; CHECK: vector_repeat result must be a scalable vector.
+define <8 x i32> @fixed_to_fixed(<8 x i32> %vec) {
+ %result = call <8 x i32> @llvm.vector.repeat.v8i32(<8 x i32> %vec)
+ ret <8 x i32> %result
+}
+
+; CHECK: vector_repeat argument must be a fixed-length vector.
+define <vscale x 8 x i32> @scalable_to_scalable(<vscale x 8 x i32> %vec) {
+ %result = call <vscale x 8 x i32> @llvm.vector.repeat.nxv8i32(<vscale x 8 x i32> %vec)
+ ret <vscale x 8 x i32> %result
+}
+
+; CHECK: vector_repeat argument and result must have the same minimum element count.
+define <vscale x 8 x i32> @mismatched_minimum_element_count(<4 x i32> %vec) {
+ %result = call <vscale x 8 x i32> @llvm.vector.repeat.nxv8i32.v4i32(<4 x i32> %vec)
+ ret <vscale x 8 x i32> %result
+}
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic.ll b/llvm/test/Verifier/vector-repeat-intrinsic.ll
similarity index 62%
rename from llvm/test/Verifier/vector-broadcast-intrinsic.ll
rename to llvm/test/Verifier/vector-repeat-intrinsic.ll
index 8487f2fff7994..9c6a6f4c40ac2 100644
--- a/llvm/test/Verifier/vector-broadcast-intrinsic.ll
+++ b/llvm/test/Verifier/vector-repeat-intrinsic.ll
@@ -1,6 +1,6 @@
; RUN: opt -passes=verify -disable-output < %s
define <vscale x 4 x i32> @fixed_to_scalable(<4 x i32> %vec) {
- %result = call <vscale x 4 x i32> @llvm.vector.broadcast.nxv4i32.v4i32(<4 x i32> %vec)
+ %result = call <vscale x 4 x i32> @llvm.vector.repeat.nxv4i32.v4i32(<4 x i32> %vec)
ret <vscale x 4 x i32> %result
}
>From 85693190d28685673e143332114df3e4f22c73d8 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Wed, 2 Sep 2026 08:08:56 +0000
Subject: [PATCH 15/19] Support nxv1 types
---
.../Target/AArch64/AArch64ISelLowering.cpp | 17 ++++
.../test/CodeGen/AArch64/sve-vector-repeat.ll | 94 +++++++++++++++++++
2 files changed, 111 insertions(+)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index d8b79f69f40ad..47ca991ce4876 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2001,6 +2001,11 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
MVT::nxv4bf16})
setOperationAction(ISD::VECTOR_REPEAT, VT, Custom);
+ // VECTOR_REPEAT to nxv1 types is just a VECTOR_SPLAT
+ for (auto VT : {MVT::nxv1f16, MVT::nxv1f32, MVT::nxv1f64, MVT::nxv1bf16,
+ MVT::nxv1i8, MVT::nxv1i16, MVT::nxv1i32, MVT::nxv1i64})
+ setOperationAction(ISD::VECTOR_REPEAT, VT, Custom);
+
if (Subtarget->hasSVEB16B16() &&
Subtarget->isNonStreamingSVEorSME2Available()) {
// Note: Use SVE for bfloat16 operations when +sve-b16b16 is available.
@@ -33016,6 +33021,18 @@ void AArch64TargetLowering::ReplaceNodeResults(
Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, V));
return;
}
+ case ISD::VECTOR_REPEAT: {
+ EVT VT = N->getValueType(0);
+ assert(VT.getVectorElementCount() == ElementCount::getScalable(1) &&
+ "Expected an nxv1 type!");
+
+ SDLoc DL(N);
+ SDValue Elt =
+ DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VT.getVectorElementType(),
+ N->getOperand(0), DAG.getVectorIdxConstant(0, DL));
+ Results.push_back(DAG.getSplatVector(VT, DL, Elt));
+ return;
+ }
case ISD::INTRINSIC_WO_CHAIN: {
EVT VT = N->getValueType(0);
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-repeat.ll b/llvm/test/CodeGen/AArch64/sve-vector-repeat.ll
index f3ca78ab2368b..1d21957ea7c7b 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-repeat.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-repeat.ll
@@ -37,6 +37,18 @@ define <vscale x 8 x i16> @repeat_double_i8_to_double_sve(<8 x i8> %a) {
ret <vscale x 8 x i16> %out.legal
}
+define <vscale x 16 x i8> @repeat_one_i8(<8 x i8> %a.legal) {
+; CHECK-LABEL: repeat_one_i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.b, b0
+; CHECK-NEXT: ret
+ %a = call <1 x i8> @llvm.vector.extract.v1i8.nxv1i8(<8 x i8> %a.legal, i64 0)
+ %out = call <vscale x 1 x i8> @llvm.vector.repeat.nxv1i8.v1i8(<1 x i8> %a)
+ %out.legal = call <vscale x 16 x i8> @llvm.vector.insert.nxv16i8.nxv1i8(<vscale x 16 x i8> poison, <vscale x 1 x i8> %out, i64 0)
+ ret <vscale x 16 x i8> %out.legal
+}
+
define <vscale x 8 x i16> @repeat_quad_i16(<8 x i16> %a) {
; CHECK-LABEL: repeat_quad_i16:
; CHECK: // %bb.0:
@@ -77,6 +89,18 @@ define <vscale x 8 x i16> @repeat_double_i16(<4 x i16> %a) {
ret <vscale x 8 x i16> %out
}
+define <vscale x 8 x i16> @repeat_one_i16(<4 x i16> %a.legal) {
+; CHECK-LABEL: repeat_one_i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.h, h0
+; CHECK-NEXT: ret
+ %a = call <1 x i16> @llvm.vector.extract.v1i16.nxv1i16(<4 x i16> %a.legal, i64 0)
+ %out = call <vscale x 1 x i16> @llvm.vector.repeat.nxv1i16.v1i16(<1 x i16> %a)
+ %out.legal = call <vscale x 8 x i16> @llvm.vector.insert.nxv8i16.nxv1i16(<vscale x 8 x i16> poison, <vscale x 1 x i16> %out, i64 0)
+ ret <vscale x 8 x i16> %out.legal
+}
+
define <vscale x 4 x i32> @repeat_double_i16_to_double_sve(<4 x i16> %a) {
; CHECK-LABEL: repeat_double_i16_to_double_sve:
; CHECK: // %bb.0:
@@ -111,6 +135,18 @@ define <vscale x 4 x i32> @repeat_double_i32(<2 x i32> %a) {
ret <vscale x 4 x i32> %out
}
+define <vscale x 4 x i32> @repeat_one_i32(<2 x i32> %a.legal) {
+; CHECK-LABEL: repeat_one_i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.s, s0
+; CHECK-NEXT: ret
+ %a = call <1 x i32> @llvm.vector.extract.v1i32.nxv1i32(<2 x i32> %a.legal, i64 0)
+ %out = call <vscale x 1 x i32> @llvm.vector.repeat.nxv1i32.v1i32(<1 x i32> %a)
+ %out.legal = call <vscale x 4 x i32> @llvm.vector.insert.nxv4i32.nxv1i32(<vscale x 4 x i32> poison, <vscale x 1 x i32> %out, i64 0)
+ ret <vscale x 4 x i32> %out.legal
+}
+
define <vscale x 2 x i64> @repeat_quad_i64(<2 x i64> %a) {
; CHECK-LABEL: repeat_quad_i64:
; CHECK: // %bb.0:
@@ -133,6 +169,17 @@ define <vscale x 2 x i64> @repeat_double_i64(<1 x i64> %a) {
ret <vscale x 2 x i64> %out
}
+define <vscale x 2 x i64> @repeat_one_i64(<1 x i64> %a) {
+; CHECK-LABEL: repeat_one_i64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %out = call <vscale x 1 x i64> @llvm.vector.repeat.nxv1i64.v1i64(<1 x i64> %a)
+ %out.legal = call <vscale x 2 x i64> @llvm.vector.insert.nxv2i64.nxv1i64(<vscale x 2 x i64> poison, <vscale x 1 x i64> %out, i64 0)
+ ret <vscale x 2 x i64> %out.legal
+}
+
; vscale_range tests
; wider-than-NEON fixed-length source and wide destination
@@ -206,6 +253,18 @@ define <vscale x 2 x half> @repeat_2f16_to_nxv2f16(<4 x half> %a) {
ret <vscale x 2 x half> %out
}
+define <vscale x 8 x half> @repeat_one_f16(<4 x half> %a.legal) {
+; CHECK-LABEL: repeat_one_f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.h, h0
+; CHECK-NEXT: ret
+ %a = call <1 x half> @llvm.vector.extract.v1f16.v4f16(<4 x half> %a.legal, i64 0)
+ %out = call <vscale x 1 x half> @llvm.vector.repeat.nxv1f16.v1f16(<1 x half> %a)
+ %out.legal = call <vscale x 8 x half> @llvm.vector.insert.nxv8f16.nxv1f16(<vscale x 8 x half> poison, <vscale x 1 x half> %out, i64 0)
+ ret <vscale x 8 x half> %out.legal
+}
+
define <vscale x 8 x bfloat> @repeat_quad_bf16(<8 x bfloat> %a) #0 {
; CHECK-LABEL: repeat_quad_bf16:
; CHECK: // %bb.0:
@@ -216,6 +275,18 @@ define <vscale x 8 x bfloat> @repeat_quad_bf16(<8 x bfloat> %a) #0 {
ret <vscale x 8 x bfloat> %out
}
+define <vscale x 8 x bfloat> @repeat_one_bf16(<4 x bfloat> %a.legal) #0 {
+; CHECK-LABEL: repeat_one_bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.h, h0
+; CHECK-NEXT: ret
+ %a = call <1 x bfloat> @llvm.vector.extract.v1bf16.v4bf16(<4 x bfloat> %a.legal, i64 0)
+ %out = call <vscale x 1 x bfloat> @llvm.vector.repeat.nxv1bf16.v1bf16(<1 x bfloat> %a)
+ %out.legal = call <vscale x 8 x bfloat> @llvm.vector.insert.nxv8bf16.nxv1bf16(<vscale x 8 x bfloat> poison, <vscale x 1 x bfloat> %out, i64 0)
+ ret <vscale x 8 x bfloat> %out.legal
+}
+
define <vscale x 4 x float> @repeat_quad_f32(<4 x float> %a) {
; CHECK-LABEL: repeat_quad_f32:
; CHECK: // %bb.0:
@@ -238,6 +309,18 @@ define <vscale x 2 x float> @repeat_double_f32_to_nxv2f32(<2 x float> %a) {
ret <vscale x 2 x float> %out
}
+define <vscale x 4 x float> @repeat_one_f32(<2 x float> %a.legal) {
+; CHECK-LABEL: repeat_one_f32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.s, s0
+; CHECK-NEXT: ret
+ %a = call <1 x float> @llvm.vector.extract.v1f32.v2f32(<2 x float> %a.legal, i64 0)
+ %out = call <vscale x 1 x float> @llvm.vector.repeat.nxv1f32.v1f32(<1 x float> %a)
+ %out.legal = call <vscale x 4 x float> @llvm.vector.insert.nxv4f32.nxv1f32(<vscale x 4 x float> poison, <vscale x 1 x float> %out, i64 0)
+ ret <vscale x 4 x float> %out.legal
+}
+
define <vscale x 4 x bfloat> @repeat_double_bf16_to_nxv4bf16(<4 x bfloat> %a) #0 {
; CHECK-LABEL: repeat_double_bf16_to_nxv4bf16:
; CHECK: // %bb.0:
@@ -260,6 +343,17 @@ define <vscale x 2 x double> @repeat_quad_f64(<2 x double> %a) {
ret <vscale x 2 x double> %out
}
+define <vscale x 2 x double> @repeat_one_f64(<1 x double> %a) {
+; CHECK-LABEL: repeat_one_f64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %out = call <vscale x 1 x double> @llvm.vector.repeat.nxv1f64.v1f64(<1 x double> %a)
+ %out.legal = call <vscale x 2 x double> @llvm.vector.insert.nxv2f64.nxv1f64(<vscale x 2 x double> poison, <vscale x 1 x double> %out, i64 0)
+ ret <vscale x 2 x double> %out.legal
+}
+
; Predicates
define <vscale x 16 x i1> @repeat_v16i1_to_nxv16i1(<16 x i8> %a) {
>From dbe0aa604e6054b602dcaa487f16fa22813e3d98 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 21 Sep 2026 16:50:55 +0000
Subject: [PATCH 16/19] Address comments
---
llvm/docs/LangRef.md | 7 ++--
.../SelectionDAG/LegalizeVectorTypes.cpp | 22 ++++++------
.../lib/CodeGen/SelectionDAG/SelectionDAG.cpp | 12 +++++++
llvm/lib/IR/Verifier.cpp | 14 ++++----
.../Target/AArch64/AArch64ISelLowering.cpp | 34 ++++---------------
llvm/lib/Target/AArch64/SVEInstrFormats.td | 2 --
.../SelectionDAGNodeConstructionTest.cpp | 17 ++++++++++
7 files changed, 57 insertions(+), 51 deletions(-)
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index 970f85e1a0d6b..6d1fed90433d6 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -20816,10 +20816,9 @@ filled. For example, repeating `<A, B>` produces a scalable vector containing
##### Arguments:
-The argument must be a fixed-length vector and the result must be a scalable
-vector with the same element type and minimum element count. In other words,
-the result type is formed by adding `vscale x` in front of the argument type's
-element count.
+The argument must be a fixed-length vector (i.e. `<N x Ty>`) and the result a
+scalable vector that is exactly `vscale` times longer (i.e.
+`<vscale x N x Ty>`).
#### '`llvm.vector.reverse`' Intrinsic
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 9e5746ddf3c9e..161c753d448b0 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -3831,7 +3831,6 @@ bool DAGTypeLegalizer::SplitVectorOperand(SDNode *N, unsigned OpNo) {
case ISD::INSERT_SUBVECTOR: Res = SplitVecOp_INSERT_SUBVECTOR(N, OpNo); break;
case ISD::EXTRACT_VECTOR_ELT:Res = SplitVecOp_EXTRACT_VECTOR_ELT(N); break;
case ISD::CONCAT_VECTORS: Res = SplitVecOp_CONCAT_VECTORS(N); break;
- break;
case ISD::VECTOR_FIND_LAST_ACTIVE:
Res = SplitVecOp_VECTOR_FIND_LAST_ACTIVE(N);
break;
@@ -8299,21 +8298,24 @@ SDValue DAGTypeLegalizer::WidenVecOp_VECTOR_REPEAT(SDNode *N) {
EVT VT = N->getValueType(0);
SDValue Src = N->getOperand(0);
EVT SrcVT = Src.getValueType();
- EVT WidennedSrcVT = TLI.getTypeToTransformTo(*DAG.getContext(), SrcVT);
- assert(WidennedSrcVT.getVectorElementCount().isKnownMultipleOf(
- SrcVT.getVectorElementCount()) &&
- "Cannot widen VECTOR_REPEAT operand to an ElementCount that's not "
- "a multiple of the input ElementCount.");
- unsigned NumConcat =
- WidennedSrcVT.getVectorMinNumElements() / SrcVT.getVectorMinNumElements();
+ EVT WidenedSrcVT = TLI.getTypeToTransformTo(*DAG.getContext(), SrcVT);
+
+ if (!WidenedSrcVT.getVectorElementCount().hasKnownScalarFactor(
+ SrcVT.getVectorElementCount()))
+ report_fatal_error(
+ "Cannot widen VECTOR_REPEAT operand to an ElementCount that's not "
+ "a multiple of the input ElementCount.");
// Repeat the original source because the extra lanes of its widened value
// are unspecified.
+ unsigned NumConcat =
+ WidenedSrcVT.getVectorElementCount().getKnownScalarFactor(
+ SrcVT.getVectorElementCount());
SmallVector<SDValue, 8> Ops(NumConcat, Src);
- SDValue WidenedSrc = DAG.getNode(ISD::CONCAT_VECTORS, DL, WidennedSrcVT, Ops);
+ SDValue WidenedSrc = DAG.getNode(ISD::CONCAT_VECTORS, DL, WidenedSrcVT, Ops);
EVT WidenedVT = VT.changeVectorElementCount(
*DAG.getContext(),
- ElementCount::getScalable(WidennedSrcVT.getVectorMinNumElements()));
+ ElementCount::getScalable(WidenedSrcVT.getVectorMinNumElements()));
SDValue Widened = DAG.getNode(ISD::VECTOR_REPEAT, DL, WidenedVT, WidenedSrc);
return DAG.getExtractSubvector(DL, VT, Widened, 0);
}
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
index 0a3eb65965826..c10f738ea7d5a 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
@@ -7444,6 +7444,18 @@ SDValue SelectionDAG::getNode(unsigned Opcode, const SDLoc &DL, EVT VT,
if (N1.getValueType().getScalarType() == MVT::i1)
return getNode(ISD::VECREDUCE_AND, DL, VT, N1);
break;
+ case ISD::VECTOR_REPEAT:
+ assert(N1.getValueType().isFixedLengthVector() &&
+ "VECTOR_REPEAT requires a fixed-length vector operand");
+ assert(VT.isScalableVector() &&
+ "VECTOR_REPEAT requires a scalable vector result");
+ assert(N1.getValueType().getVectorMinNumElements() ==
+ VT.getVectorMinNumElements() &&
+ "VECTOR_REPEAT operand and result element counts must match");
+ if (VT.getVectorMinNumElements() == 1)
+ return getSplatVector(
+ VT, DL, getExtractVectorElt(DL, VT.getVectorElementType(), N1, 0));
+ break;
case ISD::SPLAT_VECTOR:
assert(VT.isVector() && "Wrong return type!");
// FIXME: Hexagon uses i32 scalar for a floating point zero vector so allow
diff --git a/llvm/lib/IR/Verifier.cpp b/llvm/lib/IR/Verifier.cpp
index 6abccff8d8432..7af5cf037373f 100644
--- a/llvm/lib/IR/Verifier.cpp
+++ b/llvm/lib/IR/Verifier.cpp
@@ -7058,19 +7058,17 @@ void Verifier::visitIntrinsicCall(Intrinsic::ID ID, CallBase &Call) {
break;
}
case Intrinsic::vector_repeat: {
- auto *ResultTy = cast<VectorType>(Call.getType());
- auto *ArgTy = cast<VectorType>(Call.getArgOperand(0)->getType());
+ auto *ResultTy = dyn_cast<ScalableVectorType>(Call.getType());
+ auto *ArgTy = dyn_cast<FixedVectorType>(Call.getArgOperand(0)->getType());
+ Check(ArgTy, "vector_repeat argument must be a fixed-length vector.",
+ &Call);
+ Check(ResultTy, "vector_repeat result must be a scalable vector.", &Call);
Check(ResultTy->getElementType() == ArgTy->getElementType(),
"vector_repeat argument and result must have the same element "
"type.",
&Call);
- Check(ArgTy->getElementCount().isFixed(),
- "vector_repeat argument must be a fixed-length vector.", &Call);
- Check(ResultTy->getElementCount().isScalable(),
- "vector_repeat result must be a scalable vector.", &Call);
- Check(ArgTy->getElementCount().getKnownMinValue() ==
- ResultTy->getElementCount().getKnownMinValue(),
+ Check(ArgTy->getNumElements() == ResultTy->getMinNumElements(),
"vector_repeat argument and result must have the same minimum "
"element count.",
&Call);
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 47ca991ce4876..96833bd90b4e5 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -2001,11 +2001,6 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
MVT::nxv4bf16})
setOperationAction(ISD::VECTOR_REPEAT, VT, Custom);
- // VECTOR_REPEAT to nxv1 types is just a VECTOR_SPLAT
- for (auto VT : {MVT::nxv1f16, MVT::nxv1f32, MVT::nxv1f64, MVT::nxv1bf16,
- MVT::nxv1i8, MVT::nxv1i16, MVT::nxv1i32, MVT::nxv1i64})
- setOperationAction(ISD::VECTOR_REPEAT, VT, Custom);
-
if (Subtarget->hasSVEB16B16() &&
Subtarget->isNonStreamingSVEorSME2Available()) {
// Note: Use SVE for bfloat16 operations when +sve-b16b16 is available.
@@ -17847,21 +17842,18 @@ SDValue AArch64TargetLowering::LowerEXTRACT_SUBVECTOR(SDValue Op,
SDValue AArch64TargetLowering::LowerVECTOR_REPEAT(SDValue Op,
SelectionDAG &DAG) const {
SDLoc DL(Op);
+ SDValue Src = Op.getOperand(0);
EVT VT = Op.getValueType();
+ EVT SrcVT = Src.getValueType();
assert(isUnpackedType(VT, DAG) && "Expected an unpacked vector type!");
+ assert(SrcVT.is64BitVector() && "Expected 64bit source!");
- // Broadcast into a packed container before extracting the low lanes, which
+ // Repeat into a packed container before extracting the low lanes, which
// places the result elements at the spacing required by the unpacked type.
EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
- SDValue Src = Op.getOperand(0);
- EVT SrcVT = Src.getValueType();
- unsigned NumConcat =
- PackedVT.getVectorMinNumElements() / SrcVT.getVectorMinNumElements();
- SmallVector<SDValue, 8> Ops(NumConcat, Src);
- EVT PackedSrcVT = SrcVT.changeVectorElementCount(
- *DAG.getContext(),
- ElementCount::getFixed(PackedVT.getVectorMinNumElements()));
- SDValue PackedSrc = DAG.getNode(ISD::CONCAT_VECTORS, DL, PackedSrcVT, Ops);
+ EVT PackedSrcVT = SrcVT.getDoubleNumVectorElementsVT(*DAG.getContext());
+ SDValue PackedSrc =
+ DAG.getNode(ISD::CONCAT_VECTORS, DL, PackedSrcVT, Src, Src);
SDValue Broadcast = DAG.getNode(ISD::VECTOR_REPEAT, DL, PackedVT, PackedSrc);
return DAG.getExtractSubvector(DL, VT, Broadcast, 0);
}
@@ -33021,18 +33013,6 @@ void AArch64TargetLowering::ReplaceNodeResults(
Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, V));
return;
}
- case ISD::VECTOR_REPEAT: {
- EVT VT = N->getValueType(0);
- assert(VT.getVectorElementCount() == ElementCount::getScalable(1) &&
- "Expected an nxv1 type!");
-
- SDLoc DL(N);
- SDValue Elt =
- DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VT.getVectorElementType(),
- N->getOperand(0), DAG.getVectorIdxConstant(0, DL));
- Results.push_back(DAG.getSplatVector(VT, DL, Elt));
- return;
- }
case ISD::INTRINSIC_WO_CHAIN: {
EVT VT = N->getValueType(0);
diff --git a/llvm/lib/Target/AArch64/SVEInstrFormats.td b/llvm/lib/Target/AArch64/SVEInstrFormats.td
index f2e3bfb3ac866..4b6a89d863d89 100644
--- a/llvm/lib/Target/AArch64/SVEInstrFormats.td
+++ b/llvm/lib/Target/AArch64/SVEInstrFormats.td
@@ -1593,8 +1593,6 @@ multiclass sve_int_perm_dup_i<string asm> {
nxv4i32, nxv4f32, nxv2i64, nxv2f64] in {
def : Pat<(VT (vector_repeat (SVEType<VT>.ZSub V128:$vec))),
(!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
- def : Pat<(VT (vector_repeat (SVEType<VT>.DSub V64:$vec))),
- (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
// When extracting from an unpacked vector the index must be scaled to account
diff --git a/llvm/unittests/CodeGen/SelectionDAGNodeConstructionTest.cpp b/llvm/unittests/CodeGen/SelectionDAGNodeConstructionTest.cpp
index c2c1b52926c5a..52b1eac2194e9 100644
--- a/llvm/unittests/CodeGen/SelectionDAGNodeConstructionTest.cpp
+++ b/llvm/unittests/CodeGen/SelectionDAGNodeConstructionTest.cpp
@@ -525,3 +525,20 @@ TEST_F(SelectionDAGNodeConstructionTest, ExpandPartialReduceSUMLA) {
EXPECT_EQ(NumSignExtends, 1u);
EXPECT_EQ(NumZeroExtends, 1u);
}
+
+TEST_F(SelectionDAGNodeConstructionTest, VectorRepeat) {
+ SDLoc DL;
+ SDValue V1 = DAG->getCopyFromReg(DAG->getEntryNode(), DL,
+ Register::index2VirtReg(1), MVT::v1i32);
+ SDValue NXV1 = DAG->getNode(ISD::VECTOR_REPEAT, DL, MVT::nxv1i32, V1);
+
+ ASSERT_EQ(NXV1.getOpcode(), ISD::SPLAT_VECTOR);
+ ASSERT_EQ(NXV1.getOperand(0).getOpcode(), ISD::EXTRACT_VECTOR_ELT);
+ EXPECT_EQ(NXV1.getOperand(0).getOperand(0), V1);
+ EXPECT_TRUE(isNullConstant(NXV1.getOperand(0).getOperand(1)));
+
+ SDValue V2 = DAG->getCopyFromReg(DAG->getEntryNode(), DL,
+ Register::index2VirtReg(2), MVT::v2i32);
+ SDValue NXV2 = DAG->getNode(ISD::VECTOR_REPEAT, DL, MVT::nxv2i32, V2);
+ EXPECT_EQ(NXV2.getOpcode(), ISD::VECTOR_REPEAT);
+}
>From bbde247c06fc2bb67ba8cdb0035383ad275e9d4d Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Wed, 23 Sep 2026 12:38:29 +0000
Subject: [PATCH 17/19] Address comments and restore newline
---
llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp | 1 -
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 1 +
llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp | 3 +--
llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp | 2 +-
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 1 -
...repeat-intrinsic.ll => vector-repeat-intrinsic-valid.ll} | 2 ++
llvm/unittests/CodeGen/SelectionDAGNodeConstructionTest.cpp | 6 +-----
7 files changed, 6 insertions(+), 10 deletions(-)
rename llvm/test/Verifier/{vector-repeat-intrinsic.ll => vector-repeat-intrinsic-valid.ll} (76%)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
index 794cb1039f556..e8cebfa19623c 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
@@ -6128,7 +6128,6 @@ SDValue DAGTypeLegalizer::PromoteIntRes_VECTOR_REPEAT(SDNode *N) {
EVT OutVT = N->getValueType(0);
EVT NOutVT = TLI.getTypeToTransformTo(*DAG.getContext(), OutVT);
- assert(NOutVT.isVector() && "This type must be promoted to a vector type");
EVT NInVT = N->getOperand(0).getValueType().changeVectorElementType(
*DAG.getContext(), NOutVT.getVectorElementType());
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index 0b0ac2d387665..1cab31267838a 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -318,6 +318,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue PromoteIntRes_MLOAD(MaskedLoadSDNode *N);
SDValue PromoteIntRes_MGATHER(MaskedGatherSDNode *N);
SDValue PromoteIntRes_VECTOR_COMPRESS(SDNode *N);
+
SDValue PromoteIntRes_Overflow(SDNode *N);
SDValue PromoteIntRes_FFREXP(SDNode *N);
SDValue PromoteIntRes_SADDSUBO(SDNode *N, unsigned ResNo);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 161c753d448b0..51c38c2704b99 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -3492,8 +3492,7 @@ void DAGTypeLegalizer::SplitVecRes_VECTOR_REPEAT(SDNode *N, SDValue &Lo,
SDValue &Hi) {
EVT VT = N->getValueType(0);
SDValue Src = N->getOperand(0);
- EVT LoVT, HiVT;
- std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(VT);
+ auto [LoVT, HiVT] = DAG.GetSplitDestVTs(VT);
assert(LoVT == HiVT && "Expected equal split types");
// Use smaller even/odd source vectors so their broadcasts can be
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
index c10f738ea7d5a..dbc8050cbe7a7 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
@@ -7449,7 +7449,7 @@ SDValue SelectionDAG::getNode(unsigned Opcode, const SDLoc &DL, EVT VT,
"VECTOR_REPEAT requires a fixed-length vector operand");
assert(VT.isScalableVector() &&
"VECTOR_REPEAT requires a scalable vector result");
- assert(N1.getValueType().getVectorMinNumElements() ==
+ assert(N1.getValueType().getVectorNumElements() ==
VT.getVectorMinNumElements() &&
"VECTOR_REPEAT operand and result element counts must match");
if (VT.getVectorMinNumElements() == 1)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 96833bd90b4e5..4c8c68d98b7b8 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -17845,7 +17845,6 @@ SDValue AArch64TargetLowering::LowerVECTOR_REPEAT(SDValue Op,
SDValue Src = Op.getOperand(0);
EVT VT = Op.getValueType();
EVT SrcVT = Src.getValueType();
- assert(isUnpackedType(VT, DAG) && "Expected an unpacked vector type!");
assert(SrcVT.is64BitVector() && "Expected 64bit source!");
// Repeat into a packed container before extracting the low lanes, which
diff --git a/llvm/test/Verifier/vector-repeat-intrinsic.ll b/llvm/test/Verifier/vector-repeat-intrinsic-valid.ll
similarity index 76%
rename from llvm/test/Verifier/vector-repeat-intrinsic.ll
rename to llvm/test/Verifier/vector-repeat-intrinsic-valid.ll
index 9c6a6f4c40ac2..a6bf98e38d7b4 100644
--- a/llvm/test/Verifier/vector-repeat-intrinsic.ll
+++ b/llvm/test/Verifier/vector-repeat-intrinsic-valid.ll
@@ -1,5 +1,7 @@
; RUN: opt -passes=verify -disable-output < %s
+; Test that a correctly formed vector.repeat passes verifier checks.
+
define <vscale x 4 x i32> @fixed_to_scalable(<4 x i32> %vec) {
%result = call <vscale x 4 x i32> @llvm.vector.repeat.nxv4i32.v4i32(<4 x i32> %vec)
ret <vscale x 4 x i32> %result
diff --git a/llvm/unittests/CodeGen/SelectionDAGNodeConstructionTest.cpp b/llvm/unittests/CodeGen/SelectionDAGNodeConstructionTest.cpp
index 52b1eac2194e9..2f5ae1ac52c1c 100644
--- a/llvm/unittests/CodeGen/SelectionDAGNodeConstructionTest.cpp
+++ b/llvm/unittests/CodeGen/SelectionDAGNodeConstructionTest.cpp
@@ -526,6 +526,7 @@ TEST_F(SelectionDAGNodeConstructionTest, ExpandPartialReduceSUMLA) {
EXPECT_EQ(NumZeroExtends, 1u);
}
+// Verify that a nxv1 vector_repeat gets canonicalised as splat_vector
TEST_F(SelectionDAGNodeConstructionTest, VectorRepeat) {
SDLoc DL;
SDValue V1 = DAG->getCopyFromReg(DAG->getEntryNode(), DL,
@@ -536,9 +537,4 @@ TEST_F(SelectionDAGNodeConstructionTest, VectorRepeat) {
ASSERT_EQ(NXV1.getOperand(0).getOpcode(), ISD::EXTRACT_VECTOR_ELT);
EXPECT_EQ(NXV1.getOperand(0).getOperand(0), V1);
EXPECT_TRUE(isNullConstant(NXV1.getOperand(0).getOperand(1)));
-
- SDValue V2 = DAG->getCopyFromReg(DAG->getEntryNode(), DL,
- Register::index2VirtReg(2), MVT::v2i32);
- SDValue NXV2 = DAG->getNode(ISD::VECTOR_REPEAT, DL, MVT::nxv2i32, V2);
- EXPECT_EQ(NXV2.getOpcode(), ISD::VECTOR_REPEAT);
}
>From c6155a564f6fed2c7101395dfb6e365065ea2683 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Wed, 23 Sep 2026 14:00:01 +0000
Subject: [PATCH 18/19] Change lowering for 64b vector_repeats
This avoids VECTOR_CONCAT and directly exposes a d -> z.d splat
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 11 +++++------
llvm/test/CodeGen/AArch64/sve-vector-repeat.ll | 12 ++++--------
2 files changed, 9 insertions(+), 14 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 4c8c68d98b7b8..9b0e0d9612438 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -17849,12 +17849,11 @@ SDValue AArch64TargetLowering::LowerVECTOR_REPEAT(SDValue Op,
// Repeat into a packed container before extracting the low lanes, which
// places the result elements at the spacing required by the unpacked type.
- EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
- EVT PackedSrcVT = SrcVT.getDoubleNumVectorElementsVT(*DAG.getContext());
- SDValue PackedSrc =
- DAG.getNode(ISD::CONCAT_VECTORS, DL, PackedSrcVT, Src, Src);
- SDValue Broadcast = DAG.getNode(ISD::VECTOR_REPEAT, DL, PackedVT, PackedSrc);
- return DAG.getExtractSubvector(DL, VT, Broadcast, 0);
+ SDValue SrcAsScalar =
+ DAG.getExtractVectorElt(DL, MVT::i64, DAG.getBitcast(MVT::v1i64, Src), 0);
+ SDValue Splat = DAG.getSplat(MVT::nxv2i64, DL, SrcAsScalar);
+ EVT PackedVT = VT.getDoubleNumVectorElementsVT(*DAG.getContext());
+ return DAG.getExtractSubvector(DL, VT, DAG.getBitcast(PackedVT, Splat), 0);
}
SDValue AArch64TargetLowering::LowerINSERT_SUBVECTOR(SDValue Op,
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-repeat.ll b/llvm/test/CodeGen/AArch64/sve-vector-repeat.ll
index 1d21957ea7c7b..5d2f550e9b2f4 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-repeat.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-repeat.ll
@@ -230,8 +230,7 @@ define <vscale x 4 x half> @repeat_double_f16_to_double_sve(<4 x half> %a) {
; CHECK-LABEL: repeat_double_f16_to_double_sve:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
-; CHECK-NEXT: mov v0.d[1], v0.d[0]
-; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: mov z0.d, d0
; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: ret
%out = call <vscale x 4 x half> @llvm.vector.repeat.nxv4f16(<4 x half> %a)
@@ -243,8 +242,7 @@ define <vscale x 2 x half> @repeat_2f16_to_nxv2f16(<4 x half> %a) {
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
; CHECK-NEXT: dup v0.2s, v0.s[0]
-; CHECK-NEXT: mov v0.d[1], v0.d[0]
-; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: mov z0.d, d0
; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: uunpklo z0.d, z0.s
; CHECK-NEXT: ret
@@ -301,8 +299,7 @@ define <vscale x 2 x float> @repeat_double_f32_to_nxv2f32(<2 x float> %a) {
; CHECK-LABEL: repeat_double_f32_to_nxv2f32:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
-; CHECK-NEXT: mov v0.d[1], v0.d[0]
-; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: mov z0.d, d0
; CHECK-NEXT: uunpklo z0.d, z0.s
; CHECK-NEXT: ret
%out = call <vscale x 2 x float> @llvm.vector.repeat.nxv2f32.v2f32(<2 x float> %a)
@@ -325,8 +322,7 @@ define <vscale x 4 x bfloat> @repeat_double_bf16_to_nxv4bf16(<4 x bfloat> %a) #0
; CHECK-LABEL: repeat_double_bf16_to_nxv4bf16:
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
-; CHECK-NEXT: mov v0.d[1], v0.d[0]
-; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: mov z0.d, d0
; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: ret
%out = call <vscale x 4 x bfloat> @llvm.vector.repeat.nxv4bf16.v4bf16(<4 x bfloat> %a)
>From b6c53a99393234359080dd53c6bb5039de2a318f Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Thu, 24 Sep 2026 08:32:20 +0000
Subject: [PATCH 19/19] Comments
---
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 2 +-
llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp | 7 +++----
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 7 +++----
3 files changed, 7 insertions(+), 9 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index 1cab31267838a..b638d00673490 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -318,7 +318,6 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue PromoteIntRes_MLOAD(MaskedLoadSDNode *N);
SDValue PromoteIntRes_MGATHER(MaskedGatherSDNode *N);
SDValue PromoteIntRes_VECTOR_COMPRESS(SDNode *N);
-
SDValue PromoteIntRes_Overflow(SDNode *N);
SDValue PromoteIntRes_FFREXP(SDNode *N);
SDValue PromoteIntRes_SADDSUBO(SDNode *N, unsigned ResNo);
@@ -980,6 +979,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue SplitVecOp_UnaryOp(SDNode *N);
SDValue SplitVecOp_TruncateHelper(SDNode *N);
SDValue SplitVecOp_VECTOR_COMPRESS(SDNode *N, unsigned OpNo);
+
SDValue SplitVecOp_BITCAST(SDNode *N);
SDValue SplitVecOp_INSERT_SUBVECTOR(SDNode *N, unsigned OpNo);
SDValue SplitVecOp_EXTRACT_SUBVECTOR(SDNode *N);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 51c38c2704b99..97d613ee6ac5f 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -8303,18 +8303,17 @@ SDValue DAGTypeLegalizer::WidenVecOp_VECTOR_REPEAT(SDNode *N) {
SrcVT.getVectorElementCount()))
report_fatal_error(
"Cannot widen VECTOR_REPEAT operand to an ElementCount that's not "
- "a multiple of the input ElementCount.");
+ "a known scalar multiple of the input ElementCount.");
// Repeat the original source because the extra lanes of its widened value
// are unspecified.
unsigned NumConcat =
- WidenedSrcVT.getVectorElementCount().getKnownScalarFactor(
- SrcVT.getVectorElementCount());
+ WidenedSrcVT.getVectorNumElements() / SrcVT.getVectorNumElements();
SmallVector<SDValue, 8> Ops(NumConcat, Src);
SDValue WidenedSrc = DAG.getNode(ISD::CONCAT_VECTORS, DL, WidenedSrcVT, Ops);
EVT WidenedVT = VT.changeVectorElementCount(
*DAG.getContext(),
- ElementCount::getScalable(WidenedSrcVT.getVectorMinNumElements()));
+ ElementCount::getScalable(WidenedSrcVT.getVectorNumElements()));
SDValue Widened = DAG.getNode(ISD::VECTOR_REPEAT, DL, WidenedVT, WidenedSrc);
return DAG.getExtractSubvector(DL, VT, Widened, 0);
}
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 9b0e0d9612438..178f3e0cd0311 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -1995,10 +1995,9 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
MVT::nxv8f16, MVT::nxv4f32, MVT::nxv2f64, MVT::nxv8bf16})
setOperationAction(ISD::VECTOR_REPEAT, VT, Legal);
- // VECTOR_REPEAT to unpacked SVE types require explicit unpacking to add
- // spacing between elements.
- for (auto VT : {MVT::nxv2f16, MVT::nxv4f16, MVT::nxv2f32, MVT::nxv2bf16,
- MVT::nxv4bf16})
+ // VECTOR_REPEAT to legal unpacked SVE types require explicit unpacking to
+ // add spacing between elements.
+ for (auto VT : {MVT::nxv4f16, MVT::nxv2f32, MVT::nxv4bf16})
setOperationAction(ISD::VECTOR_REPEAT, VT, Custom);
if (Subtarget->hasSVEB16B16() &&
More information about the llvm-commits
mailing list