[llvm] [IR][REVEC] Define llvm.vector.broadcast intrinsic (PR #208212)
Gaƫtan Bossu via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 24 03:51:00 PDT 2026
https://github.com/gbossu updated https://github.com/llvm/llvm-project/pull/208212
>From eaa902917a66f68db7056b77037cbb269fa290e7 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Fri, 3 Jul 2026 11:25:48 +0000
Subject: [PATCH 01/12] [IR] Define llvm.vector.broadcast intrinsic
This is used broadcast a smaller vector into a wider one. The patch adds
basic legalisation support and ISel for AArch64.
---
llvm/include/llvm/CodeGen/ISDOpcodes.h | 5 +
llvm/include/llvm/IR/Intrinsics.td | 7 +
.../include/llvm/Target/TargetSelectionDAG.td | 4 +
.../SelectionDAG/LegalizeIntegerTypes.cpp | 30 ++
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 2 +
.../lib/CodeGen/SelectionDAG/SelectionDAG.cpp | 12 +
.../SelectionDAG/SelectionDAGBuilder.cpp | 6 +
.../SelectionDAG/SelectionDAGDumper.cpp | 1 +
llvm/lib/Target/AArch64/SVEInstrFormats.td | 42 ++-
.../sve-vector-broadcast-unsupported.ll | 9 +
.../CodeGen/AArch64/sve-vector-broadcast.ll | 318 ++++++++++++++++++
11 files changed, 430 insertions(+), 6 deletions(-)
create mode 100644 llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
create mode 100644 llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index 23ef5d22963ae..2e850da5d49b5 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -636,6 +636,11 @@ enum NodeType {
/// Result[J] = EXTRACT_SUBVECTOR(Interleaved, J * getVectorMinNumElements())
VECTOR_INTERLEAVE,
+ /// VECTOR_BROADCAST(SRC_SUBVEC)
+ /// Duplicate a vector in a larger vector. The element count of the result
+ /// type is expected to be a multiple of the input vector's.
+ VECTOR_BROADCAST,
+
/// VECTOR_REVERSE(VECTOR) - Returns a vector, of the same type as VECTOR,
/// whose elements are shuffled using the following algorithm:
/// RESULT[i] = VECTOR[VECTOR.ElementCount - 1 - i]
diff --git a/llvm/include/llvm/IR/Intrinsics.td b/llvm/include/llvm/IR/Intrinsics.td
index 37c9c783465d6..25cb948fc4b8f 100644
--- a/llvm/include/llvm/IR/Intrinsics.td
+++ b/llvm/include/llvm/IR/Intrinsics.td
@@ -3088,6 +3088,13 @@ foreach n = 2...8 in {
[IntrNoMem, IntrSpeculatable]>;
}
+// vector_broadcast( SrcVector )
+// Broadcast a vector in a larger one.
+// This is the equivalent of a vector splat for vector types.
+def int_vector_broadcast : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
+ [llvm_anyvector_ty],
+ [IntrNoMem, IntrSpeculatable]>;
+
//===-------------- Intrinsics to perform partial reduction ---------------===//
def int_vector_partial_reduce_add : DefaultAttrsIntrinsic<[LLVMMatchType<0>],
diff --git a/llvm/include/llvm/Target/TargetSelectionDAG.td b/llvm/include/llvm/Target/TargetSelectionDAG.td
index 37b7b8716bcd8..d67a25e189efa 100644
--- a/llvm/include/llvm/Target/TargetSelectionDAG.td
+++ b/llvm/include/llvm/Target/TargetSelectionDAG.td
@@ -926,6 +926,10 @@ def vector_insert_subvec : SDNode<"ISD::INSERT_SUBVECTOR",
def extract_subvector : SDNode<"ISD::EXTRACT_SUBVECTOR", SDTSubVecExtract, []>;
def insert_subvector : SDNode<"ISD::INSERT_SUBVECTOR", SDTSubVecInsert, []>;
+def vector_broadcast : SDNode<"ISD::VECTOR_BROADCAST",
+ SDTypeProfile<1, 1, [SDTCisVec<1>, SDTCisVec<0>]>,
+ []>;
+
def find_last_active
: SDNode<"ISD::VECTOR_FIND_LAST_ACTIVE",
SDTypeProfile<1, 1, [SDTCisInt<0>, SDTCisVec<1>]>, []>;
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
index 2fee0b280ed9b..35d1e248a052f 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeIntegerTypes.cpp
@@ -139,6 +139,9 @@ void DAGTypeLegalizer::PromoteIntegerResult(SDNode *N, unsigned ResNo) {
case ISD::VECTOR_SPLICE_RIGHT:
Res = PromoteIntRes_VECTOR_SPLICE(N);
break;
+ case ISD::VECTOR_BROADCAST:
+ Res = PromoteIntRes_VECTOR_BROADCAST(N);
+ break;
case ISD::VECTOR_INTERLEAVE:
case ISD::VECTOR_DEINTERLEAVE:
Res = PromoteIntRes_VECTOR_INTERLEAVE_DEINTERLEAVE(N);
@@ -2304,6 +2307,9 @@ bool DAGTypeLegalizer::PromoteIntegerOperand(SDNode *N, unsigned OpNo) {
case ISD::PARTIAL_REDUCE_SUMLA:
Res = PromoteIntOp_PARTIAL_REDUCE_MLA(N);
break;
+ case ISD::VECTOR_BROADCAST:
+ Res = PromoteIntOp_VECTOR_BROADCAST(N);
+ break;
case ISD::LOOP_DEPENDENCE_RAW_MASK:
case ISD::LOOP_DEPENDENCE_WAR_MASK:
Res = PromoteIntOp_LOOP_DEPENDENCE_MASK(N);
@@ -3203,6 +3209,17 @@ SDValue DAGTypeLegalizer::PromoteIntOp_LOOP_DEPENDENCE_MASK(SDNode *N) {
return SDValue(DAG.UpdateNodeOperands(N, NewOps), 0);
}
+SDValue DAGTypeLegalizer::PromoteIntOp_VECTOR_BROADCAST(SDNode *N) {
+ SDLoc DL(N);
+ SDValue Src = GetPromotedInteger(N->getOperand(0));
+ EVT SrcVT = Src.getValueType();
+ EVT OrigVT = N->getValueType(0);
+ EVT NewVT = EVT::getVectorVT(*DAG.getContext(), SrcVT.getVectorElementType(),
+ OrigVT.getVectorElementCount());
+ SDValue Res = DAG.getNode(ISD::VECTOR_BROADCAST, DL, NewVT, Src);
+ return DAG.getNode(ISD::TRUNCATE, DL, OrigVT, Res);
+}
+
//===----------------------------------------------------------------------===//
// Integer Result Expansion
//===----------------------------------------------------------------------===//
@@ -6271,6 +6288,19 @@ SDValue DAGTypeLegalizer::PromoteIntRes_VECTOR_SPLICE(SDNode *N) {
return DAG.getNode(N->getOpcode(), dl, OutVT, V0, V1, N->getOperand(2));
}
+SDValue DAGTypeLegalizer::PromoteIntRes_VECTOR_BROADCAST(SDNode *N) {
+ SDLoc DL(N);
+
+ EVT OutVT = N->getValueType(0);
+ EVT NOutVT = TLI.getTypeToTransformTo(*DAG.getContext(), OutVT);
+ assert(NOutVT.isVector() && "This type must be promoted to a vector type");
+ EVT NInVT = N->getOperand(0).getValueType().changeVectorElementType(
+ *DAG.getContext(), NOutVT.getVectorElementType());
+
+ SDValue Op = DAG.getNode(ISD::ANY_EXTEND, DL, NInVT, N->getOperand(0));
+ return DAG.getNode(N->getOpcode(), DL, NOutVT, Op);
+}
+
SDValue DAGTypeLegalizer::PromoteIntRes_VECTOR_INTERLEAVE_DEINTERLEAVE(SDNode *N) {
SDLoc DL(N);
unsigned Factor = N->getNumOperands();
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index 5050c3fe5ac22..dd4f15f07c46b 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -292,6 +292,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue PromoteIntRes_VECTOR_REVERSE(SDNode *N);
SDValue PromoteIntRes_VECTOR_SHUFFLE(SDNode *N);
SDValue PromoteIntRes_VECTOR_SPLICE(SDNode *N);
+ SDValue PromoteIntRes_VECTOR_BROADCAST(SDNode *N);
SDValue PromoteIntRes_VECTOR_INTERLEAVE_DEINTERLEAVE(SDNode *N);
SDValue PromoteIntRes_BUILD_VECTOR(SDNode *N);
SDValue PromoteIntRes_ScalarOp(SDNode *N);
@@ -426,6 +427,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue PromoteIntOp_UnaryBooleanVectorOp(SDNode *N, unsigned OpNo);
SDValue PromoteIntOp_GET_ACTIVE_LANE_MASK(SDNode *N);
SDValue PromoteIntOp_PARTIAL_REDUCE_MLA(SDNode *N);
+ SDValue PromoteIntOp_VECTOR_BROADCAST(SDNode *N);
SDValue PromoteIntOp_LOOP_DEPENDENCE_MASK(SDNode *N);
SDValue PromoteIntOp_MaskedBinOp(SDNode *N, unsigned OpNo);
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
index b9d0ed344663b..cee3ca37e8e4f 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
@@ -9127,6 +9127,18 @@ SDValue SelectionDAG::getNode(unsigned Opcode, const SDLoc &DL, EVT VT,
}
break;
}
+ case ISD::VECTOR_BROADCAST: {
+ [[maybe_unused]] EVT InputVT = N1.getValueType();
+ assert(InputVT.isVector() && VT.isVector() &&
+ "Expected the input and output of the VECTOR_BROADCAST node to be "
+ "vectors!");
+ assert(VT.getVectorElementCount().hasKnownScalarFactor(
+ InputVT.getVectorElementCount()) &&
+ "Expected the element count of the output of the VECTOR_BROADCAST "
+ "node to be a positive integer multiple of the element count of the "
+ "source operand!");
+ break;
+ }
case ISD::BITCAST:
// Fold bit_convert nodes from a type to themselves.
if (N1.getValueType() == VT)
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
index dea008fd252a9..90dd08090a723 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGBuilder.cpp
@@ -8610,6 +8610,12 @@ void SelectionDAGBuilder::visitIntrinsicCall(const CallInst &I,
case Intrinsic::vector_deinterleave8:
visitVectorDeinterleave(I, 8);
return;
+ case Intrinsic::vector_broadcast: {
+ SDValue Vec = getValue(I.getOperand(0));
+ EVT ResultVT = TLI.getValueType(DAG.getDataLayout(), I.getType());
+ setValue(&I, DAG.getNode(ISD::VECTOR_BROADCAST, sdl, ResultVT, Vec));
+ return;
+ }
case Intrinsic::experimental_vector_compress:
setValue(&I, DAG.getNode(ISD::VECTOR_COMPRESS, sdl,
getValue(I.getArgOperand(0)).getValueType(),
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
index 028f0b9486c01..3aaab645de639 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAGDumper.cpp
@@ -360,6 +360,7 @@ std::string SDNode::getOperationName(const SelectionDAG *G) const {
case ISD::EXTRACT_SUBVECTOR: return "extract_subvector";
case ISD::VECTOR_DEINTERLEAVE: return "vector_deinterleave";
case ISD::VECTOR_INTERLEAVE: return "vector_interleave";
+ case ISD::VECTOR_BROADCAST: return "vector_broadcast";
case ISD::SCALAR_TO_VECTOR: return "scalar_to_vector";
case ISD::VECTOR_SHUFFLE: return "vector_shuffle";
case ISD::VECTOR_SPLICE_LEFT: return "vector_splice_left";
diff --git a/llvm/lib/Target/AArch64/SVEInstrFormats.td b/llvm/lib/Target/AArch64/SVEInstrFormats.td
index 557f0b59d749c..f818f33154b61 100644
--- a/llvm/lib/Target/AArch64/SVEInstrFormats.td
+++ b/llvm/lib/Target/AArch64/SVEInstrFormats.td
@@ -1553,12 +1553,21 @@ multiclass sve_int_perm_dup_i<string asm> {
// Duplicate an extracted vector element across a vector.
- def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (nxv16i8 ZPR:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) ZPR:$vec, sve_elm_idx_extdup_b:$index)>;
- def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (v16i8 V128:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_b:$index)>;
- def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (v8i8 V64:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_b:$index)>;
+ foreach VT = [nxv16i8] in {
+ def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.Packed ZPR:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) ZPR:$vec, sve_elm_idx_extdup_b:$index)>;
+ def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.ZSub V128:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_b:$index)>;
+ def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_b:$index)>;
+
+ // Broadcast a whole 128-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
+ (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
+ // Broadcast a whole 64-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
+ (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
+ }
foreach VT = [nxv8i16, nxv2f16, nxv4f16, nxv8f16, nxv2bf16, nxv4bf16, nxv8bf16] in {
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.Packed ZPR:$vec), sve_elm_idx_extdup_h:$index)))),
@@ -1567,6 +1576,13 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _H) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_h:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_h:$index)))),
(!cast<Instruction>(NAME # _H) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_h:$index)>;
+
+ // Broadcast a whole 128-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
+ (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
+ // Broadcast a whole 64-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
+ (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
foreach VT = [nxv4i32, nxv2f32, nxv4f32 ] in {
@@ -1576,6 +1592,13 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _S) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_s:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_s:$index)))),
(!cast<Instruction>(NAME # _S) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_s:$index)>;
+
+ // Broadcast a whole 128-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
+ (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
+ // Broadcast a whole 64-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
+ (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
foreach VT = [nxv2i64, nxv2f64] in {
@@ -1585,6 +1608,13 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_d:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_d:$index)))),
(!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_d:$index)>;
+
+ // Broadcast a whole 128-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
+ (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
+ // Broadcast a whole 64-bit vector
+ def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
+ (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
// When extracting from an unpacked vector the index must be scaled to account
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
new file mode 100644
index 0000000000000..49152c702505a
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
@@ -0,0 +1,9 @@
+; RUN: not --crash llc -mtriple=aarch64-linux-gnu -mattr=+sve < %s 2>&1 | FileCheck %s
+
+; CHECK: LLVM ERROR: Do not know how to widen this operator's operand!
+
+; TODO: Support broadcasts from 32-bit vec
+define <vscale x 8 x half> @broadcast_single_f16(<2 x half> %a) {
+ %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<2 x half> %a)
+ ret <vscale x 8 x half> %out
+}
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
new file mode 100644
index 0000000000000..f0ef8b2c2e350
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -0,0 +1,318 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -mtriple=aarch64-linux-gnu -mattr=+sve < %s | FileCheck %s
+
+
+define <vscale x 16 x i8> @broadcast_quad_i8(<16 x i8> %a) {
+; CHECK-LABEL: broadcast_quad_i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v16i8(<16 x i8> %a)
+ ret <vscale x 16 x i8> %out
+}
+
+define <vscale x 16 x i8> @broadcast_double_i8(<8 x i8> %a) {
+; CHECK-LABEL: broadcast_double_i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %out = call <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v8i8(<8 x i8> %a)
+ ret <vscale x 16 x i8> %out
+}
+
+define <vscale x 8 x i16> @broadcast_double_i8_to_double_sve(<8 x i8> %a) {
+; CHECK-LABEL: broadcast_double_i8_to_double_sve:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ushll v0.8h, v0.8b, #0
+; CHECK-NEXT: ptrue p0.h
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: sxtb z0.h, p0/m, z0.h
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x i8> @llvm.vector.broadcast.nxv8i8.v8i8(<8 x i8> %a)
+ %out.legal = sext <vscale x 8 x i8> %out to <vscale x 8 x i16>
+ ret <vscale x 8 x i16> %out.legal
+}
+
+define <vscale x 8 x i16> @broadcast_quad_i16(<8 x i16> %a) {
+; CHECK-LABEL: broadcast_quad_i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x i16> @llvm.vector.broadcast.nxv8i16.v8i16(<8 x i16> %a)
+ ret <vscale x 8 x i16> %out
+}
+
+define <vscale x 8 x i16> @broadcast_double_i16(<4 x i16> %a) {
+; CHECK-LABEL: broadcast_double_i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x i16> @llvm.vector.broadcast.nxv8i16.v4i16(<4 x i16> %a)
+ ret <vscale x 8 x i16> %out
+}
+
+define <vscale x 4 x i32> @broadcast_double_i16_to_double_sve(<4 x i16> %a) {
+; CHECK-LABEL: broadcast_double_i16_to_double_sve:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ushll v0.4s, v0.4h, #0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.s, z0.s, #0xffff
+; CHECK-NEXT: ret
+ %out = call <vscale x 4 x i16> @llvm.vector.broadcast.nxv4i16.v4i16(<4 x i16> %a)
+ %out.legal = zext <vscale x 4 x i16> %out to <vscale x 4 x i32>
+ ret <vscale x 4 x i32> %out.legal
+}
+
+define <vscale x 4 x i32> @broadcast_quad_i32(<4 x i32> %a) {
+; CHECK-LABEL: broadcast_quad_i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 4 x i32> @llvm.vector.broadcast.nxv4i32.v4i32(<4 x i32> %a)
+ ret <vscale x 4 x i32> %out
+}
+
+define <vscale x 2 x i64> @broadcast_quad_i64(<2 x i64> %a) {
+; CHECK-LABEL: broadcast_quad_i64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v2i64(<2 x i64> %a)
+ ret <vscale x 2 x i64> %out
+}
+
+define <vscale x 8 x half> @broadcast_quad_f16(<8 x half> %a) {
+; CHECK-LABEL: broadcast_quad_f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<8 x half> %a)
+ ret <vscale x 8 x half> %out
+}
+
+define <vscale x 8 x half> @broadcast_double_f16(<4 x half> %a) {
+; CHECK-LABEL: broadcast_double_f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<4 x half> %a)
+ ret <vscale x 8 x half> %out
+}
+
+define <vscale x 4 x half> @broadcast_double_f16_to_double_sve(<4 x half> %a) {
+; CHECK-LABEL: broadcast_double_f16_to_double_sve:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<4 x half> %a)
+ ret <vscale x 4 x half> %out
+}
+
+define <vscale x 8 x bfloat> @broadcast_quad_bf16(<8 x bfloat> %a) #0 {
+; CHECK-LABEL: broadcast_quad_bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 8 x bfloat> @llvm.vector.broadcast.nxv8bf16.v8bf16(<8 x bfloat> %a)
+ ret <vscale x 8 x bfloat> %out
+}
+
+define <vscale x 4 x float> @broadcast_quad_f32(<4 x float> %a) {
+; CHECK-LABEL: broadcast_quad_f32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 4 x float> @llvm.vector.broadcast.nxv4f32.v4f32(<4 x float> %a)
+ ret <vscale x 4 x float> %out
+}
+
+define <vscale x 2 x double> @broadcast_quad_f64(<2 x double> %a) {
+; CHECK-LABEL: broadcast_quad_f64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: ret
+ %out = call <vscale x 2 x double> @llvm.vector.broadcast.nxv2f64.v2f64(<2 x double> %a)
+ ret <vscale x 2 x double> %out
+}
+
+; Predicates
+
+define <vscale x 16 x i1> @broadcast_v16i1_to_nxv16i1(<16 x i8> %a) {
+; CHECK-LABEL: broadcast_v16i1_to_nxv16i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: ptrue p0.b
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.b, z0.b, #0x1
+; CHECK-NEXT: cmpne p0.b, p0/z, z0.b, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <16 x i8> %a to <16 x i1>
+ %out = call <vscale x 16 x i1> @llvm.vector.broadcast.nxv16i1.v16i1(<16 x i1> %a.legal)
+ ret <vscale x 16 x i1> %out
+}
+
+define <vscale x 16 x i1> @broadcast_v8i1_to_nxv16i1(<8 x i8> %a) {
+; CHECK-LABEL: broadcast_v8i1_to_nxv16i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: ptrue p0.b
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: and z0.b, z0.b, #0x1
+; CHECK-NEXT: cmpne p0.b, p0/z, z0.b, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <8 x i8> %a to <8 x i1>
+ %out = call <vscale x 16 x i1> @llvm.vector.broadcast.nxv16i1.v8i1(<8 x i1> %a.legal)
+ ret <vscale x 16 x i1> %out
+}
+
+define <vscale x 8 x i1> @broadcast_v8i1_double_to_nxv8i1(<8 x i8> %a) {
+; CHECK-LABEL: broadcast_v8i1_double_to_nxv8i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ushll v0.8h, v0.8b, #0
+; CHECK-NEXT: ptrue p0.h
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.h, z0.h, #0x1
+; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <8 x i8> %a to <8 x i1>
+ %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v8i1(<8 x i1> %a.legal)
+ ret <vscale x 8 x i1> %out
+}
+
+define <vscale x 8 x i1> @broadcast_v8i1_quad_to_nxv8i1(<8 x i16> %a) {
+; CHECK-LABEL: broadcast_v8i1_quad_to_nxv8i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: ptrue p0.h
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.h, z0.h, #0x1
+; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <8 x i16> %a to <8 x i1>
+ %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v8i1(<8 x i1> %a.legal)
+ ret <vscale x 8 x i1> %out
+}
+
+define <vscale x 8 x i1> @broadcast_v4i1_double_to_nxv8i1(<4 x i16> %a) {
+; CHECK-LABEL: broadcast_v4i1_double_to_nxv8i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: ptrue p0.h
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: and z0.h, z0.h, #0x1
+; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <4 x i16> %a to <4 x i1>
+ %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v4i1(<4 x i1> %a.legal)
+ ret <vscale x 8 x i1> %out
+}
+
+define <vscale x 8 x i1> @broadcast_v4i1_quad_to_nxv8i1(<4 x i32> %a) {
+; CHECK-LABEL: broadcast_v4i1_quad_to_nxv8i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: xtn v0.4h, v0.4s
+; CHECK-NEXT: ptrue p0.h
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: and z0.h, z0.h, #0x1
+; CHECK-NEXT: cmpne p0.h, p0/z, z0.h, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <4 x i32> %a to <4 x i1>
+ %out = call <vscale x 8 x i1> @llvm.vector.broadcast.nxv8i1.v4i1(<4 x i1> %a.legal)
+ ret <vscale x 8 x i1> %out
+}
+
+define <vscale x 4 x i1> @broadcast_v4i1_double_to_nxv4i1(<4 x i16> %a) {
+; CHECK-LABEL: broadcast_v4i1_double_to_nxv4i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ushll v0.4s, v0.4h, #0
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.s, z0.s, #0x1
+; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <4 x i16> %a to <4 x i1>
+ %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v4i1(<4 x i1> %a.legal)
+ ret <vscale x 4 x i1> %out
+}
+
+define <vscale x 4 x i1> @broadcast_v4i1_quad_to_nxv4i1(<4 x i32> %a) {
+; CHECK-LABEL: broadcast_v4i1_quad_to_nxv4i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.s, z0.s, #0x1
+; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <4 x i32> %a to <4 x i1>
+ %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v4i1(<4 x i1> %a.legal)
+ ret <vscale x 4 x i1> %out
+}
+
+define <vscale x 4 x i1> @broadcast_v2i1_double_to_nxv4i1(<2 x i32> %a) {
+; CHECK-LABEL: broadcast_v2i1_double_to_nxv4i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: and z0.s, z0.s, #0x1
+; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <2 x i32> %a to <2 x i1>
+ %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v2i1(<2 x i1> %a.legal)
+ ret <vscale x 4 x i1> %out
+}
+
+define <vscale x 4 x i1> @broadcast_v2i1_quad_to_nxv4i1(<2 x i64> %a) {
+; CHECK-LABEL: broadcast_v2i1_quad_to_nxv4i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: xtn v0.2s, v0.2d
+; CHECK-NEXT: ptrue p0.s
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: and z0.s, z0.s, #0x1
+; CHECK-NEXT: cmpne p0.s, p0/z, z0.s, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <2 x i64> %a to <2 x i1>
+ %out = call <vscale x 4 x i1> @llvm.vector.broadcast.nxv4i1.v2i1(<2 x i1> %a.legal)
+ ret <vscale x 4 x i1> %out
+}
+
+define <vscale x 2 x i1> @broadcast_v2i1_double_to_nxv2i1(<2 x i32> %a) {
+; CHECK-LABEL: broadcast_v2i1_double_to_nxv2i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ushll v0.2d, v0.2s, #0
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.d, z0.d, #0x1
+; CHECK-NEXT: cmpne p0.d, p0/z, z0.d, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <2 x i32> %a to <2 x i1>
+ %out = call <vscale x 2 x i1> @llvm.vector.broadcast.nxv2i1.v2i1(<2 x i1> %a.legal)
+ ret <vscale x 2 x i1> %out
+}
+
+define <vscale x 2 x i1> @broadcast_v2i1_quad_to_nxv2i1(<2 x i64> %a) {
+; CHECK-LABEL: broadcast_v2i1_quad_to_nxv2i1:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: ptrue p0.d
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: and z0.d, z0.d, #0x1
+; CHECK-NEXT: cmpne p0.d, p0/z, z0.d, #0
+; CHECK-NEXT: ret
+ %a.legal = trunc <2 x i64> %a to <2 x i1>
+ %out = call <vscale x 2 x i1> @llvm.vector.broadcast.nxv2i1.v2i1(<2 x i1> %a.legal)
+ ret <vscale x 2 x i1> %out
+}
>From f399c1cead3e8538bb9d9ae3b7699ea1e8a64d81 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Fri, 7 Aug 2026 09:28:26 +0000
Subject: [PATCH 02/12] Support splitting results
---
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 1 +
.../SelectionDAG/LegalizeVectorTypes.cpp | 46 +++++++++++++++++
.../CodeGen/AArch64/sve-vector-broadcast.ll | 51 +++++++++++++++++++
3 files changed, 98 insertions(+)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index dd4f15f07c46b..422cf2dd70d5f 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -966,6 +966,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
void SplitVecRes_ScalarOp(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_STEP_VECTOR(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_SETCC(SDNode *N, SDValue &Lo, SDValue &Hi);
+ void SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_VECTOR_REVERSE(SDNode *N, SDValue &Lo, SDValue &Hi);
void SplitVecRes_VECTOR_SHUFFLE(ShuffleVectorSDNode *N, SDValue &Lo,
SDValue &Hi);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 5f01a97664073..02ab49693bfba 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -1425,6 +1425,9 @@ void DAGTypeLegalizer::SplitVectorResult(SDNode *N, unsigned ResNo) {
case ISD::VP_SETCC:
SplitVecRes_SETCC(N, Lo, Hi);
break;
+ case ISD::VECTOR_BROADCAST:
+ SplitVecRes_VECTOR_BROADCAST(N, Lo, Hi);
+ break;
case ISD::VECTOR_REVERSE:
SplitVecRes_VECTOR_REVERSE(N, Lo, Hi);
break;
@@ -3535,6 +3538,49 @@ void DAGTypeLegalizer::SplitVecRes_FP_TO_XINT_SAT(SDNode *N, SDValue &Lo,
Hi = DAG.getNode(N->getOpcode(), dl, DstVTHi, SrcHi, N->getOperand(1));
}
+void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
+ SDValue &Hi) {
+ EVT VT = N->getValueType(0);
+ SDValue Src = N->getOperand(0);
+ EVT SrcVT = Src.getValueType();
+ EVT LoVT, HiVT;
+ std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(VT);
+ assert(LoVT == HiVT && "Expected equal split types");
+
+ // Simple case: The split type is same as SrcVT.
+ if (LoVT == SrcVT) {
+ Lo = Hi = Src;
+ return;
+ }
+
+ // Second case: Src is known to be wider than LoVT.
+ SDLoc DL(N);
+ if (LoVT.getVectorMinNumElements() >= SrcVT.getVectorMinNumElements()) {
+ Lo = Hi = DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Src);
+ return;
+ }
+
+ // Final case: VT is scalable and has the same minimum EC.
+ // Use smaller even/odd source vectors so their broadcasts can be
+ // reinterleaved in the original lane order for every value of vscale.
+ assert(VT.getVectorMinNumElements() == SrcVT.getVectorMinNumElements() &&
+ VT.isScalableVector());
+ SDValue SrcLo, SrcHi;
+ std::tie(SrcLo, SrcHi) = DAG.SplitVector(Src, DL);
+ EVT SrcSplitVT = SrcLo.getValueType();
+ SDValue Deinterleaved =
+ DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL,
+ DAG.getVTList(SrcSplitVT, SrcSplitVT), SrcLo, SrcHi);
+ SDValue Even =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(0));
+ SDValue Odd =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(1));
+ SDValue Interleaved = DAG.getNode(ISD::VECTOR_INTERLEAVE, DL,
+ DAG.getVTList(LoVT, HiVT), Even, Odd);
+ Lo = Interleaved.getValue(0);
+ Hi = Interleaved.getValue(1);
+}
+
void DAGTypeLegalizer::SplitVecRes_VECTOR_REVERSE(SDNode *N, SDValue &Lo,
SDValue &Hi) {
SDValue InLo, InHi;
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
index f0ef8b2c2e350..7f1a37529f912 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -45,6 +45,45 @@ define <vscale x 8 x i16> @broadcast_quad_i16(<8 x i16> %a) {
ret <vscale x 8 x i16> %out
}
+define <vscale x 16 x i8> @broadcast_quad_i16_to_wide_sve(<8 x i16> %a) {
+; CHECK-LABEL: broadcast_quad_i16_to_wide_sve:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: uzp1 z0.b, z0.b, z0.b
+; CHECK-NEXT: ret
+ %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.v8i16(<8 x i16> %a)
+ %out.legal = trunc <vscale x 16 x i16> %out to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %out.legal
+}
+
+define <vscale x 16 x i8> @broadcast_wide_i16(<8 x i16> %a.lo, <8 x i16> %a.hi) {
+; CHECK-LABEL: broadcast_wide_i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: uzp2 v2.8h, v0.8h, v1.8h
+; CHECK-NEXT: uzp1 v0.8h, v0.8h, v1.8h
+; CHECK-NEXT: mov z1.q, q2
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: zip2 z2.h, z0.h, z1.h
+; CHECK-NEXT: zip1 z0.h, z0.h, z1.h
+; CHECK-NEXT: uzp1 z0.b, z0.b, z2.b
+; CHECK-NEXT: ret
+ %a = shufflevector <8 x i16> %a.lo, <8 x i16> %a.hi,
+ <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+ %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.v16i16(<16 x i16> %a)
+ %out.legal = trunc <vscale x 16 x i16> %out to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %out.legal
+}
+
+define <vscale x 16 x i16> @broadcast_nxv8i16_to_nxv16i16(<vscale x 8 x i16> %a) {
+; CHECK-LABEL: broadcast_nxv8i16_to_nxv16i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: mov z1.d, z0.d
+; CHECK-NEXT: ret
+ %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.nxv8i16(<vscale x 8 x i16> %a)
+ ret <vscale x 16 x i16> %out
+}
+
define <vscale x 8 x i16> @broadcast_double_i16(<4 x i16> %a) {
; CHECK-LABEL: broadcast_double_i16:
; CHECK: // %bb.0:
@@ -67,6 +106,18 @@ define <vscale x 4 x i32> @broadcast_double_i16_to_double_sve(<4 x i16> %a) {
ret <vscale x 4 x i32> %out.legal
}
+define <vscale x 16 x i8> @broadcast_double_i16_to_wide_sve(<4 x i16> %a) {
+; CHECK-LABEL: broadcast_double_i16_to_wide_sve:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uzp1 z0.b, z0.b, z0.b
+; CHECK-NEXT: ret
+ %out = call <vscale x 16 x i16> @llvm.vector.broadcast.nxv16i16.v4i16(<4 x i16> %a)
+ %out.legal = trunc <vscale x 16 x i16> %out to <vscale x 16 x i8>
+ ret <vscale x 16 x i8> %out.legal
+}
+
define <vscale x 4 x i32> @broadcast_quad_i32(<4 x i32> %a) {
; CHECK-LABEL: broadcast_quad_i32:
; CHECK: // %bb.0:
>From 59040f8f8bc7aaf09da4b920a27a0efa3b73b31a Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Fri, 7 Aug 2026 09:30:13 +0000
Subject: [PATCH 03/12] Support expansion to shufflevector
---
llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp | 19 ++++++
.../Target/AArch64/AArch64ISelLowering.cpp | 3 +
llvm/test/CodeGen/AArch64/vector-broadcast.ll | 68 +++++++++++++++++++
3 files changed, 90 insertions(+)
create mode 100644 llvm/test/CodeGen/AArch64/vector-broadcast.ll
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
index 4a2154102e60b..299f5cb466072 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeDAG.cpp
@@ -3694,6 +3694,25 @@ bool SelectionDAGLegalize::ExpandNode(SDNode *Node) {
case ISD::INSERT_VECTOR_ELT:
Results.push_back(ExpandINSERT_VECTOR_ELT(SDValue(Node, 0)));
break;
+ case ISD::VECTOR_BROADCAST: {
+ EVT VT = Node->getValueType(0);
+ EVT SrcVT = Node->getOperand(0).getValueType();
+ assert(VT.isFixedLengthVector() && SrcVT.isFixedLengthVector() &&
+ "Can only expand broadcasts of fixed-length vectors");
+
+ SDValue Src = Node->getOperand(0);
+ SDValue PaddedSrc = DAG.getInsertSubvector(dl, DAG.getUNDEF(VT), Src, 0);
+
+ // Create a shuffle mask that duplicates Src.
+ unsigned NumElts = VT.getVectorNumElements();
+ unsigned SrcNumElts = SrcVT.getVectorNumElements();
+ SmallVector<int, 8> Mask;
+ for (unsigned I = 0; I != NumElts; ++I)
+ Mask.push_back(I % SrcNumElts);
+ Results.push_back(
+ DAG.getVectorShuffle(VT, dl, PaddedSrc, DAG.getUNDEF(VT), Mask));
+ break;
+ }
case ISD::VECTOR_SHUFFLE: {
SmallVector<int, 32> NewMask;
ArrayRef<int> Mask = cast<ShuffleVectorSDNode>(Node)->getMask();
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index b01dc35531f4b..1131666e9aeb9 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -728,6 +728,9 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::ROTR, VT, Expand);
}
+ for (MVT VT : MVT::fixedlen_vector_valuetypes())
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Expand);
+
// AArch64 doesn't have i32 MULH{S|U}.
setOperationAction(ISD::MULHU, MVT::i32, Expand);
setOperationAction(ISD::MULHS, MVT::i32, Expand);
diff --git a/llvm/test/CodeGen/AArch64/vector-broadcast.ll b/llvm/test/CodeGen/AArch64/vector-broadcast.ll
new file mode 100644
index 0000000000000..b5a019bd9d5bf
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/vector-broadcast.ll
@@ -0,0 +1,68 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=aarch64-linux-gnu < %s | FileCheck %s
+
+; See sve-vector-broadcast.ll for scalable types.
+
+define <16 x i8> @broadcast_v8i8_to_v16i8(<8 x i8> %vec) {
+; CHECK-LABEL: broadcast_v8i8_to_v16i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: ret
+ %result = call <16 x i8> @llvm.vector.broadcast.v16i8.v8i8(<8 x i8> %vec)
+ ret <16 x i8> %result
+}
+
+define <16 x i8> @broadcast_v4i8_to_v16i8(<4 x i16> %wide.vec) {
+; CHECK-LABEL: broadcast_v4i8_to_v16i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: uzp1 v0.16b, v0.16b, v0.16b
+; CHECK-NEXT: ret
+ %vec = trunc <4 x i16> %wide.vec to <4 x i8>
+ %result = call <16 x i8> @llvm.vector.broadcast.v16i8.v4i8(<4 x i8> %vec)
+ ret <16 x i8> %result
+}
+
+define <8 x i16> @broadcast_v4i16_to_v8i16(<4 x i16> %vec) {
+; CHECK-LABEL: broadcast_v4i16_to_v8i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: ret
+ %result = call <8 x i16> @llvm.vector.broadcast.v8i16.v4i16(<4 x i16> %vec)
+ ret <8 x i16> %result
+}
+
+define <8 x i16> @broadcast_v2i16_to_v8i16(<2 x i32> %wide.vec) {
+; CHECK-LABEL: broadcast_v2i16_to_v8i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: uzp1 v0.8h, v0.8h, v0.8h
+; CHECK-NEXT: ret
+ %vec = trunc <2 x i32> %wide.vec to <2 x i16>
+ %result = call <8 x i16> @llvm.vector.broadcast.v8i16.v2i16(<2 x i16> %vec)
+ ret <8 x i16> %result
+}
+
+define <4 x i32> @broadcast_v2i32_to_v4i32(<2 x i32> %vec) {
+; CHECK-LABEL: broadcast_v2i32_to_v4i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: ret
+ %result = call <4 x i32> @llvm.vector.broadcast.v4i32.v2i32(<2 x i32> %vec)
+ ret <4 x i32> %result
+}
+
+define <4 x float> @broadcast_v2f32_to_v4f32(<2 x float> %vec) {
+; CHECK-LABEL: broadcast_v2f32_to_v4f32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2d, v0.d[0]
+; CHECK-NEXT: ret
+ %result = call <4 x float> @llvm.vector.broadcast.v4f32.v2f32(<2 x float> %vec)
+ ret <4 x float> %result
+}
>From d717b71b907807420825298a1aa1037546fbbd08 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Fri, 7 Aug 2026 10:19:39 +0000
Subject: [PATCH 04/12] Update LangRef
---
llvm/docs/LangRef.md | 26 ++++++++++++++++++++++++++
1 file changed, 26 insertions(+)
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index d57f619042f22..fb01472f9386a 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -20474,6 +20474,32 @@ runtime, then the result vector is a {ref}`poison value <poisonvalues>`. The
`idx` parameter must be a vector index constant type (for most targets this
will be an integer pointer type).
+#### '`llvm.vector.broadcast`' Intrinsic
+
+##### Syntax:
+This is an overloaded intrinsic.
+
+```
+declare <8 x i32> @llvm.vector.broadcast.v8i32.v2i32(<2 x i32> %vec)
+declare <vscale x 16 x i8> @llvm.vector.broadcast.nxv16i8.v16i8(<16 x i8> %vec)
+```
+
+##### Overview:
+
+The '`llvm.vector.broadcast.*`' intrinsic repeatedly copies the elements of
+the source vector, in order, until the result vector is filled. For example,
+broadcasting `<A, B>` to a vector with eight elements produces
+`<A, B, A, B, A, B, A, B>`.
+This intrinsic works for both fixed and scalable vectors but the recommended way
+to express this operation for fixed-width vectors is still to use a
+`shufflevector`, as that may allow for more optimization opportunities.
+
+##### Arguments:
+
+The argument and result must be vectors with the same element type. The element
+count of the result must be a known multiple of the element count of the
+argument.
+
#### '`llvm.vector.reverse`' Intrinsic
##### Syntax:
>From 68ef8e9ace9e13c3ac67fe6b85f9d4274567b2de Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 10 Aug 2026 11:03:21 +0000
Subject: [PATCH 05/12] Support operand widening
---
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 1 +
.../SelectionDAG/LegalizeVectorTypes.cpp | 25 ++++++++++++++++++
.../sve-vector-broadcast-unsupported.ll | 9 -------
.../CodeGen/AArch64/sve-vector-broadcast.ll | 26 +++++++++++++++++++
4 files changed, 52 insertions(+), 9 deletions(-)
delete mode 100644 llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index 422cf2dd70d5f..b2235bc5d972d 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -1110,6 +1110,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
bool WidenVectorOperand(SDNode *N, unsigned OpNo);
SDValue WidenVecOp_BITCAST(SDNode *N);
SDValue WidenVecOp_CONCAT_VECTORS(SDNode *N);
+ SDValue WidenVecOp_VECTOR_BROADCAST(SDNode *N);
SDValue WidenVecOp_EXTEND(SDNode *N);
SDValue WidenVecOp_CMP(SDNode *N);
SDValue WidenVecOp_EXTRACT_VECTOR_ELT(SDNode *N);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 02ab49693bfba..ec9b8d1113e82 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -7706,6 +7706,9 @@ bool DAGTypeLegalizer::WidenVectorOperand(SDNode *N, unsigned OpNo) {
Res = WidenVecOp_FAKE_USE(N);
break;
case ISD::CONCAT_VECTORS: Res = WidenVecOp_CONCAT_VECTORS(N); break;
+ case ISD::VECTOR_BROADCAST:
+ Res = WidenVecOp_VECTOR_BROADCAST(N);
+ break;
case ISD::INSERT_SUBVECTOR: Res = WidenVecOp_INSERT_SUBVECTOR(N); break;
case ISD::EXTRACT_SUBVECTOR: Res = WidenVecOp_EXTRACT_SUBVECTOR(N); break;
case ISD::EXTRACT_VECTOR_ELT: Res = WidenVecOp_EXTRACT_VECTOR_ELT(N); break;
@@ -8143,6 +8146,28 @@ SDValue DAGTypeLegalizer::WidenVecOp_CONCAT_VECTORS(SDNode *N) {
return DAG.getBuildVector(VT, dl, Ops);
}
+SDValue DAGTypeLegalizer::WidenVecOp_VECTOR_BROADCAST(SDNode *N) {
+ SDLoc DL(N);
+ EVT VT = N->getValueType(0);
+ SDValue Src = N->getOperand(0);
+ EVT SrcVT = Src.getValueType();
+ EVT WidenVT = TLI.getTypeToTransformTo(*DAG.getContext(), SrcVT);
+ assert(WidenVT.getVectorElementCount().isKnownMultipleOf(
+ SrcVT.getVectorElementCount()) &&
+ "Cannot widen VECTOR_BROADCAST operand to an ElementCount that's not "
+ "a multiple of the input ElementCount.");
+ unsigned NumConcat =
+ WidenVT.getVectorMinNumElements() / SrcVT.getVectorMinNumElements();
+
+ // Repeat the original source because the extra lanes of its widened value
+ // are unspecified.
+ SmallVector<SDValue, 8> Ops(NumConcat, Src);
+ SDValue WidenedSrc = DAG.getNode(ISD::CONCAT_VECTORS, DL, WidenVT, Ops);
+ if (VT == WidenVT)
+ return WidenedSrc;
+ return DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, WidenedSrc);
+}
+
SDValue DAGTypeLegalizer::WidenVecOp_INSERT_SUBVECTOR(SDNode *N) {
EVT VT = N->getValueType(0);
SDValue SubVec = N->getOperand(1);
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
deleted file mode 100644
index 49152c702505a..0000000000000
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast-unsupported.ll
+++ /dev/null
@@ -1,9 +0,0 @@
-; RUN: not --crash llc -mtriple=aarch64-linux-gnu -mattr=+sve < %s 2>&1 | FileCheck %s
-
-; CHECK: LLVM ERROR: Do not know how to widen this operator's operand!
-
-; TODO: Support broadcasts from 32-bit vec
-define <vscale x 8 x half> @broadcast_single_f16(<2 x half> %a) {
- %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<2 x half> %a)
- ret <vscale x 8 x half> %out
-}
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
index 7f1a37529f912..7ba31a9c77464 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -138,6 +138,8 @@ define <vscale x 2 x i64> @broadcast_quad_i64(<2 x i64> %a) {
ret <vscale x 2 x i64> %out
}
+; FP / BFP types
+
define <vscale x 8 x half> @broadcast_quad_f16(<8 x half> %a) {
; CHECK-LABEL: broadcast_quad_f16:
; CHECK: // %bb.0:
@@ -168,6 +170,30 @@ define <vscale x 4 x half> @broadcast_double_f16_to_double_sve(<4 x half> %a) {
ret <vscale x 4 x half> %out
}
+define <vscale x 8 x half> @broadcast_2f16_to_nxv8f16(<4 x half> %a) {
+; CHECK-LABEL: broadcast_2f16_to_nxv8f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2s, v0.s[0]
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
+ %out = call <vscale x 8 x half> @llvm.vector.broadcast.nxv8f16(<2 x half> %a.legal)
+ ret <vscale x 8 x half> %out
+}
+
+define <vscale x 4 x half> @broadcast_2f16_to_nxv4f16(<4 x half> %a) {
+; CHECK-LABEL: broadcast_2f16_to_nxv4f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2s, v0.s[0]
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
+ %out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<2 x half> %a.legal)
+ ret <vscale x 4 x half> %out
+}
+
define <vscale x 8 x bfloat> @broadcast_quad_bf16(<8 x bfloat> %a) #0 {
; CHECK-LABEL: broadcast_quad_bf16:
; CHECK: // %bb.0:
>From 9a0ca57d85a95be1554195821ff94a383af85a53 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 10 Aug 2026 12:03:39 +0000
Subject: [PATCH 06/12] Fix AArch64 ISel for unpacked types
Using DUP instructions only really works for packed types, as it will
not introduce the spacing between elements that is required for unpacked
types like nxv2f16.
For those, we'll first broadcast to their corresponding packed type, and
then extract the lo lanes into an unpacked type, effectively introducing
the required spacing.
---
.../Target/AArch64/AArch64ISelLowering.cpp | 22 ++++++++
llvm/lib/Target/AArch64/AArch64ISelLowering.h | 1 +
llvm/lib/Target/AArch64/SVEInstrFormats.td | 42 ++++-----------
.../CodeGen/AArch64/sve-vector-broadcast.ll | 51 +++++++++++++++++++
4 files changed, 85 insertions(+), 31 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 1131666e9aeb9..8c75dafe7bb1b 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -1922,6 +1922,12 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::VECTOR_SPLICE_RIGHT, VT, Custom);
}
+ // Broadcasts to unpacked SVE type require explicit unpacking to add spacing
+ // between elements.
+ for (auto VT : {MVT::nxv2f16, MVT::nxv4f16, MVT::nxv2f32, MVT::nxv2bf16,
+ MVT::nxv4bf16})
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Custom);
+
if (Subtarget->hasSVEB16B16() &&
Subtarget->isNonStreamingSVEorSME2Available()) {
// Note: Use SVE for bfloat16 operations when +sve-b16b16 is available.
@@ -8653,6 +8659,8 @@ SDValue AArch64TargetLowering::LowerOperation(SDValue Op,
return LowerEXTEND_VECTOR_INREG(Op, DAG);
case ISD::ZERO_EXTEND_VECTOR_INREG:
return LowerZERO_EXTEND_VECTOR_INREG(Op, DAG);
+ case ISD::VECTOR_BROADCAST:
+ return LowerVECTOR_BROADCAST(Op, DAG);
case ISD::VECTOR_SHUFFLE:
return LowerVECTOR_SHUFFLE(Op, DAG);
case ISD::SPLAT_VECTOR:
@@ -17586,6 +17594,20 @@ SDValue AArch64TargetLowering::LowerEXTRACT_SUBVECTOR(SDValue Op,
return SDValue();
}
+SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
+ SelectionDAG &DAG) const {
+ SDLoc DL(Op);
+ EVT VT = Op.getValueType();
+ assert(isUnpackedType(VT, DAG) && "Expected an unpacked vector type!");
+
+ // Broadcast into a packed container before extracting the low lanes, which
+ // places the result elements at the spacing required by the unpacked type.
+ EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
+ SDValue Broadcast =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, Op.getOperand(0));
+ return DAG.getExtractSubvector(DL, VT, Broadcast, 0);
+}
+
SDValue AArch64TargetLowering::LowerINSERT_SUBVECTOR(SDValue Op,
SelectionDAG &DAG) const {
assert(Op.getValueType().isScalableVector() &&
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.h b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
index 66a6261d2a991..afe278979c13f 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.h
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.h
@@ -758,6 +758,7 @@ class AArch64TargetLowering : public TargetLowering {
SDValue LowerBUILD_VECTOR(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerEXTEND_VECTOR_INREG(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerZERO_EXTEND_VECTOR_INREG(SDValue Op, SelectionDAG &DAG) const;
+ SDValue LowerVECTOR_BROADCAST(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerVECTOR_SHUFFLE(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerSPLAT_VECTOR(SDValue Op, SelectionDAG &DAG) const;
SDValue LowerDUPQLane(SDValue Op, SelectionDAG &DAG) const;
diff --git a/llvm/lib/Target/AArch64/SVEInstrFormats.td b/llvm/lib/Target/AArch64/SVEInstrFormats.td
index f818f33154b61..b83ebdc491c51 100644
--- a/llvm/lib/Target/AArch64/SVEInstrFormats.td
+++ b/llvm/lib/Target/AArch64/SVEInstrFormats.td
@@ -1553,21 +1553,12 @@ multiclass sve_int_perm_dup_i<string asm> {
// Duplicate an extracted vector element across a vector.
- foreach VT = [nxv16i8] in {
- def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.Packed ZPR:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) ZPR:$vec, sve_elm_idx_extdup_b:$index)>;
- def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.ZSub V128:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_b:$index)>;
- def : Pat<(VT (splat_vector (i32 (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_b:$index)))),
- (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_b:$index)>;
-
- // Broadcast a whole 128-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
- (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
- // Broadcast a whole 64-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
- (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
- }
+ def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (nxv16i8 ZPR:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) ZPR:$vec, sve_elm_idx_extdup_b:$index)>;
+ def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (v16i8 V128:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_b:$index)>;
+ def : Pat<(nxv16i8 (splat_vector (i32 (vector_extract (v8i8 V64:$vec), sve_elm_idx_extdup_b:$index)))),
+ (!cast<Instruction>(NAME # _B) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_b:$index)>;
foreach VT = [nxv8i16, nxv2f16, nxv4f16, nxv8f16, nxv2bf16, nxv4bf16, nxv8bf16] in {
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.Packed ZPR:$vec), sve_elm_idx_extdup_h:$index)))),
@@ -1576,13 +1567,6 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _H) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_h:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_h:$index)))),
(!cast<Instruction>(NAME # _H) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_h:$index)>;
-
- // Broadcast a whole 128-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
- (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
- // Broadcast a whole 64-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
- (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
foreach VT = [nxv4i32, nxv2f32, nxv4f32 ] in {
@@ -1592,13 +1576,6 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _S) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_s:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_s:$index)))),
(!cast<Instruction>(NAME # _S) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_s:$index)>;
-
- // Broadcast a whole 128-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
- (!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
- // Broadcast a whole 64-bit vector
- def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
- (!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
foreach VT = [nxv2i64, nxv2f64] in {
@@ -1608,11 +1585,14 @@ multiclass sve_int_perm_dup_i<string asm> {
(!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, zsub), sve_elm_idx_extdup_d:$index)>;
def : Pat<(VT (splat_vector (SVEType<VT>.EltAsScalar (vector_extract (SVEType<VT>.DSub V64:$vec), sve_elm_idx_extdup_d:$index)))),
(!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), sve_elm_idx_extdup_d:$index)>;
+ }
- // Broadcast a whole 128-bit vector
+ // Broadcast whole fixed-length vectors to packed SVE types.
+ // See LowerVECTOR_BROADCAST for handling of legal unpacked types.
+ foreach VT = [nxv16i8, nxv8i16, nxv8f16, nxv8bf16,
+ nxv4i32, nxv4f32, nxv2i64, nxv2f64] in {
def : Pat<(VT (vector_broadcast (SVEType<VT>.ZSub V128:$vec))),
(!cast<Instruction>(NAME # _Q) (SUBREG_TO_REG $vec, zsub), (i64 0))>;
- // Broadcast a whole 64-bit vector
def : Pat<(VT (vector_broadcast (SVEType<VT>.DSub V64:$vec))),
(!cast<Instruction>(NAME # _D) (SUBREG_TO_REG $vec, dsub), (i64 0))>;
}
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
index 7ba31a9c77464..34fbc4638fe4f 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -165,6 +165,7 @@ define <vscale x 4 x half> @broadcast_double_f16_to_double_sve(<4 x half> %a) {
; CHECK: // %bb.0:
; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: ret
%out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<4 x half> %a)
ret <vscale x 4 x half> %out
@@ -188,12 +189,40 @@ define <vscale x 4 x half> @broadcast_2f16_to_nxv4f16(<4 x half> %a) {
; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
; CHECK-NEXT: dup v0.2s, v0.s[0]
; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uunpklo z0.s, z0.h
; CHECK-NEXT: ret
%a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
%out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<2 x half> %a.legal)
ret <vscale x 4 x half> %out
}
+define <vscale x 8 x half> @broadcast_2f16_to_nxv8f16_lo(<4 x half> %a) {
+; CHECK-LABEL: broadcast_2f16_to_nxv8f16_lo:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2s, v0.s[0]
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: ret
+ %a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
+ %out = call <vscale x 4 x half> @llvm.vector.broadcast.nxv4f16(<2 x half> %a.legal)
+ %out.in.lo = call <vscale x 8 x half> @llvm.vector.insert.nxv8f16.nxv4f16(<vscale x 8 x half> poison, <vscale x 4 x half> %out, i64 0)
+ ret <vscale x 8 x half> %out.in.lo
+}
+
+define <vscale x 2 x half> @broadcast_2f16_to_nxv2f16(<4 x half> %a) {
+; CHECK-LABEL: broadcast_2f16_to_nxv2f16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0
+; CHECK-NEXT: dup v0.2s, v0.s[0]
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uunpklo z0.s, z0.h
+; CHECK-NEXT: uunpklo z0.d, z0.s
+; CHECK-NEXT: ret
+ %a.legal = call <2 x half> @llvm.vector.extract.v2f16.v4f16(<4 x half> %a, i64 0)
+ %out = call <vscale x 2 x half> @llvm.vector.broadcast.nxv2f16(<2 x half> %a.legal)
+ ret <vscale x 2 x half> %out
+}
+
define <vscale x 8 x bfloat> @broadcast_quad_bf16(<8 x bfloat> %a) #0 {
; CHECK-LABEL: broadcast_quad_bf16:
; CHECK: // %bb.0:
@@ -214,6 +243,28 @@ define <vscale x 4 x float> @broadcast_quad_f32(<4 x float> %a) {
ret <vscale x 4 x float> %out
}
+define <vscale x 2 x float> @broadcast_double_f32_to_nxv2f32(<2 x float> %a) {
+; CHECK-LABEL: broadcast_double_f32_to_nxv2f32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uunpklo z0.d, z0.s
+; CHECK-NEXT: ret
+ %out = call <vscale x 2 x float> @llvm.vector.broadcast.nxv2f32.v2f32(<2 x float> %a)
+ ret <vscale x 2 x float> %out
+}
+
+define <vscale x 4 x bfloat> @broadcast_double_bf16_to_nxv4bf16(<4 x bfloat> %a) #0 {
+; CHECK-LABEL: broadcast_double_bf16_to_nxv4bf16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $d0 killed $d0 def $z0
+; CHECK-NEXT: mov z0.d, d0
+; CHECK-NEXT: uunpklo z0.s, z0.h
+; CHECK-NEXT: ret
+ %out = call <vscale x 4 x bfloat> @llvm.vector.broadcast.nxv4bf16.v4bf16(<4 x bfloat> %a)
+ ret <vscale x 4 x bfloat> %out
+}
+
define <vscale x 2 x double> @broadcast_quad_f64(<2 x double> %a) {
; CHECK-LABEL: broadcast_quad_f64:
; CHECK: // %bb.0:
>From f890a300b56ec0623d8ceb82080086b21e964987 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 10 Aug 2026 12:33:35 +0000
Subject: [PATCH 07/12] Set Expand as default action for ISD::VECTOR_BROADCAST
---
llvm/lib/CodeGen/TargetLoweringBase.cpp | 4 ++++
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 8 +++++---
2 files changed, 9 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/CodeGen/TargetLoweringBase.cpp b/llvm/lib/CodeGen/TargetLoweringBase.cpp
index 051ba4670be40..1fdcf9be30441 100644
--- a/llvm/lib/CodeGen/TargetLoweringBase.cpp
+++ b/llvm/lib/CodeGen/TargetLoweringBase.cpp
@@ -996,6 +996,10 @@ void TargetLoweringBase::initActions() {
llvm::fill(RegClassForVT, nullptr);
llvm::fill(TargetDAGCombineArray, 0);
+ // Targets must explicitly opt into legal VECTOR_BROADCAST nodes.
+ for (MVT VT : MVT::vector_valuetypes())
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Expand);
+
// Let extending atomic loads be unsupported by default.
for (MVT ValVT : MVT::all_valuetypes())
for (MVT MemVT : MVT::all_valuetypes())
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 8c75dafe7bb1b..373f1f96bf192 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -728,9 +728,6 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::ROTR, VT, Expand);
}
- for (MVT VT : MVT::fixedlen_vector_valuetypes())
- setOperationAction(ISD::VECTOR_BROADCAST, VT, Expand);
-
// AArch64 doesn't have i32 MULH{S|U}.
setOperationAction(ISD::MULHU, MVT::i32, Expand);
setOperationAction(ISD::MULHS, MVT::i32, Expand);
@@ -1922,6 +1919,11 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::VECTOR_SPLICE_RIGHT, VT, Custom);
}
+ // Direct patterns exist for broadcasts to packed SVE types.
+ for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32, MVT::nxv2i64,
+ MVT::nxv8f16, MVT::nxv4f32, MVT::nxv2f64, MVT::nxv8bf16})
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Legal);
+
// Broadcasts to unpacked SVE type require explicit unpacking to add spacing
// between elements.
for (auto VT : {MVT::nxv2f16, MVT::nxv4f16, MVT::nxv2f32, MVT::nxv2bf16,
>From 2ad1783c62bba6efe5123d8a14f98b0e8eefb6f2 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Thu, 20 Aug 2026 13:57:57 +0000
Subject: [PATCH 08/12] Allow input min EC to be wider than output min EC with
vscale_range
This relaxes the requirements for vector.broadcast, allowing e.g.
<vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v4i64(<4 x i64> %a)
or
<vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %a)
when vscale is known > 1.
Langref and verifier are updated to reflect the new requirements.
Isel is now updated to properly support legal sources wider than NEON,
which can happen when the minimum vscale value is known > 1.
---
llvm/docs/LangRef.md | 8 ++-
.../lib/CodeGen/SelectionDAG/SelectionDAG.cpp | 12 ----
llvm/lib/IR/Verifier.cpp | 31 +++++++++
.../Target/AArch64/AArch64ISelLowering.cpp | 63 ++++++++++++++++++-
.../CodeGen/AArch64/sve-vector-broadcast.ll | 51 +++++++++++++++
.../vector-broadcast-intrinsic-invalid.ll | 37 +++++++++++
.../Verifier/vector-broadcast-intrinsic.ll | 16 +++++
7 files changed, 201 insertions(+), 17 deletions(-)
create mode 100644 llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
create mode 100644 llvm/test/Verifier/vector-broadcast-intrinsic.ll
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index fb01472f9386a..514ecc0a10840 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -20496,9 +20496,11 @@ to express this operation for fixed-width vectors is still to use a
##### Arguments:
-The argument and result must be vectors with the same element type. The element
-count of the result must be a known multiple of the element count of the
-argument.
+The argument and result must be vectors with the same element type. A scalable
+argument cannot be broadcast to a fixed-width result. For every possible value
+of `vscale`, the element count of the result must be a multiple of the element
+count of the argument. A `vscale_range` attribute may be used to establish this
+for a scalable result and fixed-width argument.
#### '`llvm.vector.reverse`' Intrinsic
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
index cee3ca37e8e4f..b9d0ed344663b 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
@@ -9127,18 +9127,6 @@ SDValue SelectionDAG::getNode(unsigned Opcode, const SDLoc &DL, EVT VT,
}
break;
}
- case ISD::VECTOR_BROADCAST: {
- [[maybe_unused]] EVT InputVT = N1.getValueType();
- assert(InputVT.isVector() && VT.isVector() &&
- "Expected the input and output of the VECTOR_BROADCAST node to be "
- "vectors!");
- assert(VT.getVectorElementCount().hasKnownScalarFactor(
- InputVT.getVectorElementCount()) &&
- "Expected the element count of the output of the VECTOR_BROADCAST "
- "node to be a positive integer multiple of the element count of the "
- "source operand!");
- break;
- }
case ISD::BITCAST:
// Fold bit_convert nodes from a type to themselves.
if (N1.getValueType() == VT)
diff --git a/llvm/lib/IR/Verifier.cpp b/llvm/lib/IR/Verifier.cpp
index 09429024e3ae8..dc4b4584915ed 100644
--- a/llvm/lib/IR/Verifier.cpp
+++ b/llvm/lib/IR/Verifier.cpp
@@ -6822,6 +6822,37 @@ void Verifier::visitIntrinsicCall(Intrinsic::ID ID, CallBase &Call) {
&Call);
break;
}
+ case Intrinsic::vector_broadcast: {
+ auto *ResultTy = cast<VectorType>(Call.getType());
+ auto *ArgTy = cast<VectorType>(Call.getArgOperand(0)->getType());
+ ElementCount ResultEC = ResultTy->getElementCount();
+ ElementCount InputEC = ArgTy->getElementCount();
+
+ Check(ResultTy->getElementType() == ArgTy->getElementType(),
+ "vector_broadcast argument and result must have the same element "
+ "type.",
+ &Call);
+
+ if (InputEC.isScalable() && ResultEC.isFixed()) {
+ CheckFailed("vector_broadcast cannot broadcast a scalable vector to a "
+ "fixed-width vector.",
+ &Call);
+ break;
+ }
+
+ uint64_t MinResultElements = ResultEC.getKnownMinValue();
+ if (ResultEC.isScalable() && InputEC.isFixed()) {
+ Attribute Attr =
+ Call.getFunction()->getFnAttribute(Attribute::VScaleRange);
+ if (Attr.isValid())
+ MinResultElements *= Attr.getVScaleRangeMin();
+ }
+ Check(MinResultElements % InputEC.getKnownMinValue() == 0,
+ "vector_broadcast result element count must be a multiple of the "
+ "argument element count for all possible values of vscale.",
+ &Call);
+ break;
+ }
case Intrinsic::vector_insert: {
Value *Vec = Call.getArgOperand(0);
Value *SubVec = Call.getArgOperand(1);
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 373f1f96bf192..7002aed6ed441 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -1919,10 +1919,17 @@ AArch64TargetLowering::AArch64TargetLowering(const TargetMachine &TM,
setOperationAction(ISD::VECTOR_SPLICE_RIGHT, VT, Custom);
}
- // Direct patterns exist for broadcasts to packed SVE types.
+ // Direct patterns exist for broadcasts to packed SVE types whose sources
+ // fit in a NEON register. Custom lowering handles wider fixed sources.
for (auto VT : {MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32, MVT::nxv2i64,
MVT::nxv8f16, MVT::nxv4f32, MVT::nxv2f64, MVT::nxv8bf16})
- setOperationAction(ISD::VECTOR_BROADCAST, VT, Legal);
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Custom);
+
+ // Custom legalise unpacked types to avoid promoting fixed-length sources
+ // beyond the width supported by NEON.
+ for (auto VT : {MVT::nxv2i8, MVT::nxv2i16, MVT::nxv2i32, MVT::nxv4i8,
+ MVT::nxv4i16, MVT::nxv8i8})
+ setOperationAction(ISD::VECTOR_BROADCAST, VT, Custom);
// Broadcasts to unpacked SVE type require explicit unpacking to add spacing
// between elements.
@@ -17600,6 +17607,38 @@ SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
SelectionDAG &DAG) const {
SDLoc DL(Op);
EVT VT = Op.getValueType();
+ assert(VT.isScalableVector() &&
+ "Custom lowering expected for scalable vectors only.");
+
+ if (isPackedVectorType(VT, DAG)) {
+ SDValue Src = Op.getOperand(0);
+ EVT SrcVT = Src.getValueType();
+ if (ElementCount::isKnownGE(VT.getVectorElementCount(),
+ SrcVT.getVectorElementCount()))
+ return Op;
+
+ // We are broadcasting to a packed scalable vector for a wider-than-NEON
+ // vector. Deinterleave smaller source vectors until we get to a quad.
+ // This is similar to SplitVecRes_VECTOR_BROADCAST, but we are past type
+ // legalisation so it cannot be reused.
+ // TODO: tbl might be cheaper? Or or repeated dup z.q[idx] followed by zip1
+ // z.q, but that requires +F64MM+NS
+ assert(SrcVT.isFixedLengthVector() &&
+ SrcVT.getFixedSizeInBits() > AArch64::SVEBitsPerBlock &&
+ "Expected a fixed source for a vscale-dependent broadcast");
+ auto [SrcLo, SrcHi] = DAG.SplitVector(Src, DL);
+ EVT SplitVT = SrcLo.getValueType();
+ SDValue Deinterleaved =
+ DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL,
+ DAG.getVTList(SplitVT, SplitVT), SrcLo, SrcHi);
+ SDValue Even = DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved);
+ SDValue Odd =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
+ SDValue Interleaved = DAG.getNode(ISD::VECTOR_INTERLEAVE, DL,
+ DAG.getVTList(VT, VT), Even, Odd);
+ return Interleaved.getValue(0);
+ }
+
assert(isUnpackedType(VT, DAG) && "Expected an unpacked vector type!");
// Broadcast into a packed container before extracting the low lanes, which
@@ -32206,6 +32245,26 @@ void AArch64TargetLowering::ReplaceNodeResults(
case ISD::BITCAST:
ReplaceBITCASTResults(N, Results, DAG);
return;
+ case ISD::VECTOR_BROADCAST: {
+ EVT VT = N->getValueType(0);
+ EVT PromotedVT = getTypeToTransformTo(*DAG.getContext(), VT);
+ EVT PromotedInVT = N->getOperand(0).getValueType().changeVectorElementType(
+ *DAG.getContext(), PromotedVT.getVectorElementType());
+ if (!PromotedInVT.isFixedLengthVector() ||
+ PromotedInVT.getFixedSizeInBits() <= AArch64::SVEBitsPerBlock)
+ return;
+
+ // Avoid promoting the input vector if that creates a vector wider than
+ // 128bits. Instead, keep the element type and broadcast to its packed type.
+ EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
+ SDLoc DL(N);
+ SDValue Broadcast =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, N->getOperand(0));
+ SDValue Promoted =
+ DAG.getNode(ISD::ANY_EXTEND_VECTOR_INREG, DL, PromotedVT, Broadcast);
+ Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, Promoted));
+ return;
+ }
case ISD::VECREDUCE_ADD:
case ISD::VECREDUCE_SMAX:
case ISD::VECREDUCE_SMIN:
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
index 34fbc4638fe4f..87d0783158bba 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -138,6 +138,57 @@ define <vscale x 2 x i64> @broadcast_quad_i64(<2 x i64> %a) {
ret <vscale x 2 x i64> %out
}
+; vscale_range tests
+
+define <vscale x 2 x i64> @broadcast_v4i32_to_nxv2i32(<4 x i32> %a) vscale_range(2,8) {
+; CHECK-LABEL: broadcast_v4i32_to_nxv2i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: uunpklo z0.d, z0.s
+; CHECK-NEXT: ret
+ %out = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %a)
+ %out.legal = zext <vscale x 2 x i32> %out to <vscale x 2 x i64>
+ ret <vscale x 2 x i64> %out.legal
+}
+
+; wider-than-NEON fixed-length source
+define <vscale x 2 x i64> @broadcast_v4i64_to_nxv2i64(<vscale x 2 x i64> %a.legal) vscale_range(2,8) {
+; CHECK-LABEL: broadcast_v4i64_to_nxv2i64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: movprfx z1, z0
+; CHECK-NEXT: ext z1.b, z1.b, z0.b, #16
+; CHECK-NEXT: uzp2 v2.2d, v0.2d, v1.2d
+; CHECK-NEXT: uzp1 v0.2d, v0.2d, v1.2d
+; CHECK-NEXT: mov z1.q, q2
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: zip1 z0.d, z0.d, z1.d
+; CHECK-NEXT: ret
+ %a = call <4 x i64> @llvm.vector.extract.v4i64.nxv2i64(<vscale x 2 x i64> %a.legal, i64 0)
+ %r = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v4i64(<4 x i64> %a)
+ ret <vscale x 2 x i64> %r
+}
+
+; wider-than-NEON fixed-length source and wide destination
+define <vscale x 4 x i32> @broadcast_v4i64_to_nxv4i64(<vscale x 2 x i64> %a.legal) vscale_range(2,8) {
+; CHECK-LABEL: broadcast_v4i64_to_nxv4i64:
+; CHECK: // %bb.0:
+; CHECK-NEXT: movprfx z1, z0
+; CHECK-NEXT: ext z1.b, z1.b, z0.b, #16
+; CHECK-NEXT: uzp2 v2.2d, v0.2d, v1.2d
+; CHECK-NEXT: uzp1 v0.2d, v0.2d, v1.2d
+; CHECK-NEXT: mov z1.q, q2
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: zip2 z2.d, z0.d, z1.d
+; CHECK-NEXT: zip1 z0.d, z0.d, z1.d
+; CHECK-NEXT: uzp1 z0.s, z0.s, z2.s
+; CHECK-NEXT: ret
+ %a = call <4 x i64> @llvm.vector.extract.v4i64.nxv2i64(<vscale x 2 x i64> %a.legal, i64 0)
+ %r = call <vscale x 4 x i64> @llvm.vector.broadcast.nxv4i64.v4i64(<4 x i64> %a)
+ %r.legal = trunc <vscale x 4 x i64> %r to <vscale x 4 x i32>
+ ret <vscale x 4 x i32> %r.legal
+}
+
; FP / BFP types
define <vscale x 8 x half> @broadcast_quad_f16(<8 x half> %a) {
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll b/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
new file mode 100644
index 0000000000000..ae7be4203316e
--- /dev/null
+++ b/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
@@ -0,0 +1,37 @@
+; RUN: not opt -passes=verify -disable-output < %s 2>&1 | FileCheck %s
+
+; CHECK: vector_broadcast argument and result must have the same element type.
+define <8 x i32> @mismatched_element_types(<2 x i64> %vec) {
+ %result = call <8 x i32> @llvm.vector.broadcast.v8i32.v2i64(<2 x i64> %vec)
+ ret <8 x i32> %result
+}
+
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+define <8 x i32> @non_multiple_fixed(<3 x i32> %vec) {
+ %result = call <8 x i32> @llvm.vector.broadcast.v8i32.v3i32(<3 x i32> %vec)
+ ret <8 x i32> %result
+}
+
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+define <vscale x 8 x i32> @non_multiple_scalable(<vscale x 3 x i32> %vec) {
+ %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.nxv3i32(<vscale x 3 x i32> %vec)
+ ret <vscale x 8 x i32> %result
+}
+
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+define <vscale x 2 x i32> @fixed_to_scalable_without_vscale_range(<4 x i32> %vec) {
+ %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
+ ret <vscale x 2 x i32> %result
+}
+
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+define <vscale x 2 x i32> @fixed_to_scalable_without_sufficient_vscale_range(<8 x i32> %vec) vscale_range(2,8) {
+ %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v8i32(<8 x i32> %vec)
+ ret <vscale x 2 x i32> %result
+}
+
+; CHECK: vector_broadcast cannot broadcast a scalable vector to a fixed-width vector.
+define <8 x i32> @scalable_to_fixed(<vscale x 2 x i32> %vec) {
+ %result = call <8 x i32> @llvm.vector.broadcast.v8i32.nxv2i32(<vscale x 2 x i32> %vec)
+ ret <8 x i32> %result
+}
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic.ll b/llvm/test/Verifier/vector-broadcast-intrinsic.ll
new file mode 100644
index 0000000000000..b9813e67d009d
--- /dev/null
+++ b/llvm/test/Verifier/vector-broadcast-intrinsic.ll
@@ -0,0 +1,16 @@
+; RUN: opt -passes=verify -disable-output < %s
+
+define <8 x i32> @fixed_to_fixed(<2 x i32> %vec) {
+ %result = call <8 x i32> @llvm.vector.broadcast.v8i32.v2i32(<2 x i32> %vec)
+ ret <8 x i32> %result
+}
+
+define <vscale x 8 x i32> @scalable_to_scalable(<vscale x 2 x i32> %vec) {
+ %result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.nxv2i32(<vscale x 2 x i32> %vec)
+ ret <vscale x 8 x i32> %result
+}
+
+define <vscale x 2 x i32> @fixed_to_scalable_with_vscale_range(<4 x i32> %vec) vscale_range(2, 8) {
+ %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
+ ret <vscale x 2 x i32> %result
+}
>From ada50e6b5485f85f68e3393bed2199c51e6b0e6d Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Fri, 21 Aug 2026 14:41:25 +0000
Subject: [PATCH 09/12] Do not require vscale_range
If the runtime EC of the output is smaller than the input, then the
result in poison.
---
llvm/docs/LangRef.md | 7 +--
llvm/include/llvm/CodeGen/ISDOpcodes.h | 3 +-
llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h | 1 +
.../SelectionDAG/LegalizeVectorTypes.cpp | 42 +++++++++-----
llvm/lib/IR/Verifier.cpp | 19 +++----
.../CodeGen/AArch64/sve-vector-broadcast.ll | 56 +++++++++++++++++++
.../vector-broadcast-intrinsic-invalid.ll | 16 +-----
.../Verifier/vector-broadcast-intrinsic.ll | 10 ++++
8 files changed, 110 insertions(+), 44 deletions(-)
diff --git a/llvm/docs/LangRef.md b/llvm/docs/LangRef.md
index 514ecc0a10840..ceb3d56f6efe2 100644
--- a/llvm/docs/LangRef.md
+++ b/llvm/docs/LangRef.md
@@ -20497,10 +20497,9 @@ to express this operation for fixed-width vectors is still to use a
##### Arguments:
The argument and result must be vectors with the same element type. A scalable
-argument cannot be broadcast to a fixed-width result. For every possible value
-of `vscale`, the element count of the result must be a multiple of the element
-count of the argument. A `vscale_range` attribute may be used to establish this
-for a scalable result and fixed-width argument.
+argument cannot be broadcast to a fixed-width result. At runtime, the element
+count of the result must be a multiple of the element count of the argument.
+Otherwise, the results is a {ref}`poison value <poisonvalues>`.
#### '`llvm.vector.reverse`' Intrinsic
diff --git a/llvm/include/llvm/CodeGen/ISDOpcodes.h b/llvm/include/llvm/CodeGen/ISDOpcodes.h
index 2e850da5d49b5..fac3034fafa1f 100644
--- a/llvm/include/llvm/CodeGen/ISDOpcodes.h
+++ b/llvm/include/llvm/CodeGen/ISDOpcodes.h
@@ -638,7 +638,8 @@ enum NodeType {
/// VECTOR_BROADCAST(SRC_SUBVEC)
/// Duplicate a vector in a larger vector. The element count of the result
- /// type is expected to be a multiple of the input vector's.
+ /// type is expected to be a multiple of the input vector's at runtime. If it
+ /// is not, the result is poison.
VECTOR_BROADCAST,
/// VECTOR_REVERSE(VECTOR) - Returns a vector, of the same type as VECTOR,
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index b2235bc5d972d..bb648c6ddac1b 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -990,6 +990,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
SDValue SplitVecOp_UnaryOp(SDNode *N);
SDValue SplitVecOp_TruncateHelper(SDNode *N);
SDValue SplitVecOp_VECTOR_COMPRESS(SDNode *N, unsigned OpNo);
+ SDValue SplitVecOp_VECTOR_BROADCAST(SDNode *N);
SDValue SplitVecOp_BITCAST(SDNode *N);
SDValue SplitVecOp_INSERT_SUBVECTOR(SDNode *N, unsigned OpNo);
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index ec9b8d1113e82..dc5a79fcd9de6 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -3538,6 +3538,19 @@ void DAGTypeLegalizer::SplitVecRes_FP_TO_XINT_SAT(SDNode *N, SDValue &Lo,
Hi = DAG.getNode(N->getOpcode(), dl, DstVTHi, SrcHi, N->getOperand(1));
}
+static SDValue buildSplitVectorBroadcast(SelectionDAG &DAG, SDLoc DL, EVT VT,
+ SDValue SrcLo, SDValue SrcHi) {
+ EVT SrcVT = SrcLo.getValueType();
+ SDValue Deinterleaved = DAG.getNode(
+ ISD::VECTOR_DEINTERLEAVE, DL, DAG.getVTList(SrcVT, SrcVT), SrcLo, SrcHi);
+ SDValue Even =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(0));
+ SDValue Odd =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
+ return DAG.getNode(ISD::VECTOR_INTERLEAVE, DL, DAG.getVTList(VT, VT), Even,
+ Odd);
+}
+
void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
SDValue &Hi) {
EVT VT = N->getValueType(0);
@@ -3553,30 +3566,21 @@ void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
return;
}
- // Second case: Src is known to be wider than LoVT.
+ // Second case: the split result is known to be at least as wide as Src.
SDLoc DL(N);
if (LoVT.getVectorMinNumElements() >= SrcVT.getVectorMinNumElements()) {
Lo = Hi = DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Src);
return;
}
- // Final case: VT is scalable and has the same minimum EC.
+ // Final case: VT is scalable and Src is a wider fixed-length vector.
// Use smaller even/odd source vectors so their broadcasts can be
// reinterleaved in the original lane order for every value of vscale.
- assert(VT.getVectorMinNumElements() == SrcVT.getVectorMinNumElements() &&
- VT.isScalableVector());
+ assert(VT.isScalableVector() && SrcVT.isFixedLengthVector() &&
+ "Expected a fixed source wider than the split scalable result");
SDValue SrcLo, SrcHi;
std::tie(SrcLo, SrcHi) = DAG.SplitVector(Src, DL);
- EVT SrcSplitVT = SrcLo.getValueType();
- SDValue Deinterleaved =
- DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL,
- DAG.getVTList(SrcSplitVT, SrcSplitVT), SrcLo, SrcHi);
- SDValue Even =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(0));
- SDValue Odd =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, LoVT, Deinterleaved.getValue(1));
- SDValue Interleaved = DAG.getNode(ISD::VECTOR_INTERLEAVE, DL,
- DAG.getVTList(LoVT, HiVT), Even, Odd);
+ SDValue Interleaved = buildSplitVectorBroadcast(DAG, DL, LoVT, SrcLo, SrcHi);
Lo = Interleaved.getValue(0);
Hi = Interleaved.getValue(1);
}
@@ -3885,6 +3889,9 @@ bool DAGTypeLegalizer::SplitVectorOperand(SDNode *N, unsigned OpNo) {
case ISD::INSERT_SUBVECTOR: Res = SplitVecOp_INSERT_SUBVECTOR(N, OpNo); break;
case ISD::EXTRACT_VECTOR_ELT:Res = SplitVecOp_EXTRACT_VECTOR_ELT(N); break;
case ISD::CONCAT_VECTORS: Res = SplitVecOp_CONCAT_VECTORS(N); break;
+ case ISD::VECTOR_BROADCAST:
+ Res = SplitVecOp_VECTOR_BROADCAST(N);
+ break;
case ISD::VECTOR_FIND_LAST_ACTIVE:
Res = SplitVecOp_VECTOR_FIND_LAST_ACTIVE(N);
break;
@@ -4065,6 +4072,13 @@ bool DAGTypeLegalizer::SplitVectorOperand(SDNode *N, unsigned OpNo) {
return false;
}
+SDValue DAGTypeLegalizer::SplitVecOp_VECTOR_BROADCAST(SDNode *N) {
+ SDLoc DL(N);
+ SDValue SrcLo, SrcHi;
+ GetSplitVector(N->getOperand(0), SrcLo, SrcHi);
+ return buildSplitVectorBroadcast(DAG, DL, N->getValueType(0), SrcLo, SrcHi);
+}
+
SDValue DAGTypeLegalizer::SplitVecOp_VECTOR_FIND_LAST_ACTIVE(SDNode *N) {
SDLoc DL(N);
diff --git a/llvm/lib/IR/Verifier.cpp b/llvm/lib/IR/Verifier.cpp
index dc4b4584915ed..593e3b3a74633 100644
--- a/llvm/lib/IR/Verifier.cpp
+++ b/llvm/lib/IR/Verifier.cpp
@@ -6840,17 +6840,14 @@ void Verifier::visitIntrinsicCall(Intrinsic::ID ID, CallBase &Call) {
break;
}
- uint64_t MinResultElements = ResultEC.getKnownMinValue();
- if (ResultEC.isScalable() && InputEC.isFixed()) {
- Attribute Attr =
- Call.getFunction()->getFnAttribute(Attribute::VScaleRange);
- if (Attr.isValid())
- MinResultElements *= Attr.getVScaleRangeMin();
- }
- Check(MinResultElements % InputEC.getKnownMinValue() == 0,
- "vector_broadcast result element count must be a multiple of the "
- "argument element count for all possible values of vscale.",
- &Call);
+ // We can only compare element counts when the types are both scalable or
+ // non-scalable.
+ if (ResultEC.isScalable() == InputEC.isScalable()) {
+ Check(ResultEC.isKnownMultipleOf(InputEC),
+ "vector_broadcast result element count must be a multiple of the "
+ "argument element count.",
+ &Call);
+ }
break;
}
case Intrinsic::vector_insert: {
diff --git a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
index 87d0783158bba..f4273b90ce3a1 100644
--- a/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
+++ b/llvm/test/CodeGen/AArch64/sve-vector-broadcast.ll
@@ -189,6 +189,62 @@ define <vscale x 4 x i32> @broadcast_v4i64_to_nxv4i64(<vscale x 2 x i64> %a.lega
ret <vscale x 4 x i32> %r.legal
}
+define <vscale x 2 x i64> @broadcast_v4i32_to_nxv2i32_no_vscale_range(<4 x i32> %a) {
+; CHECK-LABEL: broadcast_v4i32_to_nxv2i32_no_vscale_range:
+; CHECK: // %bb.0:
+; CHECK-NEXT: // kill: def $q0 killed $q0 def $z0
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: uunpklo z0.d, z0.s
+; CHECK-NEXT: ret
+ %out = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %a)
+ %out.legal = zext <vscale x 2 x i32> %out to <vscale x 2 x i64>
+ ret <vscale x 2 x i64> %out.legal
+}
+
+define <vscale x 2 x i64> @broadcast_v4i64_to_nxv2i64_no_vscale_range(ptr %ptr) {
+; CHECK-LABEL: broadcast_v4i64_to_nxv2i64_no_vscale_range:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ldp q1, q0, [x0]
+; CHECK-NEXT: uzp2 v2.2d, v1.2d, v0.2d
+; CHECK-NEXT: uzp1 v0.2d, v1.2d, v0.2d
+; CHECK-NEXT: mov z1.q, q2
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: zip1 z0.d, z0.d, z1.d
+; CHECK-NEXT: ret
+ %a = load <4 x i64>, ptr %ptr, align 8
+ %out = call <vscale x 2 x i64> @llvm.vector.broadcast.nxv2i64.v4i64(<4 x i64> %a)
+ ret <vscale x 2 x i64> %out
+}
+
+define <vscale x 4 x i32> @broadcast_v8i64_to_nxv4i64_no_vscale_range(ptr %ptr) {
+; CHECK-LABEL: broadcast_v8i64_to_nxv4i64_no_vscale_range:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ldp q1, q0, [x0]
+; CHECK-NEXT: ldp q3, q2, [x0, #32]
+; CHECK-NEXT: uzp2 v5.2d, v1.2d, v0.2d
+; CHECK-NEXT: uzp1 v0.2d, v1.2d, v0.2d
+; CHECK-NEXT: uzp2 v4.2d, v3.2d, v2.2d
+; CHECK-NEXT: uzp1 v2.2d, v3.2d, v2.2d
+; CHECK-NEXT: uzp2 v1.2d, v5.2d, v4.2d
+; CHECK-NEXT: uzp1 v3.2d, v5.2d, v4.2d
+; CHECK-NEXT: uzp2 v4.2d, v0.2d, v2.2d
+; CHECK-NEXT: uzp1 v0.2d, v0.2d, v2.2d
+; CHECK-NEXT: mov z1.q, q1
+; CHECK-NEXT: mov z2.q, q3
+; CHECK-NEXT: mov z3.q, q4
+; CHECK-NEXT: mov z0.q, q0
+; CHECK-NEXT: zip1 z1.d, z2.d, z1.d
+; CHECK-NEXT: zip1 z0.d, z0.d, z3.d
+; CHECK-NEXT: zip2 z2.d, z0.d, z1.d
+; CHECK-NEXT: zip1 z0.d, z0.d, z1.d
+; CHECK-NEXT: uzp1 z0.s, z0.s, z2.s
+; CHECK-NEXT: ret
+ %a = load <8 x i64>, ptr %ptr, align 8
+ %out = call <vscale x 4 x i64> @llvm.vector.broadcast.nxv4i64.v8i64(<8 x i64> %a)
+ %out.legal = trunc <vscale x 4 x i64> %out to <vscale x 4 x i32>
+ ret <vscale x 4 x i32> %out.legal
+}
+
; FP / BFP types
define <vscale x 8 x half> @broadcast_quad_f16(<8 x half> %a) {
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll b/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
index ae7be4203316e..f62c0b53bebe0 100644
--- a/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
+++ b/llvm/test/Verifier/vector-broadcast-intrinsic-invalid.ll
@@ -6,30 +6,18 @@ define <8 x i32> @mismatched_element_types(<2 x i64> %vec) {
ret <8 x i32> %result
}
-; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count.
define <8 x i32> @non_multiple_fixed(<3 x i32> %vec) {
%result = call <8 x i32> @llvm.vector.broadcast.v8i32.v3i32(<3 x i32> %vec)
ret <8 x i32> %result
}
-; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
+; CHECK: vector_broadcast result element count must be a multiple of the argument element count.
define <vscale x 8 x i32> @non_multiple_scalable(<vscale x 3 x i32> %vec) {
%result = call <vscale x 8 x i32> @llvm.vector.broadcast.nxv8i32.nxv3i32(<vscale x 3 x i32> %vec)
ret <vscale x 8 x i32> %result
}
-; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
-define <vscale x 2 x i32> @fixed_to_scalable_without_vscale_range(<4 x i32> %vec) {
- %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
- ret <vscale x 2 x i32> %result
-}
-
-; CHECK: vector_broadcast result element count must be a multiple of the argument element count for all possible values of vscale.
-define <vscale x 2 x i32> @fixed_to_scalable_without_sufficient_vscale_range(<8 x i32> %vec) vscale_range(2,8) {
- %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v8i32(<8 x i32> %vec)
- ret <vscale x 2 x i32> %result
-}
-
; CHECK: vector_broadcast cannot broadcast a scalable vector to a fixed-width vector.
define <8 x i32> @scalable_to_fixed(<vscale x 2 x i32> %vec) {
%result = call <8 x i32> @llvm.vector.broadcast.v8i32.nxv2i32(<vscale x 2 x i32> %vec)
diff --git a/llvm/test/Verifier/vector-broadcast-intrinsic.ll b/llvm/test/Verifier/vector-broadcast-intrinsic.ll
index b9813e67d009d..1f7434ec116c6 100644
--- a/llvm/test/Verifier/vector-broadcast-intrinsic.ll
+++ b/llvm/test/Verifier/vector-broadcast-intrinsic.ll
@@ -14,3 +14,13 @@ define <vscale x 2 x i32> @fixed_to_scalable_with_vscale_range(<4 x i32> %vec) v
%result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
ret <vscale x 2 x i32> %result
}
+
+define <vscale x 2 x i32> @fixed_to_scalable_without_vscale_range(<4 x i32> %vec) {
+ %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v4i32(<4 x i32> %vec)
+ ret <vscale x 2 x i32> %result
+}
+
+define <vscale x 2 x i32> @fixed_to_scalable_without_sufficient_vscale_range(<8 x i32> %vec) vscale_range(2, 8) {
+ %result = call <vscale x 2 x i32> @llvm.vector.broadcast.nxv2i32.v8i32(<8 x i32> %vec)
+ ret <vscale x 2 x i32> %result
+}
>From 82c619312f3113a225a2502b2d8c26cedbb137f6 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 24 Aug 2026 10:02:19 +0000
Subject: [PATCH 10/12] Reduce indent level and comment about legal operation
---
.../Target/AArch64/AArch64ISelLowering.cpp | 77 ++++++++++---------
1 file changed, 41 insertions(+), 36 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 7002aed6ed441..ed0f53dea646c 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -17610,43 +17610,48 @@ SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
assert(VT.isScalableVector() &&
"Custom lowering expected for scalable vectors only.");
- if (isPackedVectorType(VT, DAG)) {
- SDValue Src = Op.getOperand(0);
- EVT SrcVT = Src.getValueType();
- if (ElementCount::isKnownGE(VT.getVectorElementCount(),
- SrcVT.getVectorElementCount()))
- return Op;
+ if (isUnpackedType(VT, DAG)) { // Broadcast into a packed container before
+ // extracting the low lanes, which
+ // places the result elements at the spacing required by the unpacked type.
+ EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
+ SDValue Broadcast =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, Op.getOperand(0));
+ return DAG.getExtractSubvector(DL, VT, Broadcast, 0);
+ }
- // We are broadcasting to a packed scalable vector for a wider-than-NEON
- // vector. Deinterleave smaller source vectors until we get to a quad.
- // This is similar to SplitVecRes_VECTOR_BROADCAST, but we are past type
- // legalisation so it cannot be reused.
- // TODO: tbl might be cheaper? Or or repeated dup z.q[idx] followed by zip1
- // z.q, but that requires +F64MM+NS
- assert(SrcVT.isFixedLengthVector() &&
- SrcVT.getFixedSizeInBits() > AArch64::SVEBitsPerBlock &&
- "Expected a fixed source for a vscale-dependent broadcast");
- auto [SrcLo, SrcHi] = DAG.SplitVector(Src, DL);
- EVT SplitVT = SrcLo.getValueType();
- SDValue Deinterleaved =
- DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL,
- DAG.getVTList(SplitVT, SplitVT), SrcLo, SrcHi);
- SDValue Even = DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved);
- SDValue Odd =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
- SDValue Interleaved = DAG.getNode(ISD::VECTOR_INTERLEAVE, DL,
- DAG.getVTList(VT, VT), Even, Odd);
- return Interleaved.getValue(0);
- }
-
- assert(isUnpackedType(VT, DAG) && "Expected an unpacked vector type!");
-
- // Broadcast into a packed container before extracting the low lanes, which
- // places the result elements at the spacing required by the unpacked type.
- EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
- SDValue Broadcast =
- DAG.getNode(ISD::VECTOR_BROADCAST, DL, PackedVT, Op.getOperand(0));
- return DAG.getExtractSubvector(DL, VT, Broadcast, 0);
+ assert(isPackedVectorType(VT, DAG) && "Expected a packed vector type!");
+ SDValue Src = Op.getOperand(0);
+ EVT SrcVT = Src.getValueType();
+
+ // We are broadcasting a NEON-sized vector to a packed (legal) SVE type.
+ // This is a legal operation and there are patterns for it.
+ if (ElementCount::isKnownGE(VT.getVectorElementCount(),
+ SrcVT.getVectorElementCount())) {
+ assert(SrcVT.getFixedSizeInBits() <= AArch64::SVEBitsPerBlock &&
+ "Expected NEON-sized source");
+ return Op;
+ }
+
+ // We are broadcasting to a packed scalable vector for a wider-than-NEON
+ // vector. Deinterleave smaller source vectors until we get to a quad.
+ // This is similar to SplitVecRes_VECTOR_BROADCAST, but we are past type
+ // legalisation so it cannot be reused.
+ // TODO: tbl might be cheaper? Or repeated dup z.q[idx] followed by zip1
+ // z.q, but that requires +F64MM+NS
+ assert(SrcVT.isFixedLengthVector() &&
+ SrcVT.getFixedSizeInBits() > AArch64::SVEBitsPerBlock &&
+ "Expected a fixed source for a vscale-dependent broadcast");
+ auto [SrcLo, SrcHi] = DAG.SplitVector(Src, DL);
+ EVT SplitVT = SrcLo.getValueType();
+ SDValue Deinterleaved =
+ DAG.getNode(ISD::VECTOR_DEINTERLEAVE, DL, DAG.getVTList(SplitVT, SplitVT),
+ SrcLo, SrcHi);
+ SDValue Even = DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved);
+ SDValue Odd =
+ DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
+ SDValue Interleaved =
+ DAG.getNode(ISD::VECTOR_INTERLEAVE, DL, DAG.getVTList(VT, VT), Even, Odd);
+ return Interleaved.getValue(0);
}
SDValue AArch64TargetLowering::LowerINSERT_SUBVECTOR(SDValue Op,
>From b1e8a77e1e25ab90818988c02101647b8c1ebe22 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 24 Aug 2026 10:31:31 +0000
Subject: [PATCH 11/12] Tidy up buildSplitVectorBroadcast
This is used in two cases:
- Split the source operand, but keeping the output VT as is
- Split the output VT as well as the source VT
---
.../SelectionDAG/LegalizeVectorTypes.cpp | 29 ++++++++++++-------
1 file changed, 19 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index dc5a79fcd9de6..5dea7d5cf80d6 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -3538,8 +3538,14 @@ void DAGTypeLegalizer::SplitVecRes_FP_TO_XINT_SAT(SDNode *N, SDValue &Lo,
Hi = DAG.getNode(N->getOpcode(), dl, DstVTHi, SrcHi, N->getOperand(1));
}
-static SDValue buildSplitVectorBroadcast(SelectionDAG &DAG, SDLoc DL, EVT VT,
- SDValue SrcLo, SDValue SrcHi) {
+/// For two Lo/Hi source halves, perform a broadcast for each to VT and
+/// re-interleave the elements so that the result is equivalent to:
+/// Dst: DoubleWidthVT = VECTOR_BROADCAST(VECTOR_CONCAT(SrcLo, SrcHi))
+/// DstLo,DstHi: VT,VT = split(Dst)
+static std::pair<SDValue, SDValue> buildSplitVectorBroadcast(SelectionDAG &DAG,
+ SDLoc DL, EVT VT,
+ SDValue SrcLo,
+ SDValue SrcHi) {
EVT SrcVT = SrcLo.getValueType();
SDValue Deinterleaved = DAG.getNode(
ISD::VECTOR_DEINTERLEAVE, DL, DAG.getVTList(SrcVT, SrcVT), SrcLo, SrcHi);
@@ -3547,8 +3553,9 @@ static SDValue buildSplitVectorBroadcast(SelectionDAG &DAG, SDLoc DL, EVT VT,
DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(0));
SDValue Odd =
DAG.getNode(ISD::VECTOR_BROADCAST, DL, VT, Deinterleaved.getValue(1));
- return DAG.getNode(ISD::VECTOR_INTERLEAVE, DL, DAG.getVTList(VT, VT), Even,
- Odd);
+ SDValue Interleaved =
+ DAG.getNode(ISD::VECTOR_INTERLEAVE, DL, DAG.getVTList(VT, VT), Even, Odd);
+ return std::make_pair(Interleaved.getValue(0), Interleaved.getValue(1));
}
void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
@@ -3578,11 +3585,8 @@ void DAGTypeLegalizer::SplitVecRes_VECTOR_BROADCAST(SDNode *N, SDValue &Lo,
// reinterleaved in the original lane order for every value of vscale.
assert(VT.isScalableVector() && SrcVT.isFixedLengthVector() &&
"Expected a fixed source wider than the split scalable result");
- SDValue SrcLo, SrcHi;
- std::tie(SrcLo, SrcHi) = DAG.SplitVector(Src, DL);
- SDValue Interleaved = buildSplitVectorBroadcast(DAG, DL, LoVT, SrcLo, SrcHi);
- Lo = Interleaved.getValue(0);
- Hi = Interleaved.getValue(1);
+ auto [SrcLo, SrcHi] = DAG.SplitVector(Src, DL);
+ std::tie(Lo, Hi) = buildSplitVectorBroadcast(DAG, DL, LoVT, SrcLo, SrcHi);
}
void DAGTypeLegalizer::SplitVecRes_VECTOR_REVERSE(SDNode *N, SDValue &Lo,
@@ -4076,7 +4080,12 @@ SDValue DAGTypeLegalizer::SplitVecOp_VECTOR_BROADCAST(SDNode *N) {
SDLoc DL(N);
SDValue SrcLo, SrcHi;
GetSplitVector(N->getOperand(0), SrcLo, SrcHi);
- return buildSplitVectorBroadcast(DAG, DL, N->getValueType(0), SrcLo, SrcHi);
+
+ // We are only splitting SrcVT, not the destiniation type, which remains VT.
+ // buildSplitVectorBroadcast produces (VT,VT) so only Lo is needed.
+ EVT VT = N->getValueType(0);
+ auto [Lo, _] = buildSplitVectorBroadcast(DAG, DL, VT, SrcLo, SrcHi);
+ return Lo;
}
SDValue DAGTypeLegalizer::SplitVecOp_VECTOR_FIND_LAST_ACTIVE(SDNode *N) {
>From d66354f7976ccdade0631cc91a75e9f407cb5443 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Ga=C3=ABtan=20Bossu?= <gaetan.bossu at arm.com>
Date: Mon, 24 Aug 2026 10:50:12 +0000
Subject: [PATCH 12/12] Fix comments placement
---
llvm/lib/Target/AArch64/AArch64ISelLowering.cpp | 8 ++++----
1 file changed, 4 insertions(+), 4 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index ed0f53dea646c..202568a3063d7 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -17610,8 +17610,8 @@ SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
assert(VT.isScalableVector() &&
"Custom lowering expected for scalable vectors only.");
- if (isUnpackedType(VT, DAG)) { // Broadcast into a packed container before
- // extracting the low lanes, which
+ if (isUnpackedType(VT, DAG)) {
+ // Broadcast into a packed container before extracting the low lanes, which
// places the result elements at the spacing required by the unpacked type.
EVT PackedVT = getPackedSVEVectorVT(VT.getVectorElementType());
SDValue Broadcast =
@@ -17623,10 +17623,10 @@ SDValue AArch64TargetLowering::LowerVECTOR_BROADCAST(SDValue Op,
SDValue Src = Op.getOperand(0);
EVT SrcVT = Src.getValueType();
- // We are broadcasting a NEON-sized vector to a packed (legal) SVE type.
- // This is a legal operation and there are patterns for it.
if (ElementCount::isKnownGE(VT.getVectorElementCount(),
SrcVT.getVectorElementCount())) {
+ // We are broadcasting a NEON-sized vector to a packed (legal) SVE type.
+ // This is a legal operation and there are patterns for it.
assert(SrcVT.getFixedSizeInBits() <= AArch64::SVEBitsPerBlock &&
"Expected NEON-sized source");
return Op;
More information about the llvm-commits
mailing list