[llvm] [AArch64] Combine scalar_to_vector(C) -> buildvector (PR #220948)
David Green via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 6 23:59:59 PDT 2026
https://github.com/davemgreen updated https://github.com/llvm/llvm-project/pull/220948
>From 444ba06031824cb06f5e7b6ded8f8dcf90a9c6db Mon Sep 17 00:00:00 2001
From: David Green <david.green at arm.com>
Date: Thu, 3 Sep 2026 15:41:47 +0100
Subject: [PATCH] [AArch64] Combine scalar_to_vector(C) -> buildvector
This helps keep a single canonical for of constant vector, helping a number of
the existing buildvector combines trigger.
---
.../Target/AArch64/AArch64ISelLowering.cpp | 20 ++++---
.../CodeGen/AArch64/arm64-build-vector.ll | 2 +-
llvm/test/CodeGen/AArch64/arm64-tbl.ll | 2 +-
llvm/test/CodeGen/AArch64/neon-anyof-splat.ll | 6 +--
llvm/test/CodeGen/AArch64/pdep.ll | 41 +++++++-------
llvm/test/CodeGen/AArch64/pext.ll | 53 +++++++++----------
6 files changed, 64 insertions(+), 60 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 22198cd122fc7..622f162690f2a 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -17260,9 +17260,9 @@ SDValue AArch64TargetLowering::LowerBUILD_VECTOR(SDValue Op,
}
// Convert BUILD_VECTOR where all elements but the lowest are undef into
- // SCALAR_TO_VECTOR, except for when we have a single-element constant vector
+ // SCALAR_TO_VECTOR, except for when we have a constant vector
// as SimplifyDemandedBits will just turn that back into BUILD_VECTOR.
- if (isOnlyLowElement && !(NumElts == 1 && isIntOrFPConstant(Value))) {
+ if (isOnlyLowElement && !isIntOrFPConstant(Value)) {
LLVM_DEBUG(dbgs() << "LowerBUILD_VECTOR: only low element used, creating 1 "
"SCALAR_TO_VECTOR node\n");
return DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, VT, Value);
@@ -31114,6 +31114,8 @@ static SDValue
performScalarToVectorCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI,
SelectionDAG &DAG) {
SDLoc DL(N);
+ EVT VT = N->getValueType(0);
+ SDValue N0 = N->getOperand(0);
// If a DUP(Op0) already exists, reuse it for the scalar_to_vector.
if (DCI.isAfterLegalizeDAG()) {
@@ -31122,6 +31124,14 @@ performScalarToVectorCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI,
return SDValue(LN, 0);
}
+ if (VT.isFixedLengthVector() && VT.getVectorNumElements() > 1 &&
+ isIntOrFPConstant(N0)) {
+ SDValue Undef = DAG.getPOISON(N0.getValueType());
+ SmallVector<SDValue> Ops(VT.getVectorNumElements(), Undef);
+ Ops[0] = N0;
+ return DAG.getBuildVector(VT, DL, Ops);
+ }
+
// Let's do below transform.
//
// t34: v4i32 = AArch64ISD::UADDLV t2
@@ -31135,15 +31145,13 @@ performScalarToVectorCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI,
if (DCI.isBeforeLegalizeOps())
return SDValue();
- EVT VT = N->getValueType(0);
if (VT != MVT::v1i64)
return SDValue();
- SDValue ZEXT = N->getOperand(0);
- if (ZEXT.getOpcode() != ISD::ZERO_EXTEND || ZEXT.getValueType() != MVT::i64)
+ if (N0.getOpcode() != ISD::ZERO_EXTEND || N0.getValueType() != MVT::i64)
return SDValue();
- SDValue EXTRACT_VEC_ELT = ZEXT.getOperand(0);
+ SDValue EXTRACT_VEC_ELT = N0.getOperand(0);
if (EXTRACT_VEC_ELT.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
EXTRACT_VEC_ELT.getValueType() != MVT::i32)
return SDValue();
diff --git a/llvm/test/CodeGen/AArch64/arm64-build-vector.ll b/llvm/test/CodeGen/AArch64/arm64-build-vector.ll
index f43aa27823494..f4865fcee5c28 100644
--- a/llvm/test/CodeGen/AArch64/arm64-build-vector.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-build-vector.ll
@@ -26,7 +26,7 @@ define <8 x i16> @build_all_zero(<8 x i16> %a) #1 {
; CHECK-SD-LABEL: build_all_zero:
; CHECK-SD: // %bb.0:
; CHECK-SD-NEXT: mov w8, #44672 // =0xae80
-; CHECK-SD-NEXT: fmov s1, w8
+; CHECK-SD-NEXT: dup v1.8h, w8
; CHECK-SD-NEXT: mul v0.8h, v0.8h, v1.8h
; CHECK-SD-NEXT: ret
;
diff --git a/llvm/test/CodeGen/AArch64/arm64-tbl.ll b/llvm/test/CodeGen/AArch64/arm64-tbl.ll
index 200fc65f01828..67bfdde8d9f9a 100644
--- a/llvm/test/CodeGen/AArch64/arm64-tbl.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-tbl.ll
@@ -442,9 +442,9 @@ define <16 x i8> @shuffled_tbl2_to_tbl4_nonconst_first_mask(<16 x i8> %a, <16 x
define <16 x i8> @shuffled_tbl2_to_tbl4_nonconst_first_mask2(<16 x i8> %a, <16 x i8> %b, <16 x i8> %c, <16 x i8> %d, i8 %v) {
; CHECK-SD-LABEL: shuffled_tbl2_to_tbl4_nonconst_first_mask2:
; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: movi.16b v4, #1
; CHECK-SD-NEXT: mov w8, #1 // =0x1
; CHECK-SD-NEXT: // kill: def $q3 killed $q3 killed $q0_q1_q2_q3 def $q0_q1_q2_q3
-; CHECK-SD-NEXT: fmov s4, w8
; CHECK-SD-NEXT: // kill: def $q2 killed $q2 killed $q0_q1_q2_q3 def $q0_q1_q2_q3
; CHECK-SD-NEXT: // kill: def $q1 killed $q1 killed $q0_q1_q2_q3 def $q0_q1_q2_q3
; CHECK-SD-NEXT: // kill: def $q0 killed $q0 killed $q0_q1_q2_q3 def $q0_q1_q2_q3
diff --git a/llvm/test/CodeGen/AArch64/neon-anyof-splat.ll b/llvm/test/CodeGen/AArch64/neon-anyof-splat.ll
index bf33b348173f9..b03b9c7a40f13 100644
--- a/llvm/test/CodeGen/AArch64/neon-anyof-splat.ll
+++ b/llvm/test/CodeGen/AArch64/neon-anyof-splat.ll
@@ -16,9 +16,8 @@ define <4 x i32> @any_of_select_vf4(<4 x i32> %mask, <4 x i32> %a, <4 x i32> %b)
; CHECK-SD-LABEL: any_of_select_vf4:
; CHECK-SD: // %bb.0:
; CHECK-SD-NEXT: cmlt v0.4s, v0.4s, #0
-; CHECK-SD-NEXT: movi d3, #0000000000000000
; CHECK-SD-NEXT: addp d0, v0.2d
-; CHECK-SD-NEXT: cmeq v0.2d, v0.2d, v3.2d
+; CHECK-SD-NEXT: cmeq v0.2d, v0.2d, #0
; CHECK-SD-NEXT: dup v0.2d, v0.d[0]
; CHECK-SD-NEXT: bsl v0.16b, v1.16b, v2.16b
; CHECK-SD-NEXT: ret
@@ -106,9 +105,8 @@ define <2 x i64> @any_of_select_vf2(<2 x i64> %mask, <2 x i64> %a, <2 x i64> %b)
; CHECK-LABEL: any_of_select_vf2:
; CHECK: // %bb.0:
; CHECK-NEXT: cmlt v0.2d, v0.2d, #0
-; CHECK-NEXT: movi d3, #0000000000000000
; CHECK-NEXT: addp d0, v0.2d
-; CHECK-NEXT: cmeq v0.2d, v0.2d, v3.2d
+; CHECK-NEXT: cmeq v0.2d, v0.2d, #0
; CHECK-NEXT: dup v0.2d, v0.d[0]
; CHECK-NEXT: bsl v0.16b, v1.16b, v2.16b
; CHECK-NEXT: ret
diff --git a/llvm/test/CodeGen/AArch64/pdep.ll b/llvm/test/CodeGen/AArch64/pdep.ll
index aad1f51b2107f..7b1ad7d525967 100644
--- a/llvm/test/CodeGen/AArch64/pdep.ll
+++ b/llvm/test/CodeGen/AArch64/pdep.ll
@@ -13,26 +13,25 @@ define i8 @pdep_i8(i8 %val, i8 %mask) nounwind {
;
; NOSVE2BITPERM-LABEL: pdep_i8:
; NOSVE2BITPERM: // %bb.0:
-; NOSVE2BITPERM-NEXT: mvn w9, w1
-; NOSVE2BITPERM-NEXT: mov w8, #-1 // =0xffffffff
-; NOSVE2BITPERM-NEXT: lsl w9, w9, #1
-; NOSVE2BITPERM-NEXT: fmov s0, w8
-; NOSVE2BITPERM-NEXT: fmov s1, w9
+; NOSVE2BITPERM-NEXT: mvn w8, w1
+; NOSVE2BITPERM-NEXT: movi v0.2d, #0xffffffffffffffff
+; NOSVE2BITPERM-NEXT: lsl w8, w8, #1
+; NOSVE2BITPERM-NEXT: fmov s1, w8
; NOSVE2BITPERM-NEXT: pmul v1.8b, v1.8b, v0.8b
-; NOSVE2BITPERM-NEXT: fmov w8, s1
-; NOSVE2BITPERM-NEXT: bic w9, w9, w8
-; NOSVE2BITPERM-NEXT: and w8, w8, w1
-; NOSVE2BITPERM-NEXT: fmov s1, w9
-; NOSVE2BITPERM-NEXT: eor w11, w1, w8
-; NOSVE2BITPERM-NEXT: and w12, w8, #0xfe
+; NOSVE2BITPERM-NEXT: fmov w9, s1
+; NOSVE2BITPERM-NEXT: bic w8, w8, w9
+; NOSVE2BITPERM-NEXT: and w9, w9, w1
+; NOSVE2BITPERM-NEXT: fmov s1, w8
+; NOSVE2BITPERM-NEXT: eor w11, w1, w9
+; NOSVE2BITPERM-NEXT: and w12, w9, #0xfe
; NOSVE2BITPERM-NEXT: orr w11, w11, w12, lsr #1
; NOSVE2BITPERM-NEXT: pmul v1.8b, v1.8b, v0.8b
; NOSVE2BITPERM-NEXT: fmov w10, s1
-; NOSVE2BITPERM-NEXT: bic w9, w9, w10
-; NOSVE2BITPERM-NEXT: fmov s1, w9
-; NOSVE2BITPERM-NEXT: and w9, w10, w11
-; NOSVE2BITPERM-NEXT: eor w10, w11, w9
-; NOSVE2BITPERM-NEXT: and w11, w9, #0xfc
+; NOSVE2BITPERM-NEXT: bic w8, w8, w10
+; NOSVE2BITPERM-NEXT: fmov s1, w8
+; NOSVE2BITPERM-NEXT: and w8, w10, w11
+; NOSVE2BITPERM-NEXT: eor w10, w11, w8
+; NOSVE2BITPERM-NEXT: and w11, w8, #0xfc
; NOSVE2BITPERM-NEXT: orr w10, w10, w11, lsr #2
; NOSVE2BITPERM-NEXT: pmul v0.8b, v1.8b, v0.8b
; NOSVE2BITPERM-NEXT: fmov w11, s0
@@ -40,11 +39,11 @@ define i8 @pdep_i8(i8 %val, i8 %mask) nounwind {
; NOSVE2BITPERM-NEXT: and w11, w10, w0, lsl #4
; NOSVE2BITPERM-NEXT: bic w10, w0, w10
; NOSVE2BITPERM-NEXT: orr w10, w10, w11
-; NOSVE2BITPERM-NEXT: and w11, w9, w10, lsl #2
-; NOSVE2BITPERM-NEXT: bic w9, w10, w9
-; NOSVE2BITPERM-NEXT: orr w9, w9, w11
-; NOSVE2BITPERM-NEXT: and w10, w8, w9, lsl #1
-; NOSVE2BITPERM-NEXT: bic w8, w9, w8
+; NOSVE2BITPERM-NEXT: and w11, w8, w10, lsl #2
+; NOSVE2BITPERM-NEXT: bic w8, w10, w8
+; NOSVE2BITPERM-NEXT: orr w8, w8, w11
+; NOSVE2BITPERM-NEXT: and w10, w9, w8, lsl #1
+; NOSVE2BITPERM-NEXT: bic w8, w8, w9
; NOSVE2BITPERM-NEXT: orr w8, w8, w10
; NOSVE2BITPERM-NEXT: and w0, w8, w1
; NOSVE2BITPERM-NEXT: ret
diff --git a/llvm/test/CodeGen/AArch64/pext.ll b/llvm/test/CodeGen/AArch64/pext.ll
index ecde58f357489..5a9d8913166d6 100644
--- a/llvm/test/CodeGen/AArch64/pext.ll
+++ b/llvm/test/CodeGen/AArch64/pext.ll
@@ -14,43 +14,42 @@ define i8 @pext_i8(i8 %val, i8 %mask) nounwind {
;
; NOSVE2BITPERM-LABEL: pext_i8:
; NOSVE2BITPERM: // %bb.0:
-; NOSVE2BITPERM-NEXT: mvn w9, w1
-; NOSVE2BITPERM-NEXT: mov w8, #-1 // =0xffffffff
+; NOSVE2BITPERM-NEXT: mvn w8, w1
+; NOSVE2BITPERM-NEXT: movi v0.2d, #0xffffffffffffffff
; NOSVE2BITPERM-NEXT: and w10, w0, w1
-; NOSVE2BITPERM-NEXT: lsl w9, w9, #1
-; NOSVE2BITPERM-NEXT: fmov s0, w8
-; NOSVE2BITPERM-NEXT: fmov s1, w9
+; NOSVE2BITPERM-NEXT: lsl w8, w8, #1
+; NOSVE2BITPERM-NEXT: fmov s1, w8
; NOSVE2BITPERM-NEXT: pmul v1.8b, v1.8b, v0.8b
-; NOSVE2BITPERM-NEXT: fmov w8, s1
-; NOSVE2BITPERM-NEXT: bic w9, w9, w8
-; NOSVE2BITPERM-NEXT: and w8, w8, w1
-; NOSVE2BITPERM-NEXT: fmov s1, w9
-; NOSVE2BITPERM-NEXT: and w11, w10, w8
-; NOSVE2BITPERM-NEXT: eor w13, w1, w8
-; NOSVE2BITPERM-NEXT: and w8, w8, #0xfe
+; NOSVE2BITPERM-NEXT: fmov w9, s1
+; NOSVE2BITPERM-NEXT: bic w8, w8, w9
+; NOSVE2BITPERM-NEXT: and w9, w9, w1
+; NOSVE2BITPERM-NEXT: fmov s1, w8
+; NOSVE2BITPERM-NEXT: and w11, w10, w9
+; NOSVE2BITPERM-NEXT: eor w13, w1, w9
+; NOSVE2BITPERM-NEXT: and w9, w9, #0xfe
; NOSVE2BITPERM-NEXT: eor w10, w10, w11
; NOSVE2BITPERM-NEXT: and w11, w11, #0xfe
-; NOSVE2BITPERM-NEXT: orr w8, w13, w8, lsr #1
+; NOSVE2BITPERM-NEXT: orr w9, w13, w9, lsr #1
; NOSVE2BITPERM-NEXT: pmul v1.8b, v1.8b, v0.8b
; NOSVE2BITPERM-NEXT: fmov w12, s1
-; NOSVE2BITPERM-NEXT: bic w9, w9, w12
-; NOSVE2BITPERM-NEXT: fmov s1, w9
-; NOSVE2BITPERM-NEXT: orr w9, w10, w11, lsr #1
-; NOSVE2BITPERM-NEXT: and w10, w12, w8
-; NOSVE2BITPERM-NEXT: eor w8, w8, w10
-; NOSVE2BITPERM-NEXT: and w11, w9, w10
+; NOSVE2BITPERM-NEXT: bic w8, w8, w12
+; NOSVE2BITPERM-NEXT: fmov s1, w8
+; NOSVE2BITPERM-NEXT: orr w8, w10, w11, lsr #1
+; NOSVE2BITPERM-NEXT: and w10, w12, w9
+; NOSVE2BITPERM-NEXT: eor w9, w9, w10
+; NOSVE2BITPERM-NEXT: and w11, w8, w10
; NOSVE2BITPERM-NEXT: and w10, w10, #0xfc
; NOSVE2BITPERM-NEXT: pmul v0.8b, v1.8b, v0.8b
-; NOSVE2BITPERM-NEXT: orr w8, w8, w10, lsr #2
-; NOSVE2BITPERM-NEXT: eor w9, w9, w11
+; NOSVE2BITPERM-NEXT: orr w9, w9, w10, lsr #2
+; NOSVE2BITPERM-NEXT: eor w8, w8, w11
; NOSVE2BITPERM-NEXT: and w11, w11, #0xfc
-; NOSVE2BITPERM-NEXT: orr w9, w9, w11, lsr #2
+; NOSVE2BITPERM-NEXT: orr w8, w8, w11, lsr #2
; NOSVE2BITPERM-NEXT: fmov w10, s0
-; NOSVE2BITPERM-NEXT: and w8, w10, w8
-; NOSVE2BITPERM-NEXT: and w8, w9, w8
-; NOSVE2BITPERM-NEXT: eor w9, w9, w8
-; NOSVE2BITPERM-NEXT: and w8, w8, #0xf0
-; NOSVE2BITPERM-NEXT: orr w0, w9, w8, lsr #4
+; NOSVE2BITPERM-NEXT: and w9, w10, w9
+; NOSVE2BITPERM-NEXT: and w9, w8, w9
+; NOSVE2BITPERM-NEXT: eor w8, w8, w9
+; NOSVE2BITPERM-NEXT: and w9, w9, #0xf0
+; NOSVE2BITPERM-NEXT: orr w0, w8, w9, lsr #4
; NOSVE2BITPERM-NEXT: ret
%res = call i8 @llvm.pext.i8(i8 %val, i8 %mask)
ret i8 %res
More information about the llvm-commits
mailing list