[llvm] [AArch64] Combine scalar_to_vector(C) -> buildvector (PR #220948)

David Green via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 6 23:59:59 PDT 2026


https://github.com/davemgreen updated https://github.com/llvm/llvm-project/pull/220948

>From 444ba06031824cb06f5e7b6ded8f8dcf90a9c6db Mon Sep 17 00:00:00 2001
From: David Green <david.green at arm.com>
Date: Thu, 3 Sep 2026 15:41:47 +0100
Subject: [PATCH] [AArch64] Combine scalar_to_vector(C) -> buildvector

This helps keep a single canonical for of constant vector, helping a number of
the existing buildvector combines trigger.
---
 .../Target/AArch64/AArch64ISelLowering.cpp    | 20 ++++---
 .../CodeGen/AArch64/arm64-build-vector.ll     |  2 +-
 llvm/test/CodeGen/AArch64/arm64-tbl.ll        |  2 +-
 llvm/test/CodeGen/AArch64/neon-anyof-splat.ll |  6 +--
 llvm/test/CodeGen/AArch64/pdep.ll             | 41 +++++++-------
 llvm/test/CodeGen/AArch64/pext.ll             | 53 +++++++++----------
 6 files changed, 64 insertions(+), 60 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 22198cd122fc7..622f162690f2a 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -17260,9 +17260,9 @@ SDValue AArch64TargetLowering::LowerBUILD_VECTOR(SDValue Op,
   }
 
   // Convert BUILD_VECTOR where all elements but the lowest are undef into
-  // SCALAR_TO_VECTOR, except for when we have a single-element constant vector
+  // SCALAR_TO_VECTOR, except for when we have a constant vector
   // as SimplifyDemandedBits will just turn that back into BUILD_VECTOR.
-  if (isOnlyLowElement && !(NumElts == 1 && isIntOrFPConstant(Value))) {
+  if (isOnlyLowElement && !isIntOrFPConstant(Value)) {
     LLVM_DEBUG(dbgs() << "LowerBUILD_VECTOR: only low element used, creating 1 "
                          "SCALAR_TO_VECTOR node\n");
     return DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, VT, Value);
@@ -31114,6 +31114,8 @@ static SDValue
 performScalarToVectorCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI,
                              SelectionDAG &DAG) {
   SDLoc DL(N);
+  EVT VT = N->getValueType(0);
+  SDValue N0 = N->getOperand(0);
 
   // If a DUP(Op0) already exists, reuse it for the scalar_to_vector.
   if (DCI.isAfterLegalizeDAG()) {
@@ -31122,6 +31124,14 @@ performScalarToVectorCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI,
       return SDValue(LN, 0);
   }
 
+  if (VT.isFixedLengthVector() && VT.getVectorNumElements() > 1 &&
+      isIntOrFPConstant(N0)) {
+    SDValue Undef = DAG.getPOISON(N0.getValueType());
+    SmallVector<SDValue> Ops(VT.getVectorNumElements(), Undef);
+    Ops[0] = N0;
+    return DAG.getBuildVector(VT, DL, Ops);
+  }
+
   // Let's do below transform.
   //
   //         t34: v4i32 = AArch64ISD::UADDLV t2
@@ -31135,15 +31145,13 @@ performScalarToVectorCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI,
   if (DCI.isBeforeLegalizeOps())
     return SDValue();
 
-  EVT VT = N->getValueType(0);
   if (VT != MVT::v1i64)
     return SDValue();
 
-  SDValue ZEXT = N->getOperand(0);
-  if (ZEXT.getOpcode() != ISD::ZERO_EXTEND || ZEXT.getValueType() != MVT::i64)
+  if (N0.getOpcode() != ISD::ZERO_EXTEND || N0.getValueType() != MVT::i64)
     return SDValue();
 
-  SDValue EXTRACT_VEC_ELT = ZEXT.getOperand(0);
+  SDValue EXTRACT_VEC_ELT = N0.getOperand(0);
   if (EXTRACT_VEC_ELT.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
       EXTRACT_VEC_ELT.getValueType() != MVT::i32)
     return SDValue();
diff --git a/llvm/test/CodeGen/AArch64/arm64-build-vector.ll b/llvm/test/CodeGen/AArch64/arm64-build-vector.ll
index f43aa27823494..f4865fcee5c28 100644
--- a/llvm/test/CodeGen/AArch64/arm64-build-vector.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-build-vector.ll
@@ -26,7 +26,7 @@ define <8 x i16> @build_all_zero(<8 x i16> %a) #1 {
 ; CHECK-SD-LABEL: build_all_zero:
 ; CHECK-SD:       // %bb.0:
 ; CHECK-SD-NEXT:    mov w8, #44672 // =0xae80
-; CHECK-SD-NEXT:    fmov s1, w8
+; CHECK-SD-NEXT:    dup v1.8h, w8
 ; CHECK-SD-NEXT:    mul v0.8h, v0.8h, v1.8h
 ; CHECK-SD-NEXT:    ret
 ;
diff --git a/llvm/test/CodeGen/AArch64/arm64-tbl.ll b/llvm/test/CodeGen/AArch64/arm64-tbl.ll
index 200fc65f01828..67bfdde8d9f9a 100644
--- a/llvm/test/CodeGen/AArch64/arm64-tbl.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-tbl.ll
@@ -442,9 +442,9 @@ define <16 x i8> @shuffled_tbl2_to_tbl4_nonconst_first_mask(<16 x i8> %a, <16 x
 define <16 x i8> @shuffled_tbl2_to_tbl4_nonconst_first_mask2(<16 x i8> %a, <16 x i8> %b, <16 x i8> %c, <16 x i8> %d, i8 %v) {
 ; CHECK-SD-LABEL: shuffled_tbl2_to_tbl4_nonconst_first_mask2:
 ; CHECK-SD:       // %bb.0:
+; CHECK-SD-NEXT:    movi.16b v4, #1
 ; CHECK-SD-NEXT:    mov w8, #1 // =0x1
 ; CHECK-SD-NEXT:    // kill: def $q3 killed $q3 killed $q0_q1_q2_q3 def $q0_q1_q2_q3
-; CHECK-SD-NEXT:    fmov s4, w8
 ; CHECK-SD-NEXT:    // kill: def $q2 killed $q2 killed $q0_q1_q2_q3 def $q0_q1_q2_q3
 ; CHECK-SD-NEXT:    // kill: def $q1 killed $q1 killed $q0_q1_q2_q3 def $q0_q1_q2_q3
 ; CHECK-SD-NEXT:    // kill: def $q0 killed $q0 killed $q0_q1_q2_q3 def $q0_q1_q2_q3
diff --git a/llvm/test/CodeGen/AArch64/neon-anyof-splat.ll b/llvm/test/CodeGen/AArch64/neon-anyof-splat.ll
index bf33b348173f9..b03b9c7a40f13 100644
--- a/llvm/test/CodeGen/AArch64/neon-anyof-splat.ll
+++ b/llvm/test/CodeGen/AArch64/neon-anyof-splat.ll
@@ -16,9 +16,8 @@ define <4 x i32> @any_of_select_vf4(<4 x i32> %mask, <4 x i32> %a, <4 x i32> %b)
 ; CHECK-SD-LABEL: any_of_select_vf4:
 ; CHECK-SD:       // %bb.0:
 ; CHECK-SD-NEXT:    cmlt v0.4s, v0.4s, #0
-; CHECK-SD-NEXT:    movi d3, #0000000000000000
 ; CHECK-SD-NEXT:    addp d0, v0.2d
-; CHECK-SD-NEXT:    cmeq v0.2d, v0.2d, v3.2d
+; CHECK-SD-NEXT:    cmeq v0.2d, v0.2d, #0
 ; CHECK-SD-NEXT:    dup v0.2d, v0.d[0]
 ; CHECK-SD-NEXT:    bsl v0.16b, v1.16b, v2.16b
 ; CHECK-SD-NEXT:    ret
@@ -106,9 +105,8 @@ define <2 x i64> @any_of_select_vf2(<2 x i64> %mask, <2 x i64> %a, <2 x i64> %b)
 ; CHECK-LABEL: any_of_select_vf2:
 ; CHECK:       // %bb.0:
 ; CHECK-NEXT:    cmlt v0.2d, v0.2d, #0
-; CHECK-NEXT:    movi d3, #0000000000000000
 ; CHECK-NEXT:    addp d0, v0.2d
-; CHECK-NEXT:    cmeq v0.2d, v0.2d, v3.2d
+; CHECK-NEXT:    cmeq v0.2d, v0.2d, #0
 ; CHECK-NEXT:    dup v0.2d, v0.d[0]
 ; CHECK-NEXT:    bsl v0.16b, v1.16b, v2.16b
 ; CHECK-NEXT:    ret
diff --git a/llvm/test/CodeGen/AArch64/pdep.ll b/llvm/test/CodeGen/AArch64/pdep.ll
index aad1f51b2107f..7b1ad7d525967 100644
--- a/llvm/test/CodeGen/AArch64/pdep.ll
+++ b/llvm/test/CodeGen/AArch64/pdep.ll
@@ -13,26 +13,25 @@ define i8 @pdep_i8(i8 %val, i8 %mask) nounwind {
 ;
 ; NOSVE2BITPERM-LABEL: pdep_i8:
 ; NOSVE2BITPERM:       // %bb.0:
-; NOSVE2BITPERM-NEXT:    mvn w9, w1
-; NOSVE2BITPERM-NEXT:    mov w8, #-1 // =0xffffffff
-; NOSVE2BITPERM-NEXT:    lsl w9, w9, #1
-; NOSVE2BITPERM-NEXT:    fmov s0, w8
-; NOSVE2BITPERM-NEXT:    fmov s1, w9
+; NOSVE2BITPERM-NEXT:    mvn w8, w1
+; NOSVE2BITPERM-NEXT:    movi v0.2d, #0xffffffffffffffff
+; NOSVE2BITPERM-NEXT:    lsl w8, w8, #1
+; NOSVE2BITPERM-NEXT:    fmov s1, w8
 ; NOSVE2BITPERM-NEXT:    pmul v1.8b, v1.8b, v0.8b
-; NOSVE2BITPERM-NEXT:    fmov w8, s1
-; NOSVE2BITPERM-NEXT:    bic w9, w9, w8
-; NOSVE2BITPERM-NEXT:    and w8, w8, w1
-; NOSVE2BITPERM-NEXT:    fmov s1, w9
-; NOSVE2BITPERM-NEXT:    eor w11, w1, w8
-; NOSVE2BITPERM-NEXT:    and w12, w8, #0xfe
+; NOSVE2BITPERM-NEXT:    fmov w9, s1
+; NOSVE2BITPERM-NEXT:    bic w8, w8, w9
+; NOSVE2BITPERM-NEXT:    and w9, w9, w1
+; NOSVE2BITPERM-NEXT:    fmov s1, w8
+; NOSVE2BITPERM-NEXT:    eor w11, w1, w9
+; NOSVE2BITPERM-NEXT:    and w12, w9, #0xfe
 ; NOSVE2BITPERM-NEXT:    orr w11, w11, w12, lsr #1
 ; NOSVE2BITPERM-NEXT:    pmul v1.8b, v1.8b, v0.8b
 ; NOSVE2BITPERM-NEXT:    fmov w10, s1
-; NOSVE2BITPERM-NEXT:    bic w9, w9, w10
-; NOSVE2BITPERM-NEXT:    fmov s1, w9
-; NOSVE2BITPERM-NEXT:    and w9, w10, w11
-; NOSVE2BITPERM-NEXT:    eor w10, w11, w9
-; NOSVE2BITPERM-NEXT:    and w11, w9, #0xfc
+; NOSVE2BITPERM-NEXT:    bic w8, w8, w10
+; NOSVE2BITPERM-NEXT:    fmov s1, w8
+; NOSVE2BITPERM-NEXT:    and w8, w10, w11
+; NOSVE2BITPERM-NEXT:    eor w10, w11, w8
+; NOSVE2BITPERM-NEXT:    and w11, w8, #0xfc
 ; NOSVE2BITPERM-NEXT:    orr w10, w10, w11, lsr #2
 ; NOSVE2BITPERM-NEXT:    pmul v0.8b, v1.8b, v0.8b
 ; NOSVE2BITPERM-NEXT:    fmov w11, s0
@@ -40,11 +39,11 @@ define i8 @pdep_i8(i8 %val, i8 %mask) nounwind {
 ; NOSVE2BITPERM-NEXT:    and w11, w10, w0, lsl #4
 ; NOSVE2BITPERM-NEXT:    bic w10, w0, w10
 ; NOSVE2BITPERM-NEXT:    orr w10, w10, w11
-; NOSVE2BITPERM-NEXT:    and w11, w9, w10, lsl #2
-; NOSVE2BITPERM-NEXT:    bic w9, w10, w9
-; NOSVE2BITPERM-NEXT:    orr w9, w9, w11
-; NOSVE2BITPERM-NEXT:    and w10, w8, w9, lsl #1
-; NOSVE2BITPERM-NEXT:    bic w8, w9, w8
+; NOSVE2BITPERM-NEXT:    and w11, w8, w10, lsl #2
+; NOSVE2BITPERM-NEXT:    bic w8, w10, w8
+; NOSVE2BITPERM-NEXT:    orr w8, w8, w11
+; NOSVE2BITPERM-NEXT:    and w10, w9, w8, lsl #1
+; NOSVE2BITPERM-NEXT:    bic w8, w8, w9
 ; NOSVE2BITPERM-NEXT:    orr w8, w8, w10
 ; NOSVE2BITPERM-NEXT:    and w0, w8, w1
 ; NOSVE2BITPERM-NEXT:    ret
diff --git a/llvm/test/CodeGen/AArch64/pext.ll b/llvm/test/CodeGen/AArch64/pext.ll
index ecde58f357489..5a9d8913166d6 100644
--- a/llvm/test/CodeGen/AArch64/pext.ll
+++ b/llvm/test/CodeGen/AArch64/pext.ll
@@ -14,43 +14,42 @@ define i8 @pext_i8(i8 %val, i8 %mask) nounwind {
 ;
 ; NOSVE2BITPERM-LABEL: pext_i8:
 ; NOSVE2BITPERM:       // %bb.0:
-; NOSVE2BITPERM-NEXT:    mvn w9, w1
-; NOSVE2BITPERM-NEXT:    mov w8, #-1 // =0xffffffff
+; NOSVE2BITPERM-NEXT:    mvn w8, w1
+; NOSVE2BITPERM-NEXT:    movi v0.2d, #0xffffffffffffffff
 ; NOSVE2BITPERM-NEXT:    and w10, w0, w1
-; NOSVE2BITPERM-NEXT:    lsl w9, w9, #1
-; NOSVE2BITPERM-NEXT:    fmov s0, w8
-; NOSVE2BITPERM-NEXT:    fmov s1, w9
+; NOSVE2BITPERM-NEXT:    lsl w8, w8, #1
+; NOSVE2BITPERM-NEXT:    fmov s1, w8
 ; NOSVE2BITPERM-NEXT:    pmul v1.8b, v1.8b, v0.8b
-; NOSVE2BITPERM-NEXT:    fmov w8, s1
-; NOSVE2BITPERM-NEXT:    bic w9, w9, w8
-; NOSVE2BITPERM-NEXT:    and w8, w8, w1
-; NOSVE2BITPERM-NEXT:    fmov s1, w9
-; NOSVE2BITPERM-NEXT:    and w11, w10, w8
-; NOSVE2BITPERM-NEXT:    eor w13, w1, w8
-; NOSVE2BITPERM-NEXT:    and w8, w8, #0xfe
+; NOSVE2BITPERM-NEXT:    fmov w9, s1
+; NOSVE2BITPERM-NEXT:    bic w8, w8, w9
+; NOSVE2BITPERM-NEXT:    and w9, w9, w1
+; NOSVE2BITPERM-NEXT:    fmov s1, w8
+; NOSVE2BITPERM-NEXT:    and w11, w10, w9
+; NOSVE2BITPERM-NEXT:    eor w13, w1, w9
+; NOSVE2BITPERM-NEXT:    and w9, w9, #0xfe
 ; NOSVE2BITPERM-NEXT:    eor w10, w10, w11
 ; NOSVE2BITPERM-NEXT:    and w11, w11, #0xfe
-; NOSVE2BITPERM-NEXT:    orr w8, w13, w8, lsr #1
+; NOSVE2BITPERM-NEXT:    orr w9, w13, w9, lsr #1
 ; NOSVE2BITPERM-NEXT:    pmul v1.8b, v1.8b, v0.8b
 ; NOSVE2BITPERM-NEXT:    fmov w12, s1
-; NOSVE2BITPERM-NEXT:    bic w9, w9, w12
-; NOSVE2BITPERM-NEXT:    fmov s1, w9
-; NOSVE2BITPERM-NEXT:    orr w9, w10, w11, lsr #1
-; NOSVE2BITPERM-NEXT:    and w10, w12, w8
-; NOSVE2BITPERM-NEXT:    eor w8, w8, w10
-; NOSVE2BITPERM-NEXT:    and w11, w9, w10
+; NOSVE2BITPERM-NEXT:    bic w8, w8, w12
+; NOSVE2BITPERM-NEXT:    fmov s1, w8
+; NOSVE2BITPERM-NEXT:    orr w8, w10, w11, lsr #1
+; NOSVE2BITPERM-NEXT:    and w10, w12, w9
+; NOSVE2BITPERM-NEXT:    eor w9, w9, w10
+; NOSVE2BITPERM-NEXT:    and w11, w8, w10
 ; NOSVE2BITPERM-NEXT:    and w10, w10, #0xfc
 ; NOSVE2BITPERM-NEXT:    pmul v0.8b, v1.8b, v0.8b
-; NOSVE2BITPERM-NEXT:    orr w8, w8, w10, lsr #2
-; NOSVE2BITPERM-NEXT:    eor w9, w9, w11
+; NOSVE2BITPERM-NEXT:    orr w9, w9, w10, lsr #2
+; NOSVE2BITPERM-NEXT:    eor w8, w8, w11
 ; NOSVE2BITPERM-NEXT:    and w11, w11, #0xfc
-; NOSVE2BITPERM-NEXT:    orr w9, w9, w11, lsr #2
+; NOSVE2BITPERM-NEXT:    orr w8, w8, w11, lsr #2
 ; NOSVE2BITPERM-NEXT:    fmov w10, s0
-; NOSVE2BITPERM-NEXT:    and w8, w10, w8
-; NOSVE2BITPERM-NEXT:    and w8, w9, w8
-; NOSVE2BITPERM-NEXT:    eor w9, w9, w8
-; NOSVE2BITPERM-NEXT:    and w8, w8, #0xf0
-; NOSVE2BITPERM-NEXT:    orr w0, w9, w8, lsr #4
+; NOSVE2BITPERM-NEXT:    and w9, w10, w9
+; NOSVE2BITPERM-NEXT:    and w9, w8, w9
+; NOSVE2BITPERM-NEXT:    eor w8, w8, w9
+; NOSVE2BITPERM-NEXT:    and w9, w9, #0xf0
+; NOSVE2BITPERM-NEXT:    orr w0, w8, w9, lsr #4
 ; NOSVE2BITPERM-NEXT:    ret
   %res = call i8 @llvm.pext.i8(i8 %val, i8 %mask)
   ret i8 %res



More information about the llvm-commits mailing list