[llvm] [AArch64] Sign-extend promoted extracts feeding wider DUPs (PR #221186)

Oscar Priego via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 8 00:03:37 PDT 2026


https://github.com/Opriego updated https://github.com/llvm/llvm-project/pull/221186

>From bdc054cf6c229d29b58a6cf22d9efa81eaf8d174 Mon Sep 17 00:00:00 2001
From: Oscar Priego Verdugo <oscar.priegov at gmail.com>
Date: Fri, 4 Sep 2026 04:04:46 -0600
Subject: [PATCH] [AArch64] Sign-extend promoted extracts feeding wider DUPs

---
 .../Target/AArch64/AArch64ISelLowering.cpp    | 54 +++++++++++------
 .../AArch64/arm64-be-bool-splat-dup.ll        | 22 +++++++
 .../AArch64/intrinsic-vector-match-sve2.ll    | 58 ++++++++-----------
 3 files changed, 83 insertions(+), 51 deletions(-)
 create mode 100644 llvm/test/CodeGen/AArch64/arm64-be-bool-splat-dup.ll

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index d7fafc6c840c0..1e974c312b135 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -15879,6 +15879,21 @@ static unsigned getDUPLANEOp(EVT EltType) {
   llvm_unreachable("Invalid vector element type?");
 }
 
+static bool isPromotedExtractForWiderVectorElement(SDValue Value, EVT VecVT) {
+  if (Value.getOpcode() != ISD::EXTRACT_VECTOR_ELT)
+    return false;
+
+  // A promoted extract only defines the source vector element's bits. Scalar
+  // vector construction with wider elements would make promoted high bits
+  // observable.
+  EVT ValueVT = Value.getValueType();
+  EVT ExtractVT = Value.getOperand(0).getValueType().getVectorElementType();
+  EVT EltVT = VecVT.getVectorElementType();
+  return ValueVT.isInteger() && ExtractVT.isInteger() && EltVT.isInteger() &&
+         ExtractVT.bitsLT(ValueVT) && ExtractVT.bitsLT(EltVT) &&
+         EltVT.bitsLE(ValueVT);
+}
+
 static SDValue constructDup(SDValue V, int Lane, SDLoc DL, EVT VT,
                             unsigned Opcode, SelectionDAG &DAG) {
   // Try to eliminate a bitcasted extract subvector before a DUPLANE.
@@ -17263,8 +17278,9 @@ SDValue AArch64TargetLowering::LowerBUILD_VECTOR(SDValue Op,
   // SCALAR_TO_VECTOR, except for when we have a single-element constant vector
   // as SimplifyDemandedBits will just turn that back into BUILD_VECTOR.
   if (isOnlyLowElement && !(NumElts == 1 && isIntOrFPConstant(Value))) {
-    LLVM_DEBUG(dbgs() << "LowerBUILD_VECTOR: only low element used, creating 1 "
-                         "SCALAR_TO_VECTOR node\n");
+    LLVM_DEBUG(
+        dbgs() << "LowerBUILD_VECTOR: only low element used, creating 1 "
+                  "SCALAR_TO_VECTOR node\n");
     return DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, VT, Value);
   }
 
@@ -17349,24 +17365,26 @@ SDValue AArch64TargetLowering::LowerBUILD_VECTOR(SDValue Op,
     if (!isConstant) {
       if (Value.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
           Value.getValueType() != VT) {
-        LLVM_DEBUG(
-            dbgs() << "LowerBUILD_VECTOR: use DUP for non-constant splats\n");
-        return DAG.getNode(AArch64ISD::DUP, DL, VT, Value);
-      }
-
-      // This is actually a DUPLANExx operation, which keeps everything vectory.
+        if (!isPromotedExtractForWiderVectorElement(Value, VT)) {
+          LLVM_DEBUG(
+              dbgs() << "LowerBUILD_VECTOR: use DUP for non-constant splats\n");
+          return DAG.getNode(AArch64ISD::DUP, DL, VT, Value);
+        }
+      } else {
+        // This is actually a DUPLANExx operation, which keeps everything vectory.
+
+        SDValue Lane = Value.getOperand(1);
+        Value = Value.getOperand(0);
+        if (Value.getValueSizeInBits() == 64) {
+          LLVM_DEBUG(
+              dbgs() << "LowerBUILD_VECTOR: DUPLANE works on 128-bit vectors, "
+                        "widening it\n");
+          Value = WidenVector(Value, DAG);
+        }
 
-      SDValue Lane = Value.getOperand(1);
-      Value = Value.getOperand(0);
-      if (Value.getValueSizeInBits() == 64) {
-        LLVM_DEBUG(
-            dbgs() << "LowerBUILD_VECTOR: DUPLANE works on 128-bit vectors, "
-                      "widening it\n");
-        Value = WidenVector(Value, DAG);
+        unsigned Opcode = getDUPLANEOp(VT.getVectorElementType());
+        return DAG.getNode(Opcode, DL, VT, Value, Lane);
       }
-
-      unsigned Opcode = getDUPLANEOp(VT.getVectorElementType());
-      return DAG.getNode(Opcode, DL, VT, Value, Lane);
     }
 
     if (VT.getVectorElementType().isFloatingPoint()) {
diff --git a/llvm/test/CodeGen/AArch64/arm64-be-bool-splat-dup.ll b/llvm/test/CodeGen/AArch64/arm64-be-bool-splat-dup.ll
new file mode 100644
index 0000000000000..923a5ff84d61c
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/arm64-be-bool-splat-dup.ll
@@ -0,0 +1,22 @@
+; RUN: llc -O2 -mtriple=aarch64_be-linux-gnu -verify-machineinstrs < %s | FileCheck %s --check-prefix=BE
+
+; Regression test for https://github.com/llvm/llvm-project/issues/221122.
+; Do not form a scalar-fed halfword DUP from a promoted byte extract and then
+; consume the result through byte operations.
+
+define i1 @check(<16 x i32> %fr, <4 x i32> %fr1) {
+; BE-LABEL: check:
+; BE:       dup v[[DUP:[0-9]+]].8b, v{{[0-9]+}}.b[0]
+; BE:       and {{.*}}v[[DUP]].8b
+; BE:       uminv b0,
+entry:
+  %cmp16 = icmp eq <16 x i32> %fr, <i32 3208, i32 1334, i32 28764, i32 35679, i32 2789, i32 13028, i32 4754, i32 168364, i32 91254, i32 12399, i32 22848, i32 8174, i32 307964, i32 146829, i32 22009, i32 32668>
+  %cmp4 = icmp eq <4 x i32> %fr1, <i32 11594, i32 447564, i32 202404, i32 31619>
+  %splat = shufflevector <16 x i1> %cmp16, <16 x i1> poison, <4 x i32> zeroinitializer
+  %both = and <4 x i1> %splat, %cmp4
+  %pad = shufflevector <4 x i1> %both, <4 x i1> poison, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 4, i32 4, i32 4, i32 4, i32 4, i32 4, i32 4, i32 4, i32 4, i32 4, i32 4>
+  %merge = shufflevector <16 x i1> %pad, <16 x i1> %cmp16, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+  %bits = bitcast <16 x i1> %merge to i16
+  %ok = icmp eq i16 %bits, -1
+  ret i1 %ok
+}
diff --git a/llvm/test/CodeGen/AArch64/intrinsic-vector-match-sve2.ll b/llvm/test/CodeGen/AArch64/intrinsic-vector-match-sve2.ll
index 62b20ef9694ce..7dbabf18b958a 100644
--- a/llvm/test/CodeGen/AArch64/intrinsic-vector-match-sve2.ll
+++ b/llvm/test/CodeGen/AArch64/intrinsic-vector-match-sve2.ll
@@ -457,51 +457,43 @@ define <3 x i1> @match_v3i8_v3i1(<3 x i8> %op1, <8 x i8> %op2, <3 x i1> %mask) #
 ; CHECK:       // %bb.0:
 ; CHECK-NEXT:    fmov s1, w0
 ; CHECK-NEXT:    // kill: def $d0 killed $d0 def $q0
-; CHECK-NEXT:    umov w8, v0.b[1]
-; CHECK-NEXT:    umov w9, v0.b[0]
-; CHECK-NEXT:    umov w10, v0.b[2]
-; CHECK-NEXT:    umov w11, v0.b[3]
-; CHECK-NEXT:    umov w12, v0.b[4]
-; CHECK-NEXT:    umov w13, v0.b[5]
+; CHECK-NEXT:    dup v2.8b, v0.b[1]
+; CHECK-NEXT:    dup v3.8b, v0.b[0]
+; CHECK-NEXT:    dup v4.8b, v0.b[2]
+; CHECK-NEXT:    dup v5.8b, v0.b[3]
+; CHECK-NEXT:    dup v6.8b, v0.b[4]
+; CHECK-NEXT:    dup v7.8b, v0.b[5]
+; CHECK-NEXT:    dup v16.8b, v0.b[6]
+; CHECK-NEXT:    dup v0.8b, v0.b[7]
 ; CHECK-NEXT:    mov v1.h[1], w1
-; CHECK-NEXT:    dup v2.4h, w8
-; CHECK-NEXT:    umov w8, v0.b[6]
-; CHECK-NEXT:    dup v3.4h, w9
-; CHECK-NEXT:    dup v4.4h, w10
-; CHECK-NEXT:    dup v5.4h, w11
-; CHECK-NEXT:    dup v6.4h, w12
-; CHECK-NEXT:    dup v7.4h, w13
-; CHECK-NEXT:    mov v1.h[2], w2
-; CHECK-NEXT:    dup v16.4h, w8
 ; CHECK-NEXT:    bic v2.4h, #255, lsl #8
 ; CHECK-NEXT:    bic v3.4h, #255, lsl #8
 ; CHECK-NEXT:    bic v4.4h, #255, lsl #8
 ; CHECK-NEXT:    bic v5.4h, #255, lsl #8
 ; CHECK-NEXT:    bic v6.4h, #255, lsl #8
 ; CHECK-NEXT:    bic v7.4h, #255, lsl #8
-; CHECK-NEXT:    umov w8, v0.b[7]
-; CHECK-NEXT:    bic v1.4h, #255, lsl #8
 ; CHECK-NEXT:    bic v16.4h, #255, lsl #8
-; CHECK-NEXT:    cmeq v0.4h, v1.4h, v2.4h
-; CHECK-NEXT:    cmeq v2.4h, v1.4h, v3.4h
-; CHECK-NEXT:    cmeq v3.4h, v1.4h, v4.4h
-; CHECK-NEXT:    cmeq v4.4h, v1.4h, v5.4h
-; CHECK-NEXT:    cmeq v5.4h, v1.4h, v6.4h
-; CHECK-NEXT:    cmeq v6.4h, v1.4h, v7.4h
-; CHECK-NEXT:    orr v0.8b, v2.8b, v0.8b
-; CHECK-NEXT:    orr v2.8b, v3.8b, v4.8b
-; CHECK-NEXT:    orr v4.8b, v5.8b, v6.8b
+; CHECK-NEXT:    bic v0.4h, #255, lsl #8
+; CHECK-NEXT:    mov v1.h[2], w2
+; CHECK-NEXT:    bic v1.4h, #255, lsl #8
+; CHECK-NEXT:    cmeq v2.4h, v1.4h, v2.4h
+; CHECK-NEXT:    cmeq v3.4h, v1.4h, v3.4h
+; CHECK-NEXT:    cmeq v4.4h, v1.4h, v4.4h
+; CHECK-NEXT:    cmeq v5.4h, v1.4h, v5.4h
+; CHECK-NEXT:    cmeq v6.4h, v1.4h, v6.4h
+; CHECK-NEXT:    cmeq v7.4h, v1.4h, v7.4h
+; CHECK-NEXT:    cmeq v0.4h, v1.4h, v0.4h
+; CHECK-NEXT:    orr v2.8b, v3.8b, v2.8b
+; CHECK-NEXT:    orr v3.8b, v4.8b, v5.8b
+; CHECK-NEXT:    orr v4.8b, v6.8b, v7.8b
 ; CHECK-NEXT:    cmeq v5.4h, v1.4h, v16.4h
-; CHECK-NEXT:    dup v3.4h, w8
-; CHECK-NEXT:    orr v0.8b, v0.8b, v2.8b
-; CHECK-NEXT:    orr v2.8b, v4.8b, v5.8b
+; CHECK-NEXT:    orr v2.8b, v2.8b, v3.8b
+; CHECK-NEXT:    orr v3.8b, v4.8b, v5.8b
 ; CHECK-NEXT:    fmov s4, w3
-; CHECK-NEXT:    bic v3.4h, #255, lsl #8
 ; CHECK-NEXT:    mov v4.h[1], w4
-; CHECK-NEXT:    orr v0.8b, v0.8b, v2.8b
-; CHECK-NEXT:    cmeq v1.4h, v1.4h, v3.4h
+; CHECK-NEXT:    orr v2.8b, v2.8b, v3.8b
+; CHECK-NEXT:    orr v0.8b, v2.8b, v0.8b
 ; CHECK-NEXT:    mov v4.h[2], w5
-; CHECK-NEXT:    orr v0.8b, v0.8b, v1.8b
 ; CHECK-NEXT:    shl v0.4h, v0.4h, #8
 ; CHECK-NEXT:    shl v1.4h, v4.4h, #15
 ; CHECK-NEXT:    sshr v0.4h, v0.4h, #8



More information about the llvm-commits mailing list