[llvm] [AArch64][GlobalISel] Create uzp in shuffle(v, undefined) situations (PR #220535)

Joshua Rodriguez via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 2 09:48:46 PDT 2026


https://github.com/JoshdRod updated https://github.com/llvm/llvm-project/pull/220535

>From feee4edbd4b64fa93e5cb4e9c1ca06ac7bceca06 Mon Sep 17 00:00:00 2001
From: Josh Rodriguez <josh.rodriguez at arm.com>
Date: Wed, 2 Sep 2026 09:52:47 +0000
Subject: [PATCH 01/10] [AArch64][GlobalISel] Add shuffle v, undef -> uzp v, v
 to GISel

In SDAG, a combine exists in ISelLowering that can convert shuffle v, undef -> uzp v, v, when the operands allow it.
Add this optimisation to GlobalISel.
---
 .../GISel/AArch64PostLegalizerLowering.cpp    | 21 ++++++++++++++++++-
 1 file changed, 20 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp b/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
index cf04019c33bf8..c60ae32c8622a 100644
--- a/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
+++ b/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
@@ -205,6 +205,25 @@ bool matchTRN(MachineInstr &MI, MachineRegisterInfo &MRI,
   return true;
 }
 
+/// isUZP_v_undef_Mask - Special case of isUZPMask for canonical form of
+/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
+/// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>,
+static bool isUZP_v_undef_Mask(ArrayRef<int> M, unsigned NumElts, unsigned &WhichResult) {
+  unsigned Half = NumElts / 2;
+  WhichResult = (M[0] == 0 ? 0 : 1);
+  for (unsigned j = 0; j != 2; ++j) {
+    unsigned Idx = WhichResult;
+    for (unsigned i = 0; i != Half; ++i) {
+      int MIdx = M[i + j * Half];
+      if (MIdx >= 0 && (unsigned)MIdx != Idx)
+        return false;
+      Idx += 2;
+    }
+  }
+
+  return true;
+}
+
 /// \return true if a G_SHUFFLE_VECTOR instruction \p MI can be replaced with
 /// a G_UZP1 or G_UZP2 instruction.
 ///
@@ -217,7 +236,7 @@ bool matchUZP(MachineInstr &MI, MachineRegisterInfo &MRI,
   ArrayRef<int> ShuffleMask = MI.getOperand(3).getShuffleMask();
   Register Dst = MI.getOperand(0).getReg();
   unsigned NumElts = MRI.getType(Dst).getNumElements();
-  if (!isUZPMask(ShuffleMask, NumElts, WhichResult))
+  if (!isUZPMask(ShuffleMask, NumElts, WhichResult) && !isUZP_v_undef_Mask(ShuffleMask, NumElts, WhichResult))
     return false;
   unsigned Opc = (WhichResult == 0) ? AArch64::G_UZP1 : AArch64::G_UZP2;
   Register V1 = MI.getOperand(1).getReg();

>From 176c4eae7d979d926634016b1ea21d3ad0ba1503 Mon Sep 17 00:00:00 2001
From: Josh Rodriguez <josh.rodriguez at arm.com>
Date: Wed, 2 Sep 2026 10:09:51 +0000
Subject: [PATCH 02/10] [AArch64][GlobalISel] Add test check

TODO: Add check for upper half. Also add checks for other types.
---
 llvm/test/CodeGen/AArch64/arm64-uzp.ll | 9 +++++++++
 1 file changed, 9 insertions(+)

diff --git a/llvm/test/CodeGen/AArch64/arm64-uzp.ll b/llvm/test/CodeGen/AArch64/arm64-uzp.ll
index 79c346c7aa6d6..502c27157ade9 100644
--- a/llvm/test/CodeGen/AArch64/arm64-uzp.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-uzp.ll
@@ -140,3 +140,12 @@ define <8 x i16> @vuzpQi16_undef012(<8 x i16> %A, <8 x i16> %B) nounwind {
   %tmp5 = xor <8 x i16> %tmp3, %tmp4
   ret <8 x i16> %tmp5
 }
+
+define <8 x i8> @vuzpDi8_poison(<8 x i8> %A) {
+; CHECK-LABEL: vuzpDi8_poison:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    uzp1.8b v0, v0, v0
+; CHECK-NEXT:    ret
+    %r = shufflevector <8 x i8> %A, <8 x i8> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 0, i32 2, i32 4, i32 6>
+    ret <8 x i8> %r
+}

>From 7496ef5d57fe9011c3714406670099cf957d0688 Mon Sep 17 00:00:00 2001
From: Josh Rodriguez <josh.rodriguez at arm.com>
Date: Wed, 2 Sep 2026 10:41:07 +0000
Subject: [PATCH 03/10] [AArch64][GlobalISel] Add uzp1 and uzp2 code generation
 to test check

Previously, test only checked that a uzp1 was correctly generated. Now, check tests both uzp1 and uzp2.
---
 llvm/test/CodeGen/AArch64/arm64-uzp.ll | 10 +++++++---
 1 file changed, 7 insertions(+), 3 deletions(-)

diff --git a/llvm/test/CodeGen/AArch64/arm64-uzp.ll b/llvm/test/CodeGen/AArch64/arm64-uzp.ll
index 502c27157ade9..cea22caf61af8 100644
--- a/llvm/test/CodeGen/AArch64/arm64-uzp.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-uzp.ll
@@ -144,8 +144,12 @@ define <8 x i16> @vuzpQi16_undef012(<8 x i16> %A, <8 x i16> %B) nounwind {
 define <8 x i8> @vuzpDi8_poison(<8 x i8> %A) {
 ; CHECK-LABEL: vuzpDi8_poison:
 ; CHECK:       // %bb.0:
-; CHECK-NEXT:    uzp1.8b v0, v0, v0
+; CHECK-NEXT:    uzp1.8b v1, v0, v0
+; CHECK-NEXT:    uzp2.8b v0, v0, v0
+; CHECK-NEXT:    eor.8b v0, v1, v0
 ; CHECK-NEXT:    ret
-    %r = shufflevector <8 x i8> %A, <8 x i8> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 0, i32 2, i32 4, i32 6>
-    ret <8 x i8> %r
+    %tmp3 = shufflevector <8 x i8> %A, <8 x i8> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 0, i32 2, i32 4, i32 6>
+    %tmp4 = shufflevector <8 x i8> %A, <8 x i8> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 1, i32 3, i32 5, i32 7>
+    %tmp5 = xor <8 x i8> %tmp3, %tmp4
+    ret <8 x i8> %tmp5
 }

>From 4672732cad6cdf7634a593a593c31e5d84fba9a0 Mon Sep 17 00:00:00 2001
From: Josh Rodriguez <josh.rodriguez at arm.com>
Date: Wed, 2 Sep 2026 10:51:16 +0000
Subject: [PATCH 04/10] [AArch64][GlobalISel] Add test for v8i16

Increase test coverage by including an i16 test. Test file commonly includes i8 and i16 tests, so it seems sensible to add one.
---
 llvm/test/CodeGen/AArch64/arm64-uzp.ll | 13 +++++++++++++
 1 file changed, 13 insertions(+)

diff --git a/llvm/test/CodeGen/AArch64/arm64-uzp.ll b/llvm/test/CodeGen/AArch64/arm64-uzp.ll
index cea22caf61af8..8b15a6257f717 100644
--- a/llvm/test/CodeGen/AArch64/arm64-uzp.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-uzp.ll
@@ -153,3 +153,16 @@ define <8 x i8> @vuzpDi8_poison(<8 x i8> %A) {
     %tmp5 = xor <8 x i8> %tmp3, %tmp4
     ret <8 x i8> %tmp5
 }
+
+define <8 x i16> @vuzpQi16_poison(<8 x i16> %A) {
+; CHECK-LABEL: vuzpQi16_poison:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    uzp1.8h v1, v0, v0
+; CHECK-NEXT:    uzp2.8h v0, v0, v0
+; CHECK-NEXT:    eor.16b v0, v1, v0
+; CHECK-NEXT:    ret
+    %tmp3 = shufflevector <8 x i16> %A, <8 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 0, i32 2, i32 4, i32 6>
+    %tmp4 = shufflevector <8 x i16> %A, <8 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 1, i32 3, i32 5, i32 7>
+    %tmp5 = xor <8 x i16> %tmp3, %tmp4
+    ret <8 x i16> %tmp5
+}

>From 7580ce1eb0799adf0541cbcdefcb5e7e76901d93 Mon Sep 17 00:00:00 2001
From: Josh Rodriguez <josh.rodriguez at arm.com>
Date: Wed, 2 Sep 2026 10:51:29 +0000
Subject: [PATCH 05/10] [AArch64][GlobalISel] Add comment explaining purpose of
 tests

---
 llvm/test/CodeGen/AArch64/arm64-uzp.ll | 2 ++
 1 file changed, 2 insertions(+)

diff --git a/llvm/test/CodeGen/AArch64/arm64-uzp.ll b/llvm/test/CodeGen/AArch64/arm64-uzp.ll
index 8b15a6257f717..1b858849225ff 100644
--- a/llvm/test/CodeGen/AArch64/arm64-uzp.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-uzp.ll
@@ -141,6 +141,8 @@ define <8 x i16> @vuzpQi16_undef012(<8 x i16> %A, <8 x i16> %B) nounwind {
   ret <8 x i16> %tmp5
 }
 
+; For valid masks, VUZP should be generated for shuffles with poisons:
+
 define <8 x i8> @vuzpDi8_poison(<8 x i8> %A) {
 ; CHECK-LABEL: vuzpDi8_poison:
 ; CHECK:       // %bb.0:

>From 27eb06a4ee275459063b81714618e30862b72202 Mon Sep 17 00:00:00 2001
From: Josh Rodriguez <josh.rodriguez at arm.com>
Date: Wed, 2 Sep 2026 10:53:12 +0000
Subject: [PATCH 06/10] [AArch64][GlobalISel] Fix formatting

---
 .../Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp   | 6 ++++--
 1 file changed, 4 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp b/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
index c60ae32c8622a..6f1d6405c3b0e 100644
--- a/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
+++ b/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
@@ -208,7 +208,8 @@ bool matchTRN(MachineInstr &MI, MachineRegisterInfo &MRI,
 /// isUZP_v_undef_Mask - Special case of isUZPMask for canonical form of
 /// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
 /// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>,
-static bool isUZP_v_undef_Mask(ArrayRef<int> M, unsigned NumElts, unsigned &WhichResult) {
+static bool isUZP_v_undef_Mask(ArrayRef<int> M, unsigned NumElts,
+                               unsigned &WhichResult) {
   unsigned Half = NumElts / 2;
   WhichResult = (M[0] == 0 ? 0 : 1);
   for (unsigned j = 0; j != 2; ++j) {
@@ -236,7 +237,8 @@ bool matchUZP(MachineInstr &MI, MachineRegisterInfo &MRI,
   ArrayRef<int> ShuffleMask = MI.getOperand(3).getShuffleMask();
   Register Dst = MI.getOperand(0).getReg();
   unsigned NumElts = MRI.getType(Dst).getNumElements();
-  if (!isUZPMask(ShuffleMask, NumElts, WhichResult) && !isUZP_v_undef_Mask(ShuffleMask, NumElts, WhichResult))
+  if (!isUZPMask(ShuffleMask, NumElts, WhichResult) &&
+      !isUZP_v_undef_Mask(ShuffleMask, NumElts, WhichResult))
     return false;
   unsigned Opc = (WhichResult == 0) ? AArch64::G_UZP1 : AArch64::G_UZP2;
   Register V1 = MI.getOperand(1).getReg();

>From 8d6901b929c26787e22c62ccdabcb33e5642b4c3 Mon Sep 17 00:00:00 2001
From: Josh Rodriguez <josh.rodriguez at arm.com>
Date: Wed, 2 Sep 2026 11:00:44 +0000
Subject: [PATCH 07/10] [AArch64][GlobalISel] Combine GISel and SDAG definition
 of isUZP_v_undef_Mask

Combine definition into a single one, which is stored in AArch64PerfectShuffle.h
---
 .../Target/AArch64/AArch64ISelLowering.cpp    | 25 +++----------------
 .../Target/AArch64/AArch64PerfectShuffle.h    | 20 +++++++++++++++
 .../GISel/AArch64PostLegalizerLowering.cpp    | 19 --------------
 3 files changed, 23 insertions(+), 41 deletions(-)

diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index 9e808845954da..d0ce12b272b5e 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -15502,25 +15502,6 @@ static bool isEXTMaskWithSplat(ArrayRef<int> M, EVT VT, unsigned SplatOperand,
   return false;
 }
 
-/// isUZP_v_undef_Mask - Special case of isUZPMask for canonical form of
-/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
-/// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>,
-static bool isUZP_v_undef_Mask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
-  unsigned Half = VT.getVectorNumElements() / 2;
-  WhichResult = (M[0] == 0 ? 0 : 1);
-  for (unsigned j = 0; j != 2; ++j) {
-    unsigned Idx = WhichResult;
-    for (unsigned i = 0; i != Half; ++i) {
-      int MIdx = M[i + j * Half];
-      if (MIdx >= 0 && (unsigned)MIdx != Idx)
-        return false;
-      Idx += 2;
-    }
-  }
-
-  return true;
-}
-
 /// isTRN_v_undef_Mask - Special case of isTRNMask for canonical form of
 /// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
 /// Mask is e.g., <0, 0, 2, 2> instead of <0, 4, 2, 6>.
@@ -16254,7 +16235,7 @@ SDValue AArch64TargetLowering::LowerVECTOR_SHUFFLE(SDValue Op,
     unsigned Opc = (WhichResult == 0) ? AArch64ISD::ZIP1 : AArch64ISD::ZIP2;
     return DAG.getNode(Opc, DL, V1.getValueType(), V1, V1);
   }
-  if (isUZP_v_undef_Mask(ShuffleMask, VT, WhichResult)) {
+  if (isUZP_v_undef_Mask(ShuffleMask, NumElts, WhichResult)) {
     unsigned Opc = (WhichResult == 0) ? AArch64ISD::UZP1 : AArch64ISD::UZP2;
     return DAG.getNode(Opc, DL, V1.getValueType(), V1, V1);
   }
@@ -18042,7 +18023,7 @@ bool AArch64TargetLowering::isShuffleMaskLegal(ArrayRef<int> M, EVT VT) const {
           isUZPMask(M, NumElts, DummyUnsigned) ||
           isZIPMask(M, NumElts, DummyUnsigned, DummyUnsigned) ||
           isTRN_v_undef_Mask(M, VT, DummyUnsigned) ||
-          isUZP_v_undef_Mask(M, VT, DummyUnsigned) ||
+          isUZP_v_undef_Mask(M, NumElts, DummyUnsigned) ||
           isZIP_v_undef_Mask(M, NumElts, DummyUnsigned) ||
           isINSMask(M, NumElts, DummyBool, DummyInt) ||
           isConcatMask(M, VT, VT.getSizeInBits() == 128));
@@ -35651,7 +35632,7 @@ SDValue AArch64TargetLowering::LowerFixedLengthVECTOR_SHUFFLEToSVE(
       return convertFromScalableVector(
           DAG, VT, DAG.getNode(AArch64ISD::ZIP2, DL, ContainerVT, Op1, Op1));
 
-    if (isUZP_v_undef_Mask(ShuffleMask, VT, WhichResult)) {
+    if (isUZP_v_undef_Mask(ShuffleMask, NumElts, WhichResult)) {
       unsigned Opc = (WhichResult == 0) ? AArch64ISD::UZP1 : AArch64ISD::UZP2;
       return convertFromScalableVector(
           DAG, VT, DAG.getNode(Opc, DL, ContainerVT, Op1, Op1));
diff --git a/llvm/lib/Target/AArch64/AArch64PerfectShuffle.h b/llvm/lib/Target/AArch64/AArch64PerfectShuffle.h
index f5b84cc797b66..b4120db6de0c4 100644
--- a/llvm/lib/Target/AArch64/AArch64PerfectShuffle.h
+++ b/llvm/lib/Target/AArch64/AArch64PerfectShuffle.h
@@ -150,6 +150,26 @@ inline bool isUZPMask(ArrayRef<int> M, unsigned NumElts,
   return true;
 }
 
+/// isUZP_v_undef_Mask - Special case of isUZPMask for canonical form of
+/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
+/// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>,
+static bool isUZP_v_undef_Mask(ArrayRef<int> M, unsigned NumElts,
+                               unsigned &WhichResult) {
+  unsigned Half = NumElts / 2;
+  WhichResult = (M[0] == 0 ? 0 : 1);
+  for (unsigned j = 0; j != 2; ++j) {
+    unsigned Idx = WhichResult;
+    for (unsigned i = 0; i != Half; ++i) {
+      int MIdx = M[i + j * Half];
+      if (MIdx >= 0 && (unsigned)MIdx != Idx)
+        return false;
+      Idx += 2;
+    }
+  }
+
+  return true;
+}
+
 /// Return true for trn1 or trn2 masks of the form:
 ///  <0, 8, 2, 10, 4, 12, 6, 14> (WhichResultOut = 0, OperandOrderOut = 0) or
 ///  <1, 9, 3, 11, 5, 13, 7, 15> (WhichResultOut = 1, OperandOrderOut = 0) or
diff --git a/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp b/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
index 6f1d6405c3b0e..57e8b253b7862 100644
--- a/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
+++ b/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
@@ -205,25 +205,6 @@ bool matchTRN(MachineInstr &MI, MachineRegisterInfo &MRI,
   return true;
 }
 
-/// isUZP_v_undef_Mask - Special case of isUZPMask for canonical form of
-/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
-/// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>,
-static bool isUZP_v_undef_Mask(ArrayRef<int> M, unsigned NumElts,
-                               unsigned &WhichResult) {
-  unsigned Half = NumElts / 2;
-  WhichResult = (M[0] == 0 ? 0 : 1);
-  for (unsigned j = 0; j != 2; ++j) {
-    unsigned Idx = WhichResult;
-    for (unsigned i = 0; i != Half; ++i) {
-      int MIdx = M[i + j * Half];
-      if (MIdx >= 0 && (unsigned)MIdx != Idx)
-        return false;
-      Idx += 2;
-    }
-  }
-
-  return true;
-}
 
 /// \return true if a G_SHUFFLE_VECTOR instruction \p MI can be replaced with
 /// a G_UZP1 or G_UZP2 instruction.

>From 944738082452ae8bafec744e0ec7ff67bec7b54e Mon Sep 17 00:00:00 2001
From: Josh Rodriguez <josh.rodriguez at arm.com>
Date: Wed, 2 Sep 2026 13:56:19 +0000
Subject: [PATCH 08/10] [AArch64][GlobalISel] Fix isUZP_v_undef_Mask to inline
 function

Previously, isUZP_v_undef_Mask was a static function. This meant it was not linking to the other files.
Switch the function from static to inline, to allow the other files to use it.
---
 llvm/lib/Target/AArch64/AArch64PerfectShuffle.h | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/lib/Target/AArch64/AArch64PerfectShuffle.h b/llvm/lib/Target/AArch64/AArch64PerfectShuffle.h
index b4120db6de0c4..9683248b2e66b 100644
--- a/llvm/lib/Target/AArch64/AArch64PerfectShuffle.h
+++ b/llvm/lib/Target/AArch64/AArch64PerfectShuffle.h
@@ -153,7 +153,7 @@ inline bool isUZPMask(ArrayRef<int> M, unsigned NumElts,
 /// isUZP_v_undef_Mask - Special case of isUZPMask for canonical form of
 /// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
 /// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>,
-static bool isUZP_v_undef_Mask(ArrayRef<int> M, unsigned NumElts,
+inline bool isUZP_v_undef_Mask(ArrayRef<int> M, unsigned NumElts,
                                unsigned &WhichResult) {
   unsigned Half = NumElts / 2;
   WhichResult = (M[0] == 0 ? 0 : 1);

>From aef3b014c1ed6e25e579061fc73a123eec7e3e43 Mon Sep 17 00:00:00 2001
From: Josh Rodriguez <josh.rodriguez at arm.com>
Date: Wed, 2 Sep 2026 14:01:41 +0000
Subject: [PATCH 09/10] [AArch64][GlobalISel] Remove unnecessary whitespace

---
 llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp | 1 -
 1 file changed, 1 deletion(-)

diff --git a/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp b/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
index 57e8b253b7862..bd3ad68b984f0 100644
--- a/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
+++ b/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
@@ -205,7 +205,6 @@ bool matchTRN(MachineInstr &MI, MachineRegisterInfo &MRI,
   return true;
 }
 
-
 /// \return true if a G_SHUFFLE_VECTOR instruction \p MI can be replaced with
 /// a G_UZP1 or G_UZP2 instruction.
 ///

>From 2d1cc438e5c1b43522fe6964aeda70902bcb388d Mon Sep 17 00:00:00 2001
From: Josh Rodriguez <josh.rodriguez at arm.com>
Date: Wed, 2 Sep 2026 16:48:23 +0000
Subject: [PATCH 10/10] [AArch64][GlobalISel] Set V2 = V1

The correct transformation is:
shuffle_vector v, undef -> uzp v, v.
Currently, PostLegalizerLowering is lowering to:
uzp v, undef.
And this undef is being converted to v in the Virtual Register Rewriter pass.

To avoid the success of this transformation relying on other passes, fix PostLegalizerLowering so that the second argument of uzp is set to v.
---
 .../Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp   | 6 +++---
 1 file changed, 3 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp b/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
index bd3ad68b984f0..283e0cfbebfe1 100644
--- a/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
+++ b/llvm/lib/Target/AArch64/GISel/AArch64PostLegalizerLowering.cpp
@@ -217,12 +217,12 @@ bool matchUZP(MachineInstr &MI, MachineRegisterInfo &MRI,
   ArrayRef<int> ShuffleMask = MI.getOperand(3).getShuffleMask();
   Register Dst = MI.getOperand(0).getReg();
   unsigned NumElts = MRI.getType(Dst).getNumElements();
-  if (!isUZPMask(ShuffleMask, NumElts, WhichResult) &&
-      !isUZP_v_undef_Mask(ShuffleMask, NumElts, WhichResult))
+  bool UZPMask = isUZPMask(ShuffleMask, NumElts, WhichResult);
+  if (!UZPMask && !isUZP_v_undef_Mask(ShuffleMask, NumElts, WhichResult))
     return false;
   unsigned Opc = (WhichResult == 0) ? AArch64::G_UZP1 : AArch64::G_UZP2;
   Register V1 = MI.getOperand(1).getReg();
-  Register V2 = MI.getOperand(2).getReg();
+  Register V2 = MI.getOperand(UZPMask ? 2 : 1).getReg();
   MatchInfo = ShuffleVectorPseudo(Opc, Dst, {V1, V2});
   return true;
 }



More information about the llvm-commits mailing list