[llvm] Fold patterns which uses v4i32 type for equality comparison on v2i64 type (PR #184328)

Fuad Ismail via llvm-commits llvm-commits at lists.llvm.org
Tue Mar 3 03:50:43 PST 2026


https://github.com/fuad1502 created https://github.com/llvm/llvm-project/pull/184328

This PR partially recognize v2i64 comparison patterns described in #32961. In particular, this PR handles the equality comparison (`eq`, `ne`).

The recognized pattern along with the transformation proof can be seen here: https://alive2.llvm.org/ce/z/8QqwW1

New lit test:
- Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll

>From 5c444710bf7fd07d8405aa8b6802cff845f9994a Mon Sep 17 00:00:00 2001
From: Fuad Ismail <fuad1502 at gmail.com>
Date: Tue, 3 Mar 2026 07:15:43 +0700
Subject: [PATCH 1/5] Add folding v4i32 equals-shuffle-and pattern to v2i64
 equals lit test

---
 .../fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll  | 34 +++++++++++++++++++
 1 file changed, 34 insertions(+)
 create mode 100644 llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll

diff --git a/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll b/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
new file mode 100644
index 0000000000000..3c1b98af193f5
--- /dev/null
+++ b/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
@@ -0,0 +1,34 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=vector-combine -S | FileCheck %s
+
+define <4 x i32> @cmpeq_epi64_select(<4 x i32> noundef %a, <4 x i32> noundef %b) {
+; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_select(
+; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq <4 x i32> [[A]], [[B]]
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i1> [[CMP]], <4 x i1> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEXT:    [[SHUFFLE:%.*]] = sext <4 x i1> [[TMP1]] to <4 x i32>
+; CHECK-NEXT:    [[SELECT:%.*]] = select <4 x i1> [[CMP]], <4 x i32> [[SHUFFLE]], <4 x i32> zeroinitializer
+; CHECK-NEXT:    ret <4 x i32> [[SELECT]]
+;
+  %cmp = icmp eq <4 x i32> %a, %b
+  %sext = sext <4 x i1> %cmp to <4 x i32>
+  %shuffle = shufflevector <4 x i32> %sext, <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+  %select = select <4 x i1> %cmp, <4 x i32> %shuffle, <4 x i32> zeroinitializer
+  ret <4 x i32> %select
+}
+
+define <4 x i32> @cmpeq_epi64_and(<4 x i32> noundef %a, <4 x i32> noundef %b) {
+; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_and(
+; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq <4 x i32> [[A]], [[B]]
+; CHECK-NEXT:    [[SEXT:%.*]] = sext <4 x i1> [[CMP]] to <4 x i32>
+; CHECK-NEXT:    [[SHUFFLE:%.*]] = shufflevector <4 x i32> [[SEXT]], <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEXT:    [[AND:%.*]] = and <4 x i32> [[SEXT]], [[SHUFFLE]]
+; CHECK-NEXT:    ret <4 x i32> [[AND]]
+;
+  %cmp = icmp eq <4 x i32> %a, %b
+  %sext = sext <4 x i1> %cmp to <4 x i32>
+  %shuffle = shufflevector <4 x i32> %sext, <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+  %and = and <4 x i32> %sext, %shuffle
+  ret <4 x i32> %and
+}

>From b8f17f37a75293051498bac60b03aa40c0e3c365 Mon Sep 17 00:00:00 2001
From: Fuad Ismail <fuad1502 at gmail.com>
Date: Tue, 3 Mar 2026 12:08:15 +0700
Subject: [PATCH 2/5] Apply folding for v4i32 equals-shuffle-and pattern

---
 .../Transforms/Vectorize/VectorCombine.cpp    | 73 +++++++++++++++++++
 .../fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll  | 18 +++--
 2 files changed, 83 insertions(+), 8 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 1f37e435b8080..ee29ce690a435 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -27,12 +27,15 @@
 #include "llvm/Analysis/TargetTransformInfo.h"
 #include "llvm/Analysis/ValueTracking.h"
 #include "llvm/Analysis/VectorUtils.h"
+#include "llvm/IR/DerivedTypes.h"
 #include "llvm/IR/Dominators.h"
 #include "llvm/IR/Function.h"
 #include "llvm/IR/IRBuilder.h"
+#include "llvm/IR/Instruction.h"
 #include "llvm/IR/Instructions.h"
 #include "llvm/IR/PatternMatch.h"
 #include "llvm/Support/CommandLine.h"
+#include "llvm/Support/Debug.h"
 #include "llvm/Support/MathExtras.h"
 #include "llvm/Transforms/Utils/Local.h"
 #include "llvm/Transforms/Utils/LoopUtils.h"
@@ -153,6 +156,7 @@ class VectorCombine {
   bool foldEquivalentReductionCmp(Instruction &I);
   bool foldSelectShuffle(Instruction &I, bool FromReduction = false);
   bool foldInterleaveIntrinsics(Instruction &I);
+  bool foldEqualShuffleAnd(Instruction &I);
   bool shrinkType(Instruction &I);
   bool shrinkLoadForShuffles(Instruction &I);
   bool shrinkPhiOfShuffles(Instruction &I);
@@ -5435,6 +5439,69 @@ bool VectorCombine::foldInterleaveIntrinsics(Instruction &I) {
   return true;
 }
 
+// Prior to SSE4.1, performing equality comparison on v2i64 types require a
+// comparison on v4i32 types using the following pattern:
+//
+// ...
+// %3 = icmp eq <4 x i32> %1, %2
+// %4 = sext <4 x i1> %3 to <4 x i32>
+// %5 = shufflevector <4 x i32> %4, <4 x i32> poison, <4 x i32> <i32 1, i32 0,
+// i32 3, i32 2> %6 = select <4 x i1> %3, <4 x i32> %5, <4 x i32>
+// zeroinitializer
+// ...
+//
+// We should detect such patterns and fold them to:
+//
+// %3 = bitcast <4 x i32> %1 to <2 x i64>
+// %4 = bitcast <4 x i32> %2 to <2 x i64>
+// %5 = icmp eq <2 x i64> %3, %4
+// %6 = bitcast <2 x i64> %5 to <4 x i32>
+//
+bool VectorCombine::foldEqualShuffleAnd(Instruction &I) {
+  // Check pattern existance
+  Value *L, *R;
+  CmpPredicate Pred;
+
+  auto Equal = m_ICmp(Pred, m_Value(L), m_Value(R));
+  SmallVector<int> Mask = {1, 0, 3, 2};
+  auto Shuffle =
+      m_CombineOr(m_SExt(m_Shuffle(Equal, m_Poison(), m_SpecificMask(Mask))),
+                  m_Shuffle(m_SExt(Equal), m_Poison(), m_SpecificMask(Mask)));
+
+  if (!match(&I, m_CombineOr(m_And(m_SExt(Equal), Shuffle),
+                             m_Select(Equal, Shuffle, m_ZeroInt()))) ||
+      !ICmpInst::isEquality(Pred) || !L->getType()->isVectorTy())
+    return false;
+
+  auto *OldVecType = cast<VectorType>(L->getType());
+
+  if (OldVecType->isScalableTy() ||
+      !OldVecType->getElementType()->isIntegerTy())
+    return false;
+
+  int ElementCount = OldVecType->getElementCount().getFixedValue();
+  int ElementBitWidth = OldVecType->getElementType()->getIntegerBitWidth();
+
+  if (ElementCount != 4 || ElementBitWidth != 32)
+    return false;
+
+  LLVM_DEBUG(dbgs() << "VC: Found equal-shuffle-and pattern" << '\n');
+
+  // Perform folding
+  IRBuilder Builder(&I);
+  auto *NewElementType = IntegerType::get(I.getContext(), ElementBitWidth * 2);
+  auto *NewVecType = VectorType::get(NewElementType, ElementCount / 2, false);
+  auto *BitCastL = Builder.CreateBitCast(L, NewVecType);
+  auto *BitCastR = Builder.CreateBitCast(R, NewVecType);
+  auto *Cmp = Builder.CreateICmp(Pred, BitCastL, BitCastR);
+  auto *SExt = Builder.CreateSExt(Cmp, NewVecType);
+  auto *BitCastCmp = Builder.CreateBitCast(SExt, OldVecType);
+
+  replaceValue(I, *BitCastCmp);
+
+  return false;
+}
+
 // Attempt to shrink loads that are only used by shufflevector instructions.
 bool VectorCombine::shrinkLoadForShuffles(Instruction &I) {
   auto *OldLoad = dyn_cast<LoadInst>(&I);
@@ -5777,11 +5844,17 @@ bool VectorCombine::run() {
           return true;
         if (foldBitOpOfCastConstant(I))
           return true;
+        if (foldEqualShuffleAnd(I))
+          return true;
         break;
       case Instruction::PHI:
         if (shrinkPhiOfShuffles(I))
           return true;
         break;
+      case Instruction::Select:
+        if (foldEqualShuffleAnd(I))
+          return true;
+        break;
       default:
         if (shrinkType(I))
           return true;
diff --git a/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll b/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
index 3c1b98af193f5..42f3222ae3d27 100644
--- a/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
+++ b/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
@@ -4,10 +4,11 @@
 define <4 x i32> @cmpeq_epi64_select(<4 x i32> noundef %a, <4 x i32> noundef %b) {
 ; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_select(
 ; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
-; CHECK-NEXT:    [[CMP:%.*]] = icmp eq <4 x i32> [[A]], [[B]]
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i1> [[CMP]], <4 x i1> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
-; CHECK-NEXT:    [[SHUFFLE:%.*]] = sext <4 x i1> [[TMP1]] to <4 x i32>
-; CHECK-NEXT:    [[SELECT:%.*]] = select <4 x i1> [[CMP]], <4 x i32> [[SHUFFLE]], <4 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast <4 x i32> [[A]] to <2 x i64>
+; CHECK-NEXT:    [[TMP2:%.*]] = bitcast <4 x i32> [[B]] to <2 x i64>
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq <2 x i64> [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[TMP4:%.*]] = sext <2 x i1> [[TMP3]] to <2 x i64>
+; CHECK-NEXT:    [[SELECT:%.*]] = bitcast <2 x i64> [[TMP4]] to <4 x i32>
 ; CHECK-NEXT:    ret <4 x i32> [[SELECT]]
 ;
   %cmp = icmp eq <4 x i32> %a, %b
@@ -20,10 +21,11 @@ define <4 x i32> @cmpeq_epi64_select(<4 x i32> noundef %a, <4 x i32> noundef %b)
 define <4 x i32> @cmpeq_epi64_and(<4 x i32> noundef %a, <4 x i32> noundef %b) {
 ; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_and(
 ; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
-; CHECK-NEXT:    [[CMP:%.*]] = icmp eq <4 x i32> [[A]], [[B]]
-; CHECK-NEXT:    [[SEXT:%.*]] = sext <4 x i1> [[CMP]] to <4 x i32>
-; CHECK-NEXT:    [[SHUFFLE:%.*]] = shufflevector <4 x i32> [[SEXT]], <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
-; CHECK-NEXT:    [[AND:%.*]] = and <4 x i32> [[SEXT]], [[SHUFFLE]]
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast <4 x i32> [[A]] to <2 x i64>
+; CHECK-NEXT:    [[TMP2:%.*]] = bitcast <4 x i32> [[B]] to <2 x i64>
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq <2 x i64> [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[TMP4:%.*]] = sext <2 x i1> [[TMP3]] to <2 x i64>
+; CHECK-NEXT:    [[AND:%.*]] = bitcast <2 x i64> [[TMP4]] to <4 x i32>
 ; CHECK-NEXT:    ret <4 x i32> [[AND]]
 ;
   %cmp = icmp eq <4 x i32> %a, %b

>From 10ad4ba00e85fadba6255e53594144aa8973f068 Mon Sep 17 00:00:00 2001
From: Fuad Ismail <fuad1502 at gmail.com>
Date: Tue, 3 Mar 2026 13:13:33 +0700
Subject: [PATCH 3/5] Handle commutated and instruction

---
 llvm/lib/Transforms/Vectorize/VectorCombine.cpp |  2 +-
 .../fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll    | 17 +++++++++++++++++
 2 files changed, 18 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index ee29ce690a435..13559610e37a2 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -5468,7 +5468,7 @@ bool VectorCombine::foldEqualShuffleAnd(Instruction &I) {
       m_CombineOr(m_SExt(m_Shuffle(Equal, m_Poison(), m_SpecificMask(Mask))),
                   m_Shuffle(m_SExt(Equal), m_Poison(), m_SpecificMask(Mask)));
 
-  if (!match(&I, m_CombineOr(m_And(m_SExt(Equal), Shuffle),
+  if (!match(&I, m_CombineOr(m_c_And(m_SExt(Equal), Shuffle),
                              m_Select(Equal, Shuffle, m_ZeroInt()))) ||
       !ICmpInst::isEquality(Pred) || !L->getType()->isVectorTy())
     return false;
diff --git a/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll b/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
index 42f3222ae3d27..983a5a6708609 100644
--- a/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
+++ b/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
@@ -34,3 +34,20 @@ define <4 x i32> @cmpeq_epi64_and(<4 x i32> noundef %a, <4 x i32> noundef %b) {
   %and = and <4 x i32> %sext, %shuffle
   ret <4 x i32> %and
 }
+
+define <4 x i32> @cmpeq_epi64_commutated_and(<4 x i32> noundef %a, <4 x i32> noundef %b) {
+; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_commutated_and(
+; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast <4 x i32> [[A]] to <2 x i64>
+; CHECK-NEXT:    [[TMP2:%.*]] = bitcast <4 x i32> [[B]] to <2 x i64>
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq <2 x i64> [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[TMP4:%.*]] = sext <2 x i1> [[TMP3]] to <2 x i64>
+; CHECK-NEXT:    [[AND:%.*]] = bitcast <2 x i64> [[TMP4]] to <4 x i32>
+; CHECK-NEXT:    ret <4 x i32> [[AND]]
+;
+  %cmp = icmp eq <4 x i32> %a, %b
+  %sext = sext <4 x i1> %cmp to <4 x i32>
+  %shuffle = shufflevector <4 x i32> %sext, <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+  %and = and <4 x i32> %shuffle, %sext
+  ret <4 x i32> %and
+}

>From 79bbff7d352bfd6f49990f76b2e04621695ab128 Mon Sep 17 00:00:00 2001
From: Fuad Ismail <fuad1502 at gmail.com>
Date: Tue, 3 Mar 2026 16:44:27 +0700
Subject: [PATCH 4/5] Don't fold when intermediate instructions have uses
 outside pattern

---
 .../Transforms/Vectorize/VectorCombine.cpp    | 53 ++++++++++++-----
 .../fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll  | 57 +++++++++++++++++++
 2 files changed, 95 insertions(+), 15 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 13559610e37a2..805aedfc61c04 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -5442,37 +5442,46 @@ bool VectorCombine::foldInterleaveIntrinsics(Instruction &I) {
 // Prior to SSE4.1, performing equality comparison on v2i64 types require a
 // comparison on v4i32 types using the following pattern:
 //
-// ...
 // %3 = icmp eq <4 x i32> %1, %2
+//
 // %4 = sext <4 x i1> %3 to <4 x i32>
+//
 // %5 = shufflevector <4 x i32> %4, <4 x i32> poison, <4 x i32> <i32 1, i32 0,
-// i32 3, i32 2> %6 = select <4 x i1> %3, <4 x i32> %5, <4 x i32>
-// zeroinitializer
-// ...
+// i32 3, i32 2>
+//
+// %6 = select <4 x i1> %3, <4 x i32> %5, <4 x i32> zeroinitializer
+//
+// OR
+//
+// %6 = and <4 x i32> %sext, %shuffle
 //
-// We should detect such patterns and fold them to:
+// We should detect such patterns and fold them into:
 //
 // %3 = bitcast <4 x i32> %1 to <2 x i64>
+//
 // %4 = bitcast <4 x i32> %2 to <2 x i64>
+//
 // %5 = icmp eq <2 x i64> %3, %4
+//
 // %6 = bitcast <2 x i64> %5 to <4 x i32>
 //
 bool VectorCombine::foldEqualShuffleAnd(Instruction &I) {
-  // Check pattern existance
-  Value *L, *R;
+  Value *Equal, *Shuffle, *L, *R;
   CmpPredicate Pred;
-
-  auto Equal = m_ICmp(Pred, m_Value(L), m_Value(R));
   SmallVector<int> Mask = {1, 0, 3, 2};
-  auto Shuffle =
-      m_CombineOr(m_SExt(m_Shuffle(Equal, m_Poison(), m_SpecificMask(Mask))),
-                  m_Shuffle(m_SExt(Equal), m_Poison(), m_SpecificMask(Mask)));
 
-  if (!match(&I, m_CombineOr(m_c_And(m_SExt(Equal), Shuffle),
-                             m_Select(Equal, Shuffle, m_ZeroInt()))) ||
-      !ICmpInst::isEquality(Pred) || !L->getType()->isVectorTy())
+  // Check pattern existance
+  if (!match(&I, m_CombineOr(m_c_And(m_SExt(m_Value(Equal)),
+                                     m_SExtOrSelf(m_Value(Shuffle))),
+                             m_Select(m_Value(Equal),
+                                      m_SExtOrSelf(m_Value(Shuffle)),
+                                      m_ZeroInt()))) ||
+      !match(Shuffle, m_Shuffle(m_SExtOrSelf(m_Specific(Equal)), m_Poison(),
+                                m_SpecificMask(Mask))) ||
+      !match(Equal, m_ICmp(Pred, m_Value(L), m_Value(R))))
     return false;
 
+  // Check argument type
   auto *OldVecType = cast<VectorType>(L->getType());
 
   if (OldVecType->isScalableTy() ||
@@ -5485,6 +5494,20 @@ bool VectorCombine::foldEqualShuffleAnd(Instruction &I) {
   if (ElementCount != 4 || ElementBitWidth != 32)
     return false;
 
+  // Check uses outside pattern
+  if (!Shuffle->hasOneUse())
+    return false;
+
+  for (auto *U : Equal->users()) {
+    if (U == &I || U == Shuffle)
+      continue;
+    if (!isa<llvm::CastInst>(U))
+      return false;
+    for (auto *U : U->users())
+      if (U != &I && U != Shuffle)
+        return false;
+  }
+
   LLVM_DEBUG(dbgs() << "VC: Found equal-shuffle-and pattern" << '\n');
 
   // Perform folding
diff --git a/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll b/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
index 983a5a6708609..2d7e72d359973 100644
--- a/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
+++ b/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
@@ -51,3 +51,60 @@ define <4 x i32> @cmpeq_epi64_commutated_and(<4 x i32> noundef %a, <4 x i32> nou
   %and = and <4 x i32> %shuffle, %sext
   ret <4 x i32> %and
 }
+
+declare void @use.v4i1(<4 x i1>)
+declare void @use.v4i32(<4 x i32>)
+
+define <4 x i32> @cmpeq_epi64_multi_use_cmp(<4 x i32> noundef %a, <4 x i32> noundef %b) {
+; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_multi_use_cmp(
+; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq <4 x i32> [[A]], [[B]]
+; CHECK-NEXT:    call void @use.v4i1(<4 x i1> [[CMP]])
+; CHECK-NEXT:    [[SEXT:%.*]] = sext <4 x i1> [[CMP]] to <4 x i32>
+; CHECK-NEXT:    [[SHUFFLE:%.*]] = shufflevector <4 x i32> [[SEXT]], <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEXT:    [[AND:%.*]] = and <4 x i32> [[SHUFFLE]], [[SEXT]]
+; CHECK-NEXT:    ret <4 x i32> [[AND]]
+;
+  %cmp = icmp eq <4 x i32> %a, %b
+  call void @use.v4i1(<4 x i1> %cmp)
+  %sext = sext <4 x i1> %cmp to <4 x i32>
+  %shuffle = shufflevector <4 x i32> %sext, <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+  %and = and <4 x i32> %shuffle, %sext
+  ret <4 x i32> %and
+}
+
+define <4 x i32> @cmpeq_epi64_multi_use_sext(<4 x i32> noundef %a, <4 x i32> noundef %b) {
+; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_multi_use_sext(
+; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq <4 x i32> [[A]], [[B]]
+; CHECK-NEXT:    [[SEXT:%.*]] = sext <4 x i1> [[CMP]] to <4 x i32>
+; CHECK-NEXT:    call void @use.v4i32(<4 x i32> [[SEXT]])
+; CHECK-NEXT:    [[SHUFFLE:%.*]] = shufflevector <4 x i32> [[SEXT]], <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEXT:    [[AND:%.*]] = and <4 x i32> [[SHUFFLE]], [[SEXT]]
+; CHECK-NEXT:    ret <4 x i32> [[AND]]
+;
+  %cmp = icmp eq <4 x i32> %a, %b
+  %sext = sext <4 x i1> %cmp to <4 x i32>
+  call void @use.v4i32(<4 x i32> %sext)
+  %shuffle = shufflevector <4 x i32> %sext, <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+  %and = and <4 x i32> %shuffle, %sext
+  ret <4 x i32> %and
+}
+
+define <4 x i32> @cmpeq_epi64_multi_use_shuffle(<4 x i32> noundef %a, <4 x i32> noundef %b) {
+; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_multi_use_shuffle(
+; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq <4 x i32> [[A]], [[B]]
+; CHECK-NEXT:    [[SEXT:%.*]] = sext <4 x i1> [[CMP]] to <4 x i32>
+; CHECK-NEXT:    [[SHUFFLE:%.*]] = shufflevector <4 x i32> [[SEXT]], <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEXT:    call void @use.v4i32(<4 x i32> [[SHUFFLE]])
+; CHECK-NEXT:    [[AND:%.*]] = and <4 x i32> [[SHUFFLE]], [[SEXT]]
+; CHECK-NEXT:    ret <4 x i32> [[AND]]
+;
+  %cmp = icmp eq <4 x i32> %a, %b
+  %sext = sext <4 x i1> %cmp to <4 x i32>
+  %shuffle = shufflevector <4 x i32> %sext, <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+  call void @use.v4i32(<4 x i32> %shuffle)
+  %and = and <4 x i32> %shuffle, %sext
+  ret <4 x i32> %and
+}

>From dd0432152c35d705e03983a55f5fae2758699564 Mon Sep 17 00:00:00 2001
From: Fuad Ismail <fuad1502 at gmail.com>
Date: Tue, 3 Mar 2026 18:10:25 +0700
Subject: [PATCH 5/5] Add negative test cases and add icmp condition code check

---
 .../Transforms/Vectorize/VectorCombine.cpp    |  3 +-
 .../fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll  | 64 +++++++++++++++++++
 2 files changed, 66 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 805aedfc61c04..fd2ca3ef7c901 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -5478,7 +5478,8 @@ bool VectorCombine::foldEqualShuffleAnd(Instruction &I) {
                                       m_ZeroInt()))) ||
       !match(Shuffle, m_Shuffle(m_SExtOrSelf(m_Specific(Equal)), m_Poison(),
                                 m_SpecificMask(Mask))) ||
-      !match(Equal, m_ICmp(Pred, m_Value(L), m_Value(R))))
+      !match(Equal, m_ICmp(Pred, m_Value(L), m_Value(R))) ||
+      !CmpInst::isEquality(Pred))
     return false;
 
   // Check argument type
diff --git a/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll b/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
index 2d7e72d359973..4e13afd360673 100644
--- a/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
+++ b/llvm/test/Transforms/VectorCombine/fold-v4i32-eq-shuffle-and-to-v2i64-eq.ll
@@ -108,3 +108,67 @@ define <4 x i32> @cmpeq_epi64_multi_use_shuffle(<4 x i32> noundef %a, <4 x i32>
   %and = and <4 x i32> %shuffle, %sext
   ret <4 x i32> %and
 }
+
+define <4 x i32> @cmpeq_epi64_select_neg_0(<4 x i32> noundef %a, <4 x i32> noundef %b) {
+; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_select_neg_0(
+; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp sgt <4 x i32> [[A]], [[B]]
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i1> [[CMP]], <4 x i1> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEXT:    [[SHUFFLE:%.*]] = sext <4 x i1> [[TMP1]] to <4 x i32>
+; CHECK-NEXT:    [[SELECT:%.*]] = select <4 x i1> [[CMP]], <4 x i32> [[SHUFFLE]], <4 x i32> zeroinitializer
+; CHECK-NEXT:    ret <4 x i32> [[SELECT]]
+;
+  %cmp = icmp sgt <4 x i32> %a, %b
+  %sext = sext <4 x i1> %cmp to <4 x i32>
+  %shuffle = shufflevector <4 x i32> %sext, <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+  %select = select <4 x i1> %cmp, <4 x i32> %shuffle, <4 x i32> zeroinitializer
+  ret <4 x i32> %select
+}
+
+define <4 x i32> @cmpeq_epi64_and_neg_1(<4 x i32> noundef %a, <4 x i32> noundef %b) {
+; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_and_neg_1(
+; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq <4 x i32> [[A]], [[B]]
+; CHECK-NEXT:    [[SEXT:%.*]] = zext <4 x i1> [[CMP]] to <4 x i32>
+; CHECK-NEXT:    [[SHUFFLE:%.*]] = shufflevector <4 x i32> [[SEXT]], <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEXT:    [[AND:%.*]] = and <4 x i32> [[SEXT]], [[SHUFFLE]]
+; CHECK-NEXT:    ret <4 x i32> [[AND]]
+;
+  %cmp = icmp eq <4 x i32> %a, %b
+  %sext = zext <4 x i1> %cmp to <4 x i32>
+  %shuffle = shufflevector <4 x i32> %sext, <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+  %and = and <4 x i32> %sext, %shuffle
+  ret <4 x i32> %and
+}
+
+define <4 x i32> @cmpeq_epi64_select_neg_2(<4 x i32> noundef %a, <4 x i32> noundef %b) {
+; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_select_neg_2(
+; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq <4 x i32> [[A]], [[B]]
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i1> [[CMP]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[SHUFFLE:%.*]] = sext <4 x i1> [[TMP1]] to <4 x i32>
+; CHECK-NEXT:    [[SELECT:%.*]] = select <4 x i1> [[CMP]], <4 x i32> [[SHUFFLE]], <4 x i32> zeroinitializer
+; CHECK-NEXT:    ret <4 x i32> [[SELECT]]
+;
+  %cmp = icmp eq <4 x i32> %a, %b
+  %sext = sext <4 x i1> %cmp to <4 x i32>
+  %shuffle = shufflevector <4 x i32> %sext, <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+  %select = select <4 x i1> %cmp, <4 x i32> %shuffle, <4 x i32> zeroinitializer
+  ret <4 x i32> %select
+}
+
+define <4 x i32> @cmpeq_epi64_select_neg_3(<4 x i32> noundef %a, <4 x i32> noundef %b) {
+; CHECK-LABEL: define <4 x i32> @cmpeq_epi64_select_neg_3(
+; CHECK-SAME: <4 x i32> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq <4 x i32> [[A]], [[B]]
+; CHECK-NEXT:    [[SEXT:%.*]] = sext <4 x i1> [[CMP]] to <4 x i32>
+; CHECK-NEXT:    [[SHUFFLE:%.*]] = shufflevector <4 x i32> [[SEXT]], <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+; CHECK-NEXT:    [[SELECT:%.*]] = select <4 x i1> [[CMP]], <4 x i32> [[SHUFFLE]], <4 x i32> [[SEXT]]
+; CHECK-NEXT:    ret <4 x i32> [[SELECT]]
+;
+  %cmp = icmp eq <4 x i32> %a, %b
+  %sext = sext <4 x i1> %cmp to <4 x i32>
+  %shuffle = shufflevector <4 x i32> %sext, <4 x i32> poison, <4 x i32> <i32 1, i32 0, i32 3, i32 2>
+  %select = select <4 x i1> %cmp, <4 x i32> %shuffle, <4 x i32> %sext
+  ret <4 x i32> %select
+}



More information about the llvm-commits mailing list