[llvm] [VectorCombine] Count the shuffle as removed only when its binop is removed in foldShuffleOfBinops (PR #228473)

Weiwen He via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 07:58:09 PDT 2026


https://github.com/he-weiwen created https://github.com/llvm/llvm-project/pull/228473

In `MergeInner` , when folding `shuffle(binop(shuffle(x), z), binop(…))` to `binop(shuffle(x, …), …)` , if the binop has more than one use, both the binop and the inner shuffle cannot be removed. The current cost model only accounts for the binop but missed the shuffle.

Noticed this while working on ##228472, but it’s a separate issue I think.

>From 6fdeb24be86ad30837a16a8bb99f1821e98fc225 Mon Sep 17 00:00:00 2001
From: Weiwen He <he.weiwen at outlook.com>
Date: Thu, 1 Oct 2026 10:16:07 +0100
Subject: [PATCH 1/2] [VectorCombine] Add baseline tests for shuffles of
 multi-use binops with inner shuffles (NFC)

---
 .../VectorCombine/X86/shuffle-of-binops.ll    | 19 +++++++++++++++++++
 1 file changed, 19 insertions(+)

diff --git a/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll b/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll
index e1e95019680434..5390bf2bc6baf0 100644
--- a/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll
@@ -249,6 +249,25 @@ define <4 x i32> @shuf_sdiv_v4i32_multiuse_both(<4 x i32> %x, <4 x i32> %y, <4 x
   ret <4 x i32> %r
 }
 
+define <4 x i32> @shuf_add_v4i32_inner_shuffles_multiuse_rhs(<4 x i32> %a, <4 x i32> %b, <4 x i32> %y) {
+; CHECK-LABEL: define <4 x i32> @shuf_add_v4i32_inner_shuffles_multiuse_rhs(
+; CHECK-SAME: <4 x i32> [[A:%.*]], <4 x i32> [[B:%.*]], <4 x i32> [[Y:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[B0:%.*]] = shufflevector <4 x i32> [[B]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[R:%.*]] = add <4 x i32> [[B0]], [[Y]]
+; CHECK-NEXT:    call void @use(<4 x i32> [[R]])
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[A]], <4 x i32> [[B]], <4 x i32> <i32 3, i32 2, i32 5, i32 4>
+; CHECK-NEXT:    [[S:%.*]] = add <4 x i32> [[TMP1]], [[Y]]
+; CHECK-NEXT:    ret <4 x i32> [[S]]
+;
+  %a0 = shufflevector <4 x i32> %a, <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+  %b0 = shufflevector <4 x i32> %b, <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+  %l = add <4 x i32> %a0, %y
+  %r = add <4 x i32> %b0, %y
+  call void @use(<4 x i32> %r)
+  %s = shufflevector <4 x i32> %l, <4 x i32> %r, <4 x i32> <i32 0, i32 1, i32 6, i32 7>
+  ret <4 x i32> %s
+}
+
 ; non-matching operands (not commutable)
 
 define <4 x float> @shuf_fdiv_v4f32_no_common_op(<4 x float> %x, <4 x float> %y, <4 x float> %z, <4 x float> %w) {

>From d96fa1389166b502c4d8f998eb8cf047e68a9b63 Mon Sep 17 00:00:00 2001
From: Weiwen He <he.weiwen at outlook.com>
Date: Thu, 1 Oct 2026 10:16:16 +0100
Subject: [PATCH 2/2] [VectorCombine] Count the shuffle removed only when its
 binop is removed in foldShuffleOfBinops

---
 .../lib/Transforms/Vectorize/VectorCombine.cpp | 18 +++++++++++-------
 .../VectorCombine/X86/shuffle-of-binops.ll     |  5 +++--
 2 files changed, 14 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index d16a3a1535cb75..b34673ad80821e 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -2786,7 +2786,8 @@ bool VectorCombine::foldShuffleOfBinops(Instruction &I) {
   // often allow a major reduction in total cost that wouldn't happen as
   // individual folds.
   auto MergeInner = [&](Value *&Op, int Offset, MutableArrayRef<int> Mask,
-                        TTI::TargetCostKind CostKind) -> bool {
+                        TTI::TargetCostKind CostKind,
+                        Instruction *BinOp) -> bool {
     Value *InnerOp;
     ArrayRef<int> InnerMask;
     if (match(Op, m_OneUse(m_Shuffle(m_Value(InnerOp), m_Undef(),
@@ -2799,17 +2800,20 @@ bool VectorCombine::foldShuffleOfBinops(Instruction &I) {
           M = InnerMask[M - Offset];
           M = 0 <= M ? M + Offset : M;
         }
-      OldCost += TTI.getInstructionCost(cast<Instruction>(Op), CostKind);
+      // Op is only removed if the binop using it is removed too.
+      bool Removed = BinOp->hasOneUser();
+      if (Removed)
+        OldCost += TTI.getInstructionCost(cast<Instruction>(Op), CostKind);
       Op = InnerOp;
-      return true;
+      return Removed;
     }
     return false;
   };
   bool ReducedInstCount = false;
-  ReducedInstCount |= MergeInner(X, 0, NewMask0, CostKind);
-  ReducedInstCount |= MergeInner(Y, 0, NewMask1, CostKind);
-  ReducedInstCount |= MergeInner(Z, NumSrcElts, NewMask0, CostKind);
-  ReducedInstCount |= MergeInner(W, NumSrcElts, NewMask1, CostKind);
+  ReducedInstCount |= MergeInner(X, 0, NewMask0, CostKind, LHS);
+  ReducedInstCount |= MergeInner(Y, 0, NewMask1, CostKind, LHS);
+  ReducedInstCount |= MergeInner(Z, NumSrcElts, NewMask0, CostKind, RHS);
+  ReducedInstCount |= MergeInner(W, NumSrcElts, NewMask1, CostKind, RHS);
   bool SingleSrcBinOp = (X == Y) && (Z == W) && (NewMask0 == NewMask1);
   // SingleSrcBinOp only reduces instruction count if we also eliminate the
   // original binop(s). If binops have multiple uses, they won't be eliminated.
diff --git a/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll b/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll
index 5390bf2bc6baf0..20960f23f1dbc9 100644
--- a/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll
@@ -252,11 +252,12 @@ define <4 x i32> @shuf_sdiv_v4i32_multiuse_both(<4 x i32> %x, <4 x i32> %y, <4 x
 define <4 x i32> @shuf_add_v4i32_inner_shuffles_multiuse_rhs(<4 x i32> %a, <4 x i32> %b, <4 x i32> %y) {
 ; CHECK-LABEL: define <4 x i32> @shuf_add_v4i32_inner_shuffles_multiuse_rhs(
 ; CHECK-SAME: <4 x i32> [[A:%.*]], <4 x i32> [[B:%.*]], <4 x i32> [[Y:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[A0:%.*]] = shufflevector <4 x i32> [[A]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
 ; CHECK-NEXT:    [[B0:%.*]] = shufflevector <4 x i32> [[B]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[L:%.*]] = add <4 x i32> [[A0]], [[Y]]
 ; CHECK-NEXT:    [[R:%.*]] = add <4 x i32> [[B0]], [[Y]]
 ; CHECK-NEXT:    call void @use(<4 x i32> [[R]])
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[A]], <4 x i32> [[B]], <4 x i32> <i32 3, i32 2, i32 5, i32 4>
-; CHECK-NEXT:    [[S:%.*]] = add <4 x i32> [[TMP1]], [[Y]]
+; CHECK-NEXT:    [[S:%.*]] = shufflevector <4 x i32> [[L]], <4 x i32> [[R]], <4 x i32> <i32 0, i32 1, i32 6, i32 7>
 ; CHECK-NEXT:    ret <4 x i32> [[S]]
 ;
   %a0 = shufflevector <4 x i32> %a, <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>



More information about the llvm-commits mailing list