[llvm] [VectorCombine] Count the shuffle as removed only when its binop is removed in foldShuffleOfBinops (PR #228473)

via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 07:58:57 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms

@llvm/pr-subscribers-vectorizers

Author: Weiwen He (he-weiwen)

<details>
<summary>Changes</summary>

In `MergeInner` , when folding `shuffle(binop(shuffle(x), z), binop(…))` to `binop(shuffle(x, …), …)` , if the binop has more than one use, both the binop and the inner shuffle cannot be removed. The current cost model only accounts for the binop but missed the shuffle.

Noticed this while working on ##<!-- -->228472, but it’s a separate issue I think.

---
Full diff: https://github.com/llvm/llvm-project/pull/228473.diff


2 Files Affected:

- (modified) llvm/lib/Transforms/Vectorize/VectorCombine.cpp (+11-7) 
- (modified) llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll (+20) 


``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index d16a3a1535cb75..b34673ad80821e 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -2786,7 +2786,8 @@ bool VectorCombine::foldShuffleOfBinops(Instruction &I) {
   // often allow a major reduction in total cost that wouldn't happen as
   // individual folds.
   auto MergeInner = [&](Value *&Op, int Offset, MutableArrayRef<int> Mask,
-                        TTI::TargetCostKind CostKind) -> bool {
+                        TTI::TargetCostKind CostKind,
+                        Instruction *BinOp) -> bool {
     Value *InnerOp;
     ArrayRef<int> InnerMask;
     if (match(Op, m_OneUse(m_Shuffle(m_Value(InnerOp), m_Undef(),
@@ -2799,17 +2800,20 @@ bool VectorCombine::foldShuffleOfBinops(Instruction &I) {
           M = InnerMask[M - Offset];
           M = 0 <= M ? M + Offset : M;
         }
-      OldCost += TTI.getInstructionCost(cast<Instruction>(Op), CostKind);
+      // Op is only removed if the binop using it is removed too.
+      bool Removed = BinOp->hasOneUser();
+      if (Removed)
+        OldCost += TTI.getInstructionCost(cast<Instruction>(Op), CostKind);
       Op = InnerOp;
-      return true;
+      return Removed;
     }
     return false;
   };
   bool ReducedInstCount = false;
-  ReducedInstCount |= MergeInner(X, 0, NewMask0, CostKind);
-  ReducedInstCount |= MergeInner(Y, 0, NewMask1, CostKind);
-  ReducedInstCount |= MergeInner(Z, NumSrcElts, NewMask0, CostKind);
-  ReducedInstCount |= MergeInner(W, NumSrcElts, NewMask1, CostKind);
+  ReducedInstCount |= MergeInner(X, 0, NewMask0, CostKind, LHS);
+  ReducedInstCount |= MergeInner(Y, 0, NewMask1, CostKind, LHS);
+  ReducedInstCount |= MergeInner(Z, NumSrcElts, NewMask0, CostKind, RHS);
+  ReducedInstCount |= MergeInner(W, NumSrcElts, NewMask1, CostKind, RHS);
   bool SingleSrcBinOp = (X == Y) && (Z == W) && (NewMask0 == NewMask1);
   // SingleSrcBinOp only reduces instruction count if we also eliminate the
   // original binop(s). If binops have multiple uses, they won't be eliminated.
diff --git a/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll b/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll
index e1e95019680434..20960f23f1dbc9 100644
--- a/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/shuffle-of-binops.ll
@@ -249,6 +249,26 @@ define <4 x i32> @shuf_sdiv_v4i32_multiuse_both(<4 x i32> %x, <4 x i32> %y, <4 x
   ret <4 x i32> %r
 }
 
+define <4 x i32> @shuf_add_v4i32_inner_shuffles_multiuse_rhs(<4 x i32> %a, <4 x i32> %b, <4 x i32> %y) {
+; CHECK-LABEL: define <4 x i32> @shuf_add_v4i32_inner_shuffles_multiuse_rhs(
+; CHECK-SAME: <4 x i32> [[A:%.*]], <4 x i32> [[B:%.*]], <4 x i32> [[Y:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[A0:%.*]] = shufflevector <4 x i32> [[A]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[B0:%.*]] = shufflevector <4 x i32> [[B]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[L:%.*]] = add <4 x i32> [[A0]], [[Y]]
+; CHECK-NEXT:    [[R:%.*]] = add <4 x i32> [[B0]], [[Y]]
+; CHECK-NEXT:    call void @use(<4 x i32> [[R]])
+; CHECK-NEXT:    [[S:%.*]] = shufflevector <4 x i32> [[L]], <4 x i32> [[R]], <4 x i32> <i32 0, i32 1, i32 6, i32 7>
+; CHECK-NEXT:    ret <4 x i32> [[S]]
+;
+  %a0 = shufflevector <4 x i32> %a, <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+  %b0 = shufflevector <4 x i32> %b, <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+  %l = add <4 x i32> %a0, %y
+  %r = add <4 x i32> %b0, %y
+  call void @use(<4 x i32> %r)
+  %s = shufflevector <4 x i32> %l, <4 x i32> %r, <4 x i32> <i32 0, i32 1, i32 6, i32 7>
+  ret <4 x i32> %s
+}
+
 ; non-matching operands (not commutable)
 
 define <4 x float> @shuf_fdiv_v4f32_no_common_op(<4 x float> %x, <4 x float> %y, <4 x float> %z, <4 x float> %w) {

``````````

</details>


https://github.com/llvm/llvm-project/pull/228473


More information about the llvm-commits mailing list