[llvm] [SLP]Delete combined subnodes together with the trimmed combined root (PR #219180)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 04:16:34 PDT 2026


https://github.com/alexey-bataev created https://github.com/llvm/llvm-project/pull/219180

Skipping CombinedVectorize entries in subtree-cost aggregation also
dropped them from the ancestors' subtree node lists, so trimming a
combined root left its combined subnodes live but never vectorized.


>From 2f2adab1034b31d32f450e8e60e723c7fb504cff Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Thu, 27 Aug 2026 04:16:11 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
 =?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    |  7 ++--
 .../X86/trim-root-with-combined-node.ll       | 35 +++++++++++++++++++
 2 files changed, 38 insertions(+), 4 deletions(-)
 create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/trim-root-with-combined-node.ll

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index e5211a198acdd..358901da42266 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -19517,10 +19517,9 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals,
       };
   for (const std::unique_ptr<TreeEntry> &Ptr : VectorizableTree) {
     TreeEntry &TE = *Ptr;
-    // Combined subnodes are not costed on their own, only as a whole combined
-    // node, so only the root nodes are considered.
-    if (TE.State == TreeEntry::CombinedVectorize)
-      continue;
+    // Combined subnodes are not costed on their own (their cost is 0), but
+    // must be included into the ancestors' subtree node lists, so that they
+    // get deleted together with the trimmed combined root.
     InstructionCost C = NodesCosts.at(&TE);
     InstructionCost ExtractCost = ExtractCosts.lookup(&TE);
     std::get<0>(SubtreeCosts[TE.Idx]) += C + ExtractCost;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/trim-root-with-combined-node.ll b/llvm/test/Transforms/SLPVectorizer/X86/trim-root-with-combined-node.ll
new file mode 100644
index 0000000000000..c5281bfbff04e
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/trim-root-with-combined-node.ll
@@ -0,0 +1,35 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu | FileCheck %s
+
+define void @test(ptr %this, double %0) {
+; CHECK-LABEL: define void @test(
+; CHECK-SAME: ptr [[THIS:%.*]], double [[TMP0:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x double> poison, double [[TMP0]], i64 0
+; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP3:%.*]] = fmul <2 x double> [[TMP2]], zeroinitializer
+; CHECK-NEXT:    [[PIXEL00_LOC:%.*]] = getelementptr i8, ptr [[THIS]], i64 144
+; CHECK-NEXT:    [[MUL23_I:%.*]] = fmul contract double [[TMP0]], 0.000000e+00
+; CHECK-NEXT:    [[TMP4:%.*]] = fadd contract <2 x double> [[TMP3]], <double -0.000000e+00, double 0.000000e+00>
+; CHECK-NEXT:    [[TMP5:%.*]] = fsub <2 x double> zeroinitializer, [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = shufflevector <2 x double> [[TMP4]], <2 x double> poison, <2 x i32> <i32 1, i32 poison>
+; CHECK-NEXT:    [[TMP7:%.*]] = insertelement <2 x double> [[TMP6]], double [[MUL23_I]], i64 1
+; CHECK-NEXT:    [[TMP8:%.*]] = fsub <2 x double> [[TMP5]], [[TMP7]]
+; CHECK-NEXT:    store <2 x double> [[TMP8]], ptr [[PIXEL00_LOC]], align 8
+; CHECK-NEXT:    ret void
+;
+entry:
+  %mul.i.i141 = fmul double %0, 0.000000e+00
+  %sub.i148 = fsub double 0.000000e+00, %mul.i.i141
+  %mul12.i = fmul contract double %0, 0.000000e+00
+  %sub18.i = fadd contract double %mul12.i, 0.000000e+00
+  %sub.i165 = fsub double %sub.i148, %sub18.i
+  %pixel00_loc = getelementptr i8, ptr %this, i64 144
+  store double %sub.i165, ptr %pixel00_loc, align 8
+  %sub7.i151 = fsub double 0.000000e+00, %sub18.i
+  %mul23.i = fmul contract double %0, 0.000000e+00
+  %sub7.i168 = fsub double %sub7.i151, %mul23.i
+  %ref.tmp43.sroa.4.0.pixel00_loc.sroa_idx = getelementptr i8, ptr %this, i64 152
+  store double %sub7.i168, ptr %ref.tmp43.sroa.4.0.pixel00_loc.sroa_idx, align 8
+  ret void
+}



More information about the llvm-commits mailing list