[llvm] 8b96ddb - [SLP]Do not reuse transformed nodes in gathers emitted before their user
via llvm-commits
llvm-commits at lists.llvm.org
Sat Sep 26 05:48:31 PDT 2026
Author: Alexey Bataev
Date: 2026-09-26T08:48:23-04:00
New Revision: 8b96ddbf9cb3437ee4e47a9f60dc983a1f2be8d5
URL: https://github.com/llvm/llvm-project/commit/8b96ddbf9cb3437ee4e47a9f60dc983a1f2be8d5
DIFF: https://github.com/llvm/llvm-project/commit/8b96ddbf9cb3437ee4e47a9f60dc983a1f2be8d5.diff
LOG: [SLP]Do not reuse transformed nodes in gathers emitted before their user
Gathers of the users with all scalars used outside the block are emitted
before the user, while the transformed nodes are emitted with the user.
Reusing such a transformed node postponed the gather and moved only the
last instruction of the transformed node buildvector, breaking dominance.
Fixes #226674
Reviewers:
Pull Request: https://github.com/llvm/llvm-project/pull/226687
Added:
llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll
Modified:
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 9a1b310fd327f..f76a754418422 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -21027,11 +21027,13 @@ BoUpSLP::isGatherShuffledSingleRegisterEntry(
GatherNodes.push_back(E);
}
} else if (const TreeEntry *E = getSameValuesTreeEntry(V, TE->Scalars);
- E && TransformedToGatherNodes.contains(E) && E->UserTreeIndex &&
+ E && !TEUserNeedsEmitFirst &&
+ TransformedToGatherNodes.contains(E) && E->UserTreeIndex &&
E->UserTreeIndex.UserTE == TE->UserTreeIndex.UserTE &&
!E->UserTreeIndex.UserTE->isGather()) {
// Regular gathers reuse only perfectly matched transformed nodes of the
- // same user.
+ // same user. Gathers, emitted before their user, cannot reuse transformed
+ // nodes, which are emitted with the user.
GatherNodes.push_back(E);
}
for (const TreeEntry *TEPtr : GatherNodes) {
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll b/llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll
new file mode 100644
index 0000000000000..f86a3fc7abc7c
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll
@@ -0,0 +1,69 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s
+
+define double @test(double %x, i1 %c) {
+; CHECK-LABEL: define double @test(
+; CHECK-SAME: double [[X:%.*]], i1 [[C:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br i1 [[C]], label %[[BB:.*]], label %[[EXIT:.*]]
+; CHECK: [[BB]]:
+; CHECK-NEXT: [[ADD:%.*]] = fadd double [[X]], [[X]]
+; CHECK-NEXT: [[SUB:%.*]] = fsub double 0.000000e+00, [[X]]
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x double> poison, double [[ADD]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x double> [[TMP0]], double [[SUB]], i64 1
+; CHECK-NEXT: [[TMP2:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> [[TMP1]], <2 x double> [[TMP1]], <2 x double> zeroinitializer)
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[TMP3:%.*]] = phi <2 x double> [ [[TMP2]], %[[BB]] ], [ zeroinitializer, %[[ENTRY]] ]
+; CHECK-NEXT: [[A:%.*]] = fadd double [[X]], 0.000000e+00
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x double> [[TMP3]], i64 1
+; CHECK-NEXT: [[M:%.*]] = fmul double [[A]], [[TMP4]]
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x double> [[TMP3]], i64 0
+; CHECK-NEXT: [[R:%.*]] = call double @llvm.fmuladd.f64(double [[TMP5]], double 0.000000e+00, double [[M]])
+; CHECK-NEXT: ret double [[R]]
+;
+entry:
+ br i1 %c, label %bb, label %exit
+
+bb:
+ %add = fadd double %x, %x
+ %f0 = call double @llvm.fmuladd.f64(double %add, double %add, double 0.000000e+00)
+ %sub = fsub double 0.000000e+00, %x
+ %f1 = call double @llvm.fmuladd.f64(double %sub, double %sub, double 0.000000e+00)
+ br label %exit
+
+exit:
+ %p0 = phi double [ %f0, %bb ], [ 0.000000e+00, %entry ]
+ %p1 = phi double [ %f1, %bb ], [ 0.000000e+00, %entry ]
+ %a = fadd double %x, 0.000000e+00
+ %m = fmul double %a, %p1
+ %r = call double @llvm.fmuladd.f64(double %p0, double 0.000000e+00, double %m)
+ ret double %r
+}
+
+define void @test_loop(float %x) {
+; CHECK-LABEL: define void @test_loop(
+; CHECK-SAME: float [[X:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[TMP0:%.*]] = phi <2 x float> [ [[TMP3:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
+; CHECK-NEXT: [[ADD:%.*]] = fadd float [[X]], [[X]]
+; CHECK-NEXT: [[SUB:%.*]] = fsub float 0.000000e+00, [[X]]
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x float> poison, float [[ADD]], i64 0
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <2 x float> [[TMP1]], float [[SUB]], i64 1
+; CHECK-NEXT: [[TMP3]] = call <2 x float> @llvm.fmuladd.v2f32(<2 x float> [[TMP2]], <2 x float> [[TMP2]], <2 x float> [[TMP0]])
+; CHECK-NEXT: br label %[[LOOP]]
+;
+entry:
+ br label %loop
+
+loop:
+ %p0 = phi float [ %f0, %loop ], [ 0.000000e+00, %entry ]
+ %p1 = phi float [ %f1, %loop ], [ 0.000000e+00, %entry ]
+ %add = fadd float %x, %x
+ %f0 = call float @llvm.fmuladd.f32(float %add, float %add, float %p0)
+ %sub = fsub float 0.000000e+00, %x
+ %f1 = call float @llvm.fmuladd.f32(float %sub, float %sub, float %p1)
+ br label %loop
+}
More information about the llvm-commits
mailing list