[llvm] [SLP]Do not reuse transformed nodes in gathers emitted before their user (PR #226687)
via llvm-commits
llvm-commits at lists.llvm.org
Sat Sep 26 05:48:43 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-vectorizers
Author: Alexey Bataev (alexey-bataev)
<details>
<summary>Changes</summary>
Gathers of the users with all scalars used outside the block are emitted
before the user, while the transformed nodes are emitted with the user.
Reusing such a transformed node postponed the gather and moved only the
last instruction of the transformed node buildvector, breaking dominance.
Fixes #<!-- -->226674
---
Full diff: https://github.com/llvm/llvm-project/pull/226687.diff
2 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp (+4-2)
- (added) llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll (+69)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 9a1b310fd327ff..f76a7544184222 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -21027,11 +21027,13 @@ BoUpSLP::isGatherShuffledSingleRegisterEntry(
GatherNodes.push_back(E);
}
} else if (const TreeEntry *E = getSameValuesTreeEntry(V, TE->Scalars);
- E && TransformedToGatherNodes.contains(E) && E->UserTreeIndex &&
+ E && !TEUserNeedsEmitFirst &&
+ TransformedToGatherNodes.contains(E) && E->UserTreeIndex &&
E->UserTreeIndex.UserTE == TE->UserTreeIndex.UserTE &&
!E->UserTreeIndex.UserTE->isGather()) {
// Regular gathers reuse only perfectly matched transformed nodes of the
- // same user.
+ // same user. Gathers, emitted before their user, cannot reuse transformed
+ // nodes, which are emitted with the user.
GatherNodes.push_back(E);
}
for (const TreeEntry *TEPtr : GatherNodes) {
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll b/llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll
new file mode 100644
index 00000000000000..f86a3fc7abc7c3
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll
@@ -0,0 +1,69 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s
+
+define double @test(double %x, i1 %c) {
+; CHECK-LABEL: define double @test(
+; CHECK-SAME: double [[X:%.*]], i1 [[C:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br i1 [[C]], label %[[BB:.*]], label %[[EXIT:.*]]
+; CHECK: [[BB]]:
+; CHECK-NEXT: [[ADD:%.*]] = fadd double [[X]], [[X]]
+; CHECK-NEXT: [[SUB:%.*]] = fsub double 0.000000e+00, [[X]]
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x double> poison, double [[ADD]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x double> [[TMP0]], double [[SUB]], i64 1
+; CHECK-NEXT: [[TMP2:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> [[TMP1]], <2 x double> [[TMP1]], <2 x double> zeroinitializer)
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[TMP3:%.*]] = phi <2 x double> [ [[TMP2]], %[[BB]] ], [ zeroinitializer, %[[ENTRY]] ]
+; CHECK-NEXT: [[A:%.*]] = fadd double [[X]], 0.000000e+00
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x double> [[TMP3]], i64 1
+; CHECK-NEXT: [[M:%.*]] = fmul double [[A]], [[TMP4]]
+; CHECK-NEXT: [[TMP5:%.*]] = extractelement <2 x double> [[TMP3]], i64 0
+; CHECK-NEXT: [[R:%.*]] = call double @llvm.fmuladd.f64(double [[TMP5]], double 0.000000e+00, double [[M]])
+; CHECK-NEXT: ret double [[R]]
+;
+entry:
+ br i1 %c, label %bb, label %exit
+
+bb:
+ %add = fadd double %x, %x
+ %f0 = call double @llvm.fmuladd.f64(double %add, double %add, double 0.000000e+00)
+ %sub = fsub double 0.000000e+00, %x
+ %f1 = call double @llvm.fmuladd.f64(double %sub, double %sub, double 0.000000e+00)
+ br label %exit
+
+exit:
+ %p0 = phi double [ %f0, %bb ], [ 0.000000e+00, %entry ]
+ %p1 = phi double [ %f1, %bb ], [ 0.000000e+00, %entry ]
+ %a = fadd double %x, 0.000000e+00
+ %m = fmul double %a, %p1
+ %r = call double @llvm.fmuladd.f64(double %p0, double 0.000000e+00, double %m)
+ ret double %r
+}
+
+define void @test_loop(float %x) {
+; CHECK-LABEL: define void @test_loop(
+; CHECK-SAME: float [[X:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[TMP0:%.*]] = phi <2 x float> [ [[TMP3:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
+; CHECK-NEXT: [[ADD:%.*]] = fadd float [[X]], [[X]]
+; CHECK-NEXT: [[SUB:%.*]] = fsub float 0.000000e+00, [[X]]
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x float> poison, float [[ADD]], i64 0
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <2 x float> [[TMP1]], float [[SUB]], i64 1
+; CHECK-NEXT: [[TMP3]] = call <2 x float> @llvm.fmuladd.v2f32(<2 x float> [[TMP2]], <2 x float> [[TMP2]], <2 x float> [[TMP0]])
+; CHECK-NEXT: br label %[[LOOP]]
+;
+entry:
+ br label %loop
+
+loop:
+ %p0 = phi float [ %f0, %loop ], [ 0.000000e+00, %entry ]
+ %p1 = phi float [ %f1, %loop ], [ 0.000000e+00, %entry ]
+ %add = fadd float %x, %x
+ %f0 = call float @llvm.fmuladd.f32(float %add, float %add, float %p0)
+ %sub = fsub float 0.000000e+00, %x
+ %f1 = call float @llvm.fmuladd.f32(float %sub, float %sub, float %p1)
+ br label %loop
+}
``````````
</details>
https://github.com/llvm/llvm-project/pull/226687
More information about the llvm-commits
mailing list