[llvm] [SLP]Do not reuse transformed nodes in gathers emitted before their user (PR #226687)

via llvm-commits llvm-commits at lists.llvm.org
Sat Sep 26 05:48:43 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-vectorizers

Author: Alexey Bataev (alexey-bataev)

<details>
<summary>Changes</summary>

Gathers of the users with all scalars used outside the block are emitted
before the user, while the transformed nodes are emitted with the user.
Reusing such a transformed node postponed the gather and moved only the
last instruction of the transformed node buildvector, breaking dominance.

Fixes #<!-- -->226674


---
Full diff: https://github.com/llvm/llvm-project/pull/226687.diff


2 Files Affected:

- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp (+4-2) 
- (added) llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll (+69) 


``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 9a1b310fd327ff..f76a7544184222 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -21027,11 +21027,13 @@ BoUpSLP::isGatherShuffledSingleRegisterEntry(
         GatherNodes.push_back(E);
       }
     } else if (const TreeEntry *E = getSameValuesTreeEntry(V, TE->Scalars);
-               E && TransformedToGatherNodes.contains(E) && E->UserTreeIndex &&
+               E && !TEUserNeedsEmitFirst &&
+               TransformedToGatherNodes.contains(E) && E->UserTreeIndex &&
                E->UserTreeIndex.UserTE == TE->UserTreeIndex.UserTE &&
                !E->UserTreeIndex.UserTE->isGather()) {
       // Regular gathers reuse only perfectly matched transformed nodes of the
-      // same user.
+      // same user. Gathers, emitted before their user, cannot reuse transformed
+      // nodes, which are emitted with the user.
       GatherNodes.push_back(E);
     }
     for (const TreeEntry *TEPtr : GatherNodes) {
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll b/llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll
new file mode 100644
index 00000000000000..f86a3fc7abc7c3
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/trimmed-node-matching-gather-emitted-first.ll
@@ -0,0 +1,69 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s
+
+define double @test(double %x, i1 %c) {
+; CHECK-LABEL: define double @test(
+; CHECK-SAME: double [[X:%.*]], i1 [[C:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br i1 [[C]], label %[[BB:.*]], label %[[EXIT:.*]]
+; CHECK:       [[BB]]:
+; CHECK-NEXT:    [[ADD:%.*]] = fadd double [[X]], [[X]]
+; CHECK-NEXT:    [[SUB:%.*]] = fsub double 0.000000e+00, [[X]]
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x double> poison, double [[ADD]], i64 0
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x double> [[TMP0]], double [[SUB]], i64 1
+; CHECK-NEXT:    [[TMP2:%.*]] = call <2 x double> @llvm.fmuladd.v2f64(<2 x double> [[TMP1]], <2 x double> [[TMP1]], <2 x double> zeroinitializer)
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[TMP3:%.*]] = phi <2 x double> [ [[TMP2]], %[[BB]] ], [ zeroinitializer, %[[ENTRY]] ]
+; CHECK-NEXT:    [[A:%.*]] = fadd double [[X]], 0.000000e+00
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <2 x double> [[TMP3]], i64 1
+; CHECK-NEXT:    [[M:%.*]] = fmul double [[A]], [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <2 x double> [[TMP3]], i64 0
+; CHECK-NEXT:    [[R:%.*]] = call double @llvm.fmuladd.f64(double [[TMP5]], double 0.000000e+00, double [[M]])
+; CHECK-NEXT:    ret double [[R]]
+;
+entry:
+  br i1 %c, label %bb, label %exit
+
+bb:
+  %add = fadd double %x, %x
+  %f0 = call double @llvm.fmuladd.f64(double %add, double %add, double 0.000000e+00)
+  %sub = fsub double 0.000000e+00, %x
+  %f1 = call double @llvm.fmuladd.f64(double %sub, double %sub, double 0.000000e+00)
+  br label %exit
+
+exit:
+  %p0 = phi double [ %f0, %bb ], [ 0.000000e+00, %entry ]
+  %p1 = phi double [ %f1, %bb ], [ 0.000000e+00, %entry ]
+  %a = fadd double %x, 0.000000e+00
+  %m = fmul double %a, %p1
+  %r = call double @llvm.fmuladd.f64(double %p0, double 0.000000e+00, double %m)
+  ret double %r
+}
+
+define void @test_loop(float %x) {
+; CHECK-LABEL: define void @test_loop(
+; CHECK-SAME: float [[X:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = phi <2 x float> [ [[TMP3:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
+; CHECK-NEXT:    [[ADD:%.*]] = fadd float [[X]], [[X]]
+; CHECK-NEXT:    [[SUB:%.*]] = fsub float 0.000000e+00, [[X]]
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x float> poison, float [[ADD]], i64 0
+; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x float> [[TMP1]], float [[SUB]], i64 1
+; CHECK-NEXT:    [[TMP3]] = call <2 x float> @llvm.fmuladd.v2f32(<2 x float> [[TMP2]], <2 x float> [[TMP2]], <2 x float> [[TMP0]])
+; CHECK-NEXT:    br label %[[LOOP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %p0 = phi float [ %f0, %loop ], [ 0.000000e+00, %entry ]
+  %p1 = phi float [ %f1, %loop ], [ 0.000000e+00, %entry ]
+  %add = fadd float %x, %x
+  %f0 = call float @llvm.fmuladd.f32(float %add, float %add, float %p0)
+  %sub = fsub float 0.000000e+00, %x
+  %f1 = call float @llvm.fmuladd.f32(float %sub, float %sub, float %p1)
+  br label %loop
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/226687


More information about the llvm-commits mailing list