[llvm] [SLP]Schedule instructions with calculated deps after trimming (PR #227083)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 28 11:50:40 PDT 2026
https://github.com/alexey-bataev created https://github.com/llvm/llvm-project/pull/227083
Cancelling the bundles of the trimmed nodes clears all the dependencies
in the scheduling region, but they are recalculated only for the
remaining bundles, the trimmed scalars and their users. The
instructions, whose dependencies were calculated earlier (e.g. for the
bundles, failed to be scheduled), are not scheduled anymore and are
moved above all the scheduled instructions, breaking the original order.
Recalculate the dependencies for all such instructions.
Fixes the regression, reported in
https://github.com/llvm/llvm-project/pull/226268#issuecomment-5872192154
>From 78dbf9ab22e2d4d4eebb8d3569ecf0933ccb9d46 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Mon, 28 Sep 2026 11:50:09 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 18 +++++++--
.../AArch64/lcssa-phi-extract-scale.ll | 40 +++++++++----------
.../trimmed-subtree-schedule-operands.ll | 8 ++--
3 files changed, 38 insertions(+), 28 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 198ed26bcc719..06595912f262a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -28059,7 +28059,7 @@ void BoUpSLP::scheduleBlock(const BoUpSLP &R, BlockScheduling *BS) {
return DeletedNodes.contains(TE) || TransformedToGatherNodes.contains(TE);
};
const size_t NumBundles = BS->ScheduledBundlesList.size();
- SmallPtrSet<const ScheduleData *, 16> TrimmedSDs;
+ SmallPtrSet<const ScheduleData *, 16> ClearedSDs;
erase_if(BS->ScheduledBundlesList,
[&](const std::unique_ptr<ScheduleBundle> &Bundle) {
const TreeEntry *TE = Bundle->getTreeEntry();
@@ -28069,12 +28069,21 @@ void BoUpSLP::scheduleBlock(const BoUpSLP &R, BlockScheduling *BS) {
<< TE->Idx << " bundle " << *Bundle << "\n");
for (ScheduleEntity *SE : Bundle->getBundle())
if (auto *SD = dyn_cast<ScheduleData>(SE))
- TrimmedSDs.insert(SD);
+ ClearedSDs.insert(SD);
BS->cancelScheduling(*Bundle);
return true;
});
- if (BS->ScheduledBundlesList.size() != NumBundles)
+ if (BS->ScheduledBundlesList.size() != NumBundles) {
+ // Need to schedule all the instructions with the calculated dependencies,
+ // not only the users of the remaining bundles, otherwise they are moved
+ // above all the scheduled instructions.
+ for (Instruction &I : make_range(BS->ScheduleStart->getIterator(),
+ BS->ScheduleEnd->getIterator()))
+ if (ScheduleData *SD = BS->getScheduleData(&I);
+ SD && SD->hasValidDependencies())
+ ClearedSDs.insert(SD);
BS->clearDependencies();
+ }
// A key point - if we got here, pre-scheduling was able to find a valid
// scheduling of the sub-graph of the scheduling window which consists
@@ -28163,7 +28172,8 @@ void BoUpSLP::scheduleBlock(const BoUpSLP &R, BlockScheduling *BS) {
doesNotNeedToBeScheduled(I)) &&
"scheduler and vectorizer bundle mismatch");
SD->setSchedulingPriority(Idx++);
- if (TrimmedSDs.contains(SD) || !CopyableData.empty() ||
+ if ((!SD->hasValidDependencies() && ClearedSDs.contains(SD)) ||
+ !CopyableData.empty() ||
any_of(R.ValueToGatherNodes.lookup(I), [&](const TreeEntry *TE) {
assert(TE->isGather() && "expected gather node");
return TE->hasState() && TE->hasCopyableElements() &&
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/lcssa-phi-extract-scale.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/lcssa-phi-extract-scale.ll
index 6db60c49764dc..a372875f3b577 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/lcssa-phi-extract-scale.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/lcssa-phi-extract-scale.ll
@@ -51,28 +51,28 @@ define double @test(ptr %obj, ptr %arr, i32 %n) {
; CHECK-NEXT: [[V40:%.*]] = sub i32 [[V36]], [[V39]]
; CHECK-NEXT: [[V41:%.*]] = add nsw i32 [[V40]], 2147483647
; CHECK-NEXT: [[V42:%.*]] = icmp slt i32 [[V40]], 0
-; CHECK-NEXT: [[SPEC_SELECT_I13:%.*]] = select i1 [[V42]], i32 [[V41]], i32 [[V40]]
-; CHECK-NEXT: store i32 [[SPEC_SELECT_I13]], ptr [[V38]], align 4
; CHECK-NEXT: [[V45:%.*]] = zext nneg i32 [[TMP10]] to i64
; CHECK-NEXT: [[V46:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[ARR]], i64 [[V45]]
-; CHECK-NEXT: [[V47:%.*]] = load i32, ptr [[V46]], align 4
; CHECK-NEXT: [[V48:%.*]] = zext nneg i32 [[TMP11]] to i64
; CHECK-NEXT: [[V49:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[ARR]], i64 [[V48]]
+; CHECK-NEXT: [[DOTNOT3_I:%.*]] = icmp eq i32 [[TMP10]], 0
+; CHECK-NEXT: [[V54:%.*]] = add nsw i32 [[TMP10]], -1
+; CHECK-NEXT: [[DOTSINK_I]] = select i1 [[DOTNOT3_I]], i32 16, i32 [[V54]]
+; CHECK-NEXT: [[DOTNOT4_I:%.*]] = icmp eq i32 [[TMP11]], 0
+; CHECK-NEXT: [[V55:%.*]] = add nsw i32 [[TMP11]], -1
+; CHECK-NEXT: [[DOTSINK6_I]] = select i1 [[DOTNOT4_I]], i32 16, i32 [[V55]]
+; CHECK-NEXT: [[SPEC_SELECT_I13:%.*]] = select i1 [[V42]], i32 [[V41]], i32 [[V40]]
+; CHECK-NEXT: store i32 [[SPEC_SELECT_I13]], ptr [[V38]], align 4
+; CHECK-NEXT: [[V43:%.*]] = sitofp i32 [[SPEC_SELECT_I13]] to double
+; CHECK-NEXT: [[V47:%.*]] = load i32, ptr [[V46]], align 4
; CHECK-NEXT: [[V50:%.*]] = load i32, ptr [[V49]], align 4
; CHECK-NEXT: [[V51:%.*]] = sub i32 [[V47]], [[V50]]
; CHECK-NEXT: [[V52:%.*]] = add nsw i32 [[V51]], 2147483647
; CHECK-NEXT: [[V53:%.*]] = icmp slt i32 [[V51]], 0
; CHECK-NEXT: [[SPEC_SELECT_I:%.*]] = select i1 [[V53]], i32 [[V52]], i32 [[V51]]
; CHECK-NEXT: store i32 [[SPEC_SELECT_I]], ptr [[V49]], align 4
-; CHECK-NEXT: [[DOTNOT3_I:%.*]] = icmp eq i32 [[TMP10]], 0
-; CHECK-NEXT: [[V54:%.*]] = add nsw i32 [[TMP10]], -1
-; CHECK-NEXT: [[DOTSINK_I]] = select i1 [[DOTNOT3_I]], i32 16, i32 [[V54]]
; CHECK-NEXT: store i32 [[DOTSINK_I]], ptr [[POS1]], align 4
-; CHECK-NEXT: [[DOTNOT4_I:%.*]] = icmp eq i32 [[TMP11]], 0
-; CHECK-NEXT: [[V55:%.*]] = add nsw i32 [[TMP11]], -1
-; CHECK-NEXT: [[DOTSINK6_I]] = select i1 [[DOTNOT4_I]], i32 16, i32 [[V55]]
; CHECK-NEXT: store i32 [[DOTSINK6_I]], ptr [[POS2]], align 8
-; CHECK-NEXT: [[V43:%.*]] = sitofp i32 [[SPEC_SELECT_I13]] to double
; CHECK-NEXT: [[V56:%.*]] = sitofp i32 [[SPEC_SELECT_I]] to double
; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x double> poison, double [[V43]], i64 0
; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x double> [[TMP0]], double [[V56]], i64 1
@@ -282,28 +282,28 @@ define double @test_inloop_first(ptr %obj, ptr %arr, i32 %n) {
; CHECK-NEXT: [[V40:%.*]] = sub i32 [[V36]], [[V39]]
; CHECK-NEXT: [[V41:%.*]] = add nsw i32 [[V40]], 2147483647
; CHECK-NEXT: [[V42:%.*]] = icmp slt i32 [[V40]], 0
-; CHECK-NEXT: [[SPEC_SELECT_I13:%.*]] = select i1 [[V42]], i32 [[V41]], i32 [[V40]]
-; CHECK-NEXT: store i32 [[SPEC_SELECT_I13]], ptr [[V38]], align 4
; CHECK-NEXT: [[V45:%.*]] = zext nneg i32 [[TMP11]] to i64
; CHECK-NEXT: [[V46:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[ARR]], i64 [[V45]]
-; CHECK-NEXT: [[V47:%.*]] = load i32, ptr [[V46]], align 4
; CHECK-NEXT: [[V48:%.*]] = zext nneg i32 [[TMP12]] to i64
; CHECK-NEXT: [[V49:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[ARR]], i64 [[V48]]
+; CHECK-NEXT: [[DOTNOT3_I:%.*]] = icmp eq i32 [[TMP11]], 0
+; CHECK-NEXT: [[V54:%.*]] = add nsw i32 [[TMP11]], -1
+; CHECK-NEXT: [[DOTSINK_I]] = select i1 [[DOTNOT3_I]], i32 16, i32 [[V54]]
+; CHECK-NEXT: [[DOTNOT4_I:%.*]] = icmp eq i32 [[TMP12]], 0
+; CHECK-NEXT: [[V55:%.*]] = add nsw i32 [[TMP12]], -1
+; CHECK-NEXT: [[DOTSINK6_I]] = select i1 [[DOTNOT4_I]], i32 16, i32 [[V55]]
+; CHECK-NEXT: [[SPEC_SELECT_I13:%.*]] = select i1 [[V42]], i32 [[V41]], i32 [[V40]]
+; CHECK-NEXT: store i32 [[SPEC_SELECT_I13]], ptr [[V38]], align 4
+; CHECK-NEXT: [[V43:%.*]] = sitofp i32 [[SPEC_SELECT_I13]] to double
+; CHECK-NEXT: [[V47:%.*]] = load i32, ptr [[V46]], align 4
; CHECK-NEXT: [[V50:%.*]] = load i32, ptr [[V49]], align 4
; CHECK-NEXT: [[V51:%.*]] = sub i32 [[V47]], [[V50]]
; CHECK-NEXT: [[V52:%.*]] = add nsw i32 [[V51]], 2147483647
; CHECK-NEXT: [[V53:%.*]] = icmp slt i32 [[V51]], 0
; CHECK-NEXT: [[SPEC_SELECT_I:%.*]] = select i1 [[V53]], i32 [[V52]], i32 [[V51]]
; CHECK-NEXT: store i32 [[SPEC_SELECT_I]], ptr [[V49]], align 4
-; CHECK-NEXT: [[DOTNOT3_I:%.*]] = icmp eq i32 [[TMP11]], 0
-; CHECK-NEXT: [[V54:%.*]] = add nsw i32 [[TMP11]], -1
-; CHECK-NEXT: [[DOTSINK_I]] = select i1 [[DOTNOT3_I]], i32 16, i32 [[V54]]
; CHECK-NEXT: store i32 [[DOTSINK_I]], ptr [[POS1]], align 4
-; CHECK-NEXT: [[DOTNOT4_I:%.*]] = icmp eq i32 [[TMP12]], 0
-; CHECK-NEXT: [[V55:%.*]] = add nsw i32 [[TMP12]], -1
-; CHECK-NEXT: [[DOTSINK6_I]] = select i1 [[DOTNOT4_I]], i32 16, i32 [[V55]]
; CHECK-NEXT: store i32 [[DOTSINK6_I]], ptr [[POS2]], align 8
-; CHECK-NEXT: [[V43:%.*]] = sitofp i32 [[SPEC_SELECT_I13]] to double
; CHECK-NEXT: [[V56:%.*]] = sitofp i32 [[SPEC_SELECT_I]] to double
; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x double> poison, double [[V43]], i64 0
; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x double> [[TMP0]], double [[V56]], i64 1
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/trimmed-subtree-schedule-operands.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/trimmed-subtree-schedule-operands.ll
index 93def9c5db56d..49f10eb374186 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/trimmed-subtree-schedule-operands.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/trimmed-subtree-schedule-operands.ll
@@ -18,11 +18,7 @@ define void @test(ptr %srcGrid, ptr %dstGrid, i64 %iv) {
; CHECK-NEXT: [[ARRAYIDX144:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 56
; CHECK-NEXT: [[ARRAYIDX148:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 64
; CHECK-NEXT: [[ARRAYIDX184:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 136
-; CHECK-NEXT: [[TMP3:%.*]] = load double, ptr [[ARRAYIDX184]], align 8
; CHECK-NEXT: [[ARRAYIDX188:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 144
-; CHECK-NEXT: [[TMP4:%.*]] = load double, ptr [[ARRAYIDX188]], align 8
-; CHECK-NEXT: [[TMP7:%.*]] = fadd fast double 0.000000e+00, [[TMP4]]
-; CHECK-NEXT: [[MOMENTUM_Z:%.*]] = fsub fast double 0.000000e+00, [[TMP7]]
; CHECK-NEXT: [[ARRAYIDX675:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[DSTGRID]], i64 [[IV]]
; CHECK-NEXT: [[ARRAYIDX787:%.*]] = getelementptr inbounds nuw i8, ptr [[ARRAYIDX675]], i64 32216
; CHECK-NEXT: [[TMP1:%.*]] = load double, ptr [[ARRAYIDX148]], align 8
@@ -30,12 +26,16 @@ define void @test(ptr %srcGrid, ptr %dstGrid, i64 %iv) {
; CHECK-NEXT: [[ADD157:%.*]] = fadd fast double [[TMP1]], 0.000000e+00
; CHECK-NEXT: [[ADD177:%.*]] = fadd fast double [[ADD157]], 0.000000e+00
; CHECK-NEXT: [[ADD181:%.*]] = fadd fast double [[ADD177]], 0.000000e+00
+; CHECK-NEXT: [[TMP3:%.*]] = load double, ptr [[ARRAYIDX184]], align 8
; CHECK-NEXT: [[ADD185:%.*]] = fadd fast double [[ADD181]], 0.000000e+00
+; CHECK-NEXT: [[TMP4:%.*]] = load double, ptr [[ARRAYIDX188]], align 8
; CHECK-NEXT: [[RHO:%.*]] = fadd fast double [[ADD185]], 1.000000e+00
; CHECK-NEXT: [[TMP5:%.*]] = fadd fast double [[TMP3]], [[TMP4]]
; CHECK-NEXT: [[SUB227:%.*]] = fsub fast double 0.000000e+00, [[TMP5]]
; CHECK-NEXT: [[TMP6:%.*]] = fadd fast double 0.000000e+00, 0.000000e+00
; CHECK-NEXT: [[SUB266:%.*]] = fsub fast double 0.000000e+00, [[TMP6]]
+; CHECK-NEXT: [[TMP7:%.*]] = fadd fast double 0.000000e+00, [[TMP4]]
+; CHECK-NEXT: [[MOMENTUM_Z:%.*]] = fsub fast double 0.000000e+00, [[TMP7]]
; CHECK-NEXT: [[DIV:%.*]] = fdiv fast double [[SUB227]], [[RHO]]
; CHECK-NEXT: [[DIV306:%.*]] = fdiv fast double [[SUB266]], 1.000000e+00
; CHECK-NEXT: [[DIV307:%.*]] = fdiv fast double [[MOMENTUM_Z]], [[RHO]]
More information about the llvm-commits
mailing list