[llvm] [SLP]Bail out on non-schedulable expanded binop with stale operand deps (PR #196449)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Thu May 7 16:55:34 PDT 2026
https://github.com/alexey-bataev created https://github.com/llvm/llvm-project/pull/196449
In tryScheduleBundle's DoesNotRequireScheduling path, an expanded binop
(shl X, 1 modeled as add X, X) doubles the dependency count of the
duplicated operand. If the operand has a
single IR use yet its ScheduleData already has Dependencies populated
by an earlier calculation that did not see the expanded duplicate use,
double decrement still exceeds calculateDependencies' single increment
and UnscheduledDeps goes negative.
Fixes #196281.
>From 0936156bb05e68d0ed5b8d930d2a964b65ed0fa8 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Thu, 7 May 2026 16:55:17 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 16 ++++--
...expanded-binop-doesnotneedschedule-user.ll | 6 +--
.../X86/expanded-operand-already-scheduled.ll | 50 +++++++++++++++++++
3 files changed, 64 insertions(+), 8 deletions(-)
create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/expanded-operand-already-scheduled.ll
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 2a9d98cedc1eb..9cfeae0b8bee7 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -24608,16 +24608,22 @@ BoUpSLP::BlockScheduling::tryScheduleBundle(ArrayRef<Value *> VL, BoUpSLP *SLP,
// dependency count of the duplicated operand. This non-schedulable
// path does not clear operand ScheduleData dependencies the way the
// regular scheduling path does (see CheckIfNeedToClearDeps below), so
- // when that operand has more uses than this bundle member, the
- // schedule's decrement count exceeds calculateDependencies' increment
- // count and UnscheduledDeps goes negative. Bail out instead of
- // producing an inconsistent schedule.
+ // when that operand has more uses than this bundle member, or its
+ // ScheduleData already has computed dependencies that did not see
+ // the expanded duplicate use, the schedule's decrement count exceeds
+ // calculateDependencies' increment count and UnscheduledDeps goes
+ // negative. Bail out instead of producing an inconsistent schedule.
if (S.isExpandedBinOp(I) &&
any_of(enumerate(I->operands()), [&](const auto &P) {
if (S.isExpandedOperand(I, P.index()))
return false;
auto *OpI = dyn_cast<Instruction>(P.value());
- return OpI && !OpI->hasOneUse();
+ if (!OpI)
+ return false;
+ if (!OpI->hasOneUse())
+ return true;
+ ScheduleData *SD = getScheduleData(OpI);
+ return SD && SD->hasValidDependencies();
}))
return std::nullopt;
if (EI && EI.UserTE->State == TreeEntry::Vectorize &&
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/expanded-binop-doesnotneedschedule-user.ll b/llvm/test/Transforms/SLPVectorizer/X86/expanded-binop-doesnotneedschedule-user.ll
index 3f233d26e462b..fa0f927e36395 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/expanded-binop-doesnotneedschedule-user.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/expanded-binop-doesnotneedschedule-user.ll
@@ -8,13 +8,13 @@ define void @test() {
; CHECK-NEXT: br label %[[BB1:.*]]
; CHECK: [[BB1]]:
; CHECK-NEXT: [[TMP0:%.*]] = phi <2 x i32> [ [[TMP5:%.*]], %[[BB1]] ], [ zeroinitializer, %[[BB]] ]
-; CHECK-NEXT: [[TMP1:%.*]] = phi <2 x i32> [ [[TMP7:%.*]], %[[BB1]] ], [ zeroinitializer, %[[BB]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = phi <2 x i32> [ [[TMP6:%.*]], %[[BB1]] ], [ zeroinitializer, %[[BB]] ]
; CHECK-NEXT: [[TMP2:%.*]] = mul <2 x i32> [[TMP1]], <i32 0, i32 1>
; CHECK-NEXT: [[TMP3:%.*]] = shl <2 x i32> [[TMP2]], zeroinitializer
; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <2 x i32> [[TMP3]], <2 x i32> <i32 1, i32 poison>, <2 x i32> <i32 2, i32 1>
; CHECK-NEXT: [[TMP5]] = add <2 x i32> [[TMP4]], [[TMP3]]
-; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP4]], <2 x i32> <i32 0, i32 poison>, <2 x i32> <i32 2, i32 1>
-; CHECK-NEXT: [[TMP7]] = add <2 x i32> [[TMP6]], [[TMP4]]
+; CHECK-NEXT: [[ADD6:%.*]] = add i32 0, 1
+; CHECK-NEXT: [[TMP6]] = insertelement <2 x i32> [[TMP5]], i32 [[ADD6]], i32 0
; CHECK-NEXT: br label %[[BB1]]
;
bb:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/expanded-operand-already-scheduled.ll b/llvm/test/Transforms/SLPVectorizer/X86/expanded-operand-already-scheduled.ll
new file mode 100644
index 0000000000000..2079aac55de9c
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/expanded-operand-already-scheduled.ll
@@ -0,0 +1,50 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt --passes=slp-vectorizer -S -slp-threshold=-99999 -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s
+
+define i32 @test() {
+; CHECK-LABEL: define i32 @test() {
+; CHECK-NEXT: [[Q_EXIT8:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = add <2 x i32> zeroinitializer, <i32 1, i32 0>
+; CHECK-NEXT: br label %[[BB1:.*]]
+; CHECK: [[BB1]]:
+; CHECK-NEXT: [[TMP2:%.*]] = trunc <2 x i32> [[TMP0]] to <2 x i1>
+; CHECK-NEXT: [[TMP3:%.*]] = mul <2 x i1> [[TMP2]], zeroinitializer
+; CHECK-NEXT: [[TMP4:%.*]] = or <2 x i1> zeroinitializer, [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = lshr <2 x i1> [[TMP4]], zeroinitializer
+; CHECK-NEXT: [[TMP6:%.*]] = and <2 x i1> [[TMP4]], zeroinitializer
+; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <2 x i1> [[TMP5]], <2 x i1> [[TMP6]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT: [[TMP8:%.*]] = extractelement <2 x i1> [[TMP7]], i32 0
+; CHECK-NEXT: [[TMP9:%.*]] = zext i1 [[TMP8]] to i32
+; CHECK-NEXT: [[TMP10:%.*]] = extractelement <2 x i1> [[TMP7]], i32 1
+; CHECK-NEXT: [[TMP11:%.*]] = zext i1 [[TMP10]] to i32
+; CHECK-NEXT: [[TMP12:%.*]] = or i32 [[TMP9]], [[TMP11]]
+; CHECK-NEXT: ret i32 [[TMP12]]
+;
+q.exit8:
+ %0 = add i32 0, 0
+ %1 = or i32 %0, 0
+ %2 = shl i32 %1, 1
+ %3 = lshr i32 0, 0
+ %4 = add i32 %3, 1
+ %5 = sub i32 0, 0
+ br label %6
+
+6:
+ %7 = or i32 %5, 0
+ %8 = zext i32 %7 to i64
+ %9 = mul i64 %8, 0
+ %10 = zext i32 %2 to i64
+ %11 = mul i64 0, %10
+ %12 = or i64 %9, %11
+ %13 = zext i32 %0 to i64
+ %14 = mul i64 %13, 0
+ %15 = zext i32 %4 to i64
+ %16 = mul i64 %15, 0
+ %17 = or i64 %14, %16
+ %18 = trunc i64 %17 to i32
+ %19 = lshr i32 %18, 0
+ %20 = trunc i64 %12 to i32
+ %21 = and i32 %20, 0
+ %22 = or i32 %19, %21
+ ret i32 %22
+}
More information about the llvm-commits
mailing list