[llvm] [SLP]Bail out on non-schedulable expanded binop with stale operand deps (PR #196449)

via llvm-commits llvm-commits at lists.llvm.org
Thu May 7 16:56:13 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Alexey Bataev (alexey-bataev)

<details>
<summary>Changes</summary>

In tryScheduleBundle's DoesNotRequireScheduling path, an expanded binop
(shl X, 1 modeled as add X, X) doubles the dependency count of the
duplicated operand. If the operand has a
single IR use yet its ScheduleData already has Dependencies populated
by an earlier calculation that did not see the expanded duplicate use,
double decrement still exceeds calculateDependencies' single increment
and UnscheduledDeps goes negative.

Fixes #<!-- -->196281.


---
Full diff: https://github.com/llvm/llvm-project/pull/196449.diff


3 Files Affected:

- (modified) llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp (+11-5) 
- (modified) llvm/test/Transforms/SLPVectorizer/X86/expanded-binop-doesnotneedschedule-user.ll (+3-3) 
- (added) llvm/test/Transforms/SLPVectorizer/X86/expanded-operand-already-scheduled.ll (+50) 


``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 2a9d98cedc1eb..9cfeae0b8bee7 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -24608,16 +24608,22 @@ BoUpSLP::BlockScheduling::tryScheduleBundle(ArrayRef<Value *> VL, BoUpSLP *SLP,
       // dependency count of the duplicated operand. This non-schedulable
       // path does not clear operand ScheduleData dependencies the way the
       // regular scheduling path does (see CheckIfNeedToClearDeps below), so
-      // when that operand has more uses than this bundle member, the
-      // schedule's decrement count exceeds calculateDependencies' increment
-      // count and UnscheduledDeps goes negative. Bail out instead of
-      // producing an inconsistent schedule.
+      // when that operand has more uses than this bundle member, or its
+      // ScheduleData already has computed dependencies that did not see
+      // the expanded duplicate use, the schedule's decrement count exceeds
+      // calculateDependencies' increment count and UnscheduledDeps goes
+      // negative. Bail out instead of producing an inconsistent schedule.
       if (S.isExpandedBinOp(I) &&
           any_of(enumerate(I->operands()), [&](const auto &P) {
             if (S.isExpandedOperand(I, P.index()))
               return false;
             auto *OpI = dyn_cast<Instruction>(P.value());
-            return OpI && !OpI->hasOneUse();
+            if (!OpI)
+              return false;
+            if (!OpI->hasOneUse())
+              return true;
+            ScheduleData *SD = getScheduleData(OpI);
+            return SD && SD->hasValidDependencies();
           }))
         return std::nullopt;
       if (EI && EI.UserTE->State == TreeEntry::Vectorize &&
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/expanded-binop-doesnotneedschedule-user.ll b/llvm/test/Transforms/SLPVectorizer/X86/expanded-binop-doesnotneedschedule-user.ll
index 3f233d26e462b..fa0f927e36395 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/expanded-binop-doesnotneedschedule-user.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/expanded-binop-doesnotneedschedule-user.ll
@@ -8,13 +8,13 @@ define void @test() {
 ; CHECK-NEXT:    br label %[[BB1:.*]]
 ; CHECK:       [[BB1]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = phi <2 x i32> [ [[TMP5:%.*]], %[[BB1]] ], [ zeroinitializer, %[[BB]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = phi <2 x i32> [ [[TMP7:%.*]], %[[BB1]] ], [ zeroinitializer, %[[BB]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = phi <2 x i32> [ [[TMP6:%.*]], %[[BB1]] ], [ zeroinitializer, %[[BB]] ]
 ; CHECK-NEXT:    [[TMP2:%.*]] = mul <2 x i32> [[TMP1]], <i32 0, i32 1>
 ; CHECK-NEXT:    [[TMP3:%.*]] = shl <2 x i32> [[TMP2]], zeroinitializer
 ; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <2 x i32> [[TMP3]], <2 x i32> <i32 1, i32 poison>, <2 x i32> <i32 2, i32 1>
 ; CHECK-NEXT:    [[TMP5]] = add <2 x i32> [[TMP4]], [[TMP3]]
-; CHECK-NEXT:    [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP4]], <2 x i32> <i32 0, i32 poison>, <2 x i32> <i32 2, i32 1>
-; CHECK-NEXT:    [[TMP7]] = add <2 x i32> [[TMP6]], [[TMP4]]
+; CHECK-NEXT:    [[ADD6:%.*]] = add i32 0, 1
+; CHECK-NEXT:    [[TMP6]] = insertelement <2 x i32> [[TMP5]], i32 [[ADD6]], i32 0
 ; CHECK-NEXT:    br label %[[BB1]]
 ;
 bb:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/expanded-operand-already-scheduled.ll b/llvm/test/Transforms/SLPVectorizer/X86/expanded-operand-already-scheduled.ll
new file mode 100644
index 0000000000000..2079aac55de9c
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/expanded-operand-already-scheduled.ll
@@ -0,0 +1,50 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt --passes=slp-vectorizer -S -slp-threshold=-99999 -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s
+
+define i32 @test() {
+; CHECK-LABEL: define i32 @test() {
+; CHECK-NEXT:  [[Q_EXIT8:.*:]]
+; CHECK-NEXT:    [[TMP0:%.*]] = add <2 x i32> zeroinitializer, <i32 1, i32 0>
+; CHECK-NEXT:    br label %[[BB1:.*]]
+; CHECK:       [[BB1]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = trunc <2 x i32> [[TMP0]] to <2 x i1>
+; CHECK-NEXT:    [[TMP3:%.*]] = mul <2 x i1> [[TMP2]], zeroinitializer
+; CHECK-NEXT:    [[TMP4:%.*]] = or <2 x i1> zeroinitializer, [[TMP3]]
+; CHECK-NEXT:    [[TMP5:%.*]] = lshr <2 x i1> [[TMP4]], zeroinitializer
+; CHECK-NEXT:    [[TMP6:%.*]] = and <2 x i1> [[TMP4]], zeroinitializer
+; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <2 x i1> [[TMP5]], <2 x i1> [[TMP6]], <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x i1> [[TMP7]], i32 0
+; CHECK-NEXT:    [[TMP9:%.*]] = zext i1 [[TMP8]] to i32
+; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <2 x i1> [[TMP7]], i32 1
+; CHECK-NEXT:    [[TMP11:%.*]] = zext i1 [[TMP10]] to i32
+; CHECK-NEXT:    [[TMP12:%.*]] = or i32 [[TMP9]], [[TMP11]]
+; CHECK-NEXT:    ret i32 [[TMP12]]
+;
+q.exit8:
+  %0 = add i32 0, 0
+  %1 = or i32 %0, 0
+  %2 = shl i32 %1, 1
+  %3 = lshr i32 0, 0
+  %4 = add i32 %3, 1
+  %5 = sub i32 0, 0
+  br label %6
+
+6:
+  %7 = or i32 %5, 0
+  %8 = zext i32 %7 to i64
+  %9 = mul i64 %8, 0
+  %10 = zext i32 %2 to i64
+  %11 = mul i64 0, %10
+  %12 = or i64 %9, %11
+  %13 = zext i32 %0 to i64
+  %14 = mul i64 %13, 0
+  %15 = zext i32 %4 to i64
+  %16 = mul i64 %15, 0
+  %17 = or i64 %14, %16
+  %18 = trunc i64 %17 to i32
+  %19 = lshr i32 %18, 0
+  %20 = trunc i64 %12 to i32
+  %21 = and i32 %20, 0
+  %22 = or i32 %19, %21
+  ret i32 %22
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/196449


More information about the llvm-commits mailing list