[llvm] ce5d743 - [SLP]Fix erasing external-use replacements of narrowed reduction leaves
via llvm-commits
llvm-commits at lists.llvm.org
Fri Aug 28 05:54:24 PDT 2026
Author: Alexey Bataev
Date: 2026-08-28T08:54:20-04:00
New Revision: ce5d7430aab24cef7c9a3b786c26073001eecc4d
URL: https://github.com/llvm/llvm-project/commit/ce5d7430aab24cef7c9a3b786c26073001eecc4d
DIFF: https://github.com/llvm/llvm-project/commit/ce5d7430aab24cef7c9a3b786c26073001eecc4d.diff
LOG: [SLP]Fix erasing external-use replacements of narrowed reduction leaves
A narrowed reduction leaf that stays a leftover reduction value but is
vectorized as part of the tree has no reduction op user, so its
external-use replacement is swept as a dead operand together with the
tree scalars, and the reduction epilogue reuses the dead value. Keep
the replacements of reduced values without reduction op users.
Fixes https://github.com/llvm/llvm-project/pull/216062#issuecomment-5451142901
Reviewers:
Pull Request: https://github.com/llvm/llvm-project/pull/219460
Added:
llvm/test/Transforms/SLPVectorizer/X86/narrowed-reduction-leftover-leaf.ll
Modified:
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 358901da42266..a7eaef3701415 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -949,6 +949,7 @@ class slpvectorizer::BoUpSLP {
ExternalUses.clear();
ExternalUsesAsOriginalScalar.clear();
ExternalUsesWithNonUsers.clear();
+ ExternalUseReplacements.clear();
RTChecks.clear();
HasRuntimeCheckableBlockers = false;
HasNonCheckableMemBlocker = false;
@@ -2504,6 +2505,7 @@ class slpvectorizer::BoUpSLP {
if (auto *OpI = dyn_cast_if_present<Instruction>(U.get());
OpI && !DeletedInstructions.contains(OpI) && OpI->hasOneUser() &&
wouldInstructionBeTriviallyDead(OpI, TLI) &&
+ !ExternalUseReplacements.contains(OpI) &&
(Entries.empty() || none_of(Entries, [&](const TreeEntry *Entry) {
return Entry->VectorizedValue == OpI;
})))
@@ -2553,6 +2555,7 @@ class slpvectorizer::BoUpSLP {
// loop iteration.
if (auto *OpI = dyn_cast<Instruction>(OpV))
if (!DeletedInstructions.contains(OpI) &&
+ !ExternalUseReplacements.contains(OpI) &&
(!OpI->getType()->isVectorTy() ||
none_of(
VectorValuesAndScales,
@@ -3960,6 +3963,11 @@ class slpvectorizer::BoUpSLP {
/// uses.
SmallPtrSet<Value *, 4> ExternalUsesWithNonUsers;
+ /// Replacements emitted for the external uses without users, consumed after
+ /// the tree vectorization; must not be collected as dead operands of the
+ /// erased scalars.
+ SmallPtrSet<Value *, 4> ExternalUseReplacements;
+
/// Values used only by @llvm.assume calls.
SmallPtrSet<const Value *, 32> EphValues;
@@ -25891,7 +25899,17 @@ Value *BoUpSLP::vectorizeTree(
assert((!isa<ExtractElementInst>(Scalar) ||
!IgnoredExtracts.contains(cast<ExtractElementInst>(Scalar))) &&
"Extractelements should not be replaced.");
+ // Reduced values that are not operands of the reduction ops (e.g.
+ // narrowed reduction leaves) lose all users once the tree scalars are
+ // erased; keep the replacement alive for the reduction epilogue.
+ bool NeedsProtection =
+ UserIgnoreList && none_of(Scalar->users(), [&](llvm::User *U) {
+ return UserIgnoreList->contains(U);
+ });
Scalar->replaceAllUsesWith(NewInst);
+ if (NeedsProtection)
+ if (auto *I = dyn_cast<Instruction>(NewInst))
+ ExternalUseReplacements.insert(I);
}
continue;
}
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/narrowed-reduction-leftover-leaf.ll b/llvm/test/Transforms/SLPVectorizer/X86/narrowed-reduction-leftover-leaf.ll
new file mode 100644
index 0000000000000..223d187ea6351
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/narrowed-reduction-leftover-leaf.ll
@@ -0,0 +1,59 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu -mcpu=x86-64-v2 < %s | FileCheck %s
+
+define void @test() {
+; CHECK-LABEL: define void @test(
+; CHECK-SAME: ) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[T2:%.*]] = trunc i32 0 to i8
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[TMP0:%.*]] = phi <4 x i8> [ [[TMP4:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
+; CHECK-NEXT: [[LD:%.*]] = load i32, ptr null, align 1
+; CHECK-NEXT: [[SH:%.*]] = lshr i32 [[LD]], 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x i32> <i32 poison, i32 0, i32 0, i32 0>, i32 [[LD]], i64 0
+; CHECK-NEXT: [[TMP2:%.*]] = lshr <4 x i32> [[TMP1]], zeroinitializer
+; CHECK-NEXT: [[TMP3:%.*]] = trunc <4 x i32> [[TMP2]] to <4 x i8>
+; CHECK-NEXT: [[T3:%.*]] = trunc i32 [[SH]] to i8
+; CHECK-NEXT: [[TMP4]] = or <4 x i8> [[TMP0]], [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i32>
+; CHECK-NEXT: [[TMP6:%.*]] = call i32 @llvm.vector.reduce.or.v4i32(<4 x i32> [[TMP5]])
+; CHECK-NEXT: [[TMP7:%.*]] = zext i8 [[T2]] to i32
+; CHECK-NEXT: [[OP_RDX:%.*]] = or i32 [[TMP6]], [[TMP7]]
+; CHECK-NEXT: [[TMP8:%.*]] = zext i8 [[T2]] to i32
+; CHECK-NEXT: [[TMP9:%.*]] = zext i8 [[T3]] to i32
+; CHECK-NEXT: [[OP_RDX1:%.*]] = or i32 [[TMP8]], [[TMP9]]
+; CHECK-NEXT: [[OP_RDX2:%.*]] = or i32 [[OP_RDX]], [[OP_RDX1]]
+; CHECK-NEXT: store i32 [[OP_RDX2]], ptr null, align 1
+; CHECK-NEXT: br label %[[LOOP]]
+;
+entry:
+ br label %loop
+
+loop:
+ %p3 = phi i8 [ %o3, %loop ], [ 0, %entry ]
+ %p2 = phi i8 [ %o2, %loop ], [ 0, %entry ]
+ %p1 = phi i8 [ %o1, %loop ], [ 0, %entry ]
+ %p0 = phi i8 [ %o0, %loop ], [ 0, %entry ]
+ %t0 = trunc i32 0 to i8
+ %o0 = or i8 %p0, %t0
+ %t1 = trunc i32 0 to i8
+ %o1 = or i8 %p1, %t1
+ %t2 = trunc i32 0 to i8
+ %o2 = or i8 %p2, %t2
+ %ld = load i32, ptr null, align 1
+ %sh = lshr i32 %ld, 0
+ %t3 = trunc i32 %sh to i8
+ %o3 = or i8 %p3, %t3
+ %z3 = zext i8 %o3 to i32
+ %s3 = shl i32 %z3, 0
+ %z2 = zext i8 %o2 to i32
+ %a1 = or i32 %s3, %z2
+ %z1 = zext i8 %p1 to i32
+ %s1 = shl i32 %z1, 0
+ %a2 = or i32 %a1, %s1
+ %z0 = zext i8 %o0 to i32
+ %a3 = or i32 %a2, %z0
+ store i32 %a3, ptr null, align 1
+ br label %loop
+}
More information about the llvm-commits
mailing list