[llvm] ce5d743 - [SLP]Fix erasing external-use replacements of narrowed reduction leaves

via llvm-commits llvm-commits at lists.llvm.org
Fri Aug 28 05:54:24 PDT 2026


Author: Alexey Bataev
Date: 2026-08-28T08:54:20-04:00
New Revision: ce5d7430aab24cef7c9a3b786c26073001eecc4d

URL: https://github.com/llvm/llvm-project/commit/ce5d7430aab24cef7c9a3b786c26073001eecc4d
DIFF: https://github.com/llvm/llvm-project/commit/ce5d7430aab24cef7c9a3b786c26073001eecc4d.diff

LOG: [SLP]Fix erasing external-use replacements of narrowed reduction leaves

A narrowed reduction leaf that stays a leftover reduction value but is
vectorized as part of the tree has no reduction op user, so its
external-use replacement is swept as a dead operand together with the
tree scalars, and the reduction epilogue reuses the dead value. Keep
the replacements of reduced values without reduction op users.

Fixes https://github.com/llvm/llvm-project/pull/216062#issuecomment-5451142901

Reviewers: 

Pull Request: https://github.com/llvm/llvm-project/pull/219460

Added: 
    llvm/test/Transforms/SLPVectorizer/X86/narrowed-reduction-leftover-leaf.ll

Modified: 
    llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 358901da42266..a7eaef3701415 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -949,6 +949,7 @@ class slpvectorizer::BoUpSLP {
     ExternalUses.clear();
     ExternalUsesAsOriginalScalar.clear();
     ExternalUsesWithNonUsers.clear();
+    ExternalUseReplacements.clear();
     RTChecks.clear();
     HasRuntimeCheckableBlockers = false;
     HasNonCheckableMemBlocker = false;
@@ -2504,6 +2505,7 @@ class slpvectorizer::BoUpSLP {
         if (auto *OpI = dyn_cast_if_present<Instruction>(U.get());
             OpI && !DeletedInstructions.contains(OpI) && OpI->hasOneUser() &&
             wouldInstructionBeTriviallyDead(OpI, TLI) &&
+            !ExternalUseReplacements.contains(OpI) &&
             (Entries.empty() || none_of(Entries, [&](const TreeEntry *Entry) {
                return Entry->VectorizedValue == OpI;
              })))
@@ -2553,6 +2555,7 @@ class slpvectorizer::BoUpSLP {
         // loop iteration.
         if (auto *OpI = dyn_cast<Instruction>(OpV))
           if (!DeletedInstructions.contains(OpI) &&
+              !ExternalUseReplacements.contains(OpI) &&
               (!OpI->getType()->isVectorTy() ||
                none_of(
                    VectorValuesAndScales,
@@ -3960,6 +3963,11 @@ class slpvectorizer::BoUpSLP {
   /// uses.
   SmallPtrSet<Value *, 4> ExternalUsesWithNonUsers;
 
+  /// Replacements emitted for the external uses without users, consumed after
+  /// the tree vectorization; must not be collected as dead operands of the
+  /// erased scalars.
+  SmallPtrSet<Value *, 4> ExternalUseReplacements;
+
   /// Values used only by @llvm.assume calls.
   SmallPtrSet<const Value *, 32> EphValues;
 
@@ -25891,7 +25899,17 @@ Value *BoUpSLP::vectorizeTree(
         assert((!isa<ExtractElementInst>(Scalar) ||
                 !IgnoredExtracts.contains(cast<ExtractElementInst>(Scalar))) &&
                "Extractelements should not be replaced.");
+        // Reduced values that are not operands of the reduction ops (e.g.
+        // narrowed reduction leaves) lose all users once the tree scalars are
+        // erased; keep the replacement alive for the reduction epilogue.
+        bool NeedsProtection =
+            UserIgnoreList && none_of(Scalar->users(), [&](llvm::User *U) {
+              return UserIgnoreList->contains(U);
+            });
         Scalar->replaceAllUsesWith(NewInst);
+        if (NeedsProtection)
+          if (auto *I = dyn_cast<Instruction>(NewInst))
+            ExternalUseReplacements.insert(I);
       }
       continue;
     }

diff  --git a/llvm/test/Transforms/SLPVectorizer/X86/narrowed-reduction-leftover-leaf.ll b/llvm/test/Transforms/SLPVectorizer/X86/narrowed-reduction-leftover-leaf.ll
new file mode 100644
index 0000000000000..223d187ea6351
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/narrowed-reduction-leftover-leaf.ll
@@ -0,0 +1,59 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu -mcpu=x86-64-v2 < %s | FileCheck %s
+
+define void @test() {
+; CHECK-LABEL: define void @test(
+; CHECK-SAME: ) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[T2:%.*]] = trunc i32 0 to i8
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = phi <4 x i8> [ [[TMP4:%.*]], %[[LOOP]] ], [ zeroinitializer, %[[ENTRY]] ]
+; CHECK-NEXT:    [[LD:%.*]] = load i32, ptr null, align 1
+; CHECK-NEXT:    [[SH:%.*]] = lshr i32 [[LD]], 0
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x i32> <i32 poison, i32 0, i32 0, i32 0>, i32 [[LD]], i64 0
+; CHECK-NEXT:    [[TMP2:%.*]] = lshr <4 x i32> [[TMP1]], zeroinitializer
+; CHECK-NEXT:    [[TMP3:%.*]] = trunc <4 x i32> [[TMP2]] to <4 x i8>
+; CHECK-NEXT:    [[T3:%.*]] = trunc i32 [[SH]] to i8
+; CHECK-NEXT:    [[TMP4]] = or <4 x i8> [[TMP0]], [[TMP3]]
+; CHECK-NEXT:    [[TMP5:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i32>
+; CHECK-NEXT:    [[TMP6:%.*]] = call i32 @llvm.vector.reduce.or.v4i32(<4 x i32> [[TMP5]])
+; CHECK-NEXT:    [[TMP7:%.*]] = zext i8 [[T2]] to i32
+; CHECK-NEXT:    [[OP_RDX:%.*]] = or i32 [[TMP6]], [[TMP7]]
+; CHECK-NEXT:    [[TMP8:%.*]] = zext i8 [[T2]] to i32
+; CHECK-NEXT:    [[TMP9:%.*]] = zext i8 [[T3]] to i32
+; CHECK-NEXT:    [[OP_RDX1:%.*]] = or i32 [[TMP8]], [[TMP9]]
+; CHECK-NEXT:    [[OP_RDX2:%.*]] = or i32 [[OP_RDX]], [[OP_RDX1]]
+; CHECK-NEXT:    store i32 [[OP_RDX2]], ptr null, align 1
+; CHECK-NEXT:    br label %[[LOOP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %p3 = phi i8 [ %o3, %loop ], [ 0, %entry ]
+  %p2 = phi i8 [ %o2, %loop ], [ 0, %entry ]
+  %p1 = phi i8 [ %o1, %loop ], [ 0, %entry ]
+  %p0 = phi i8 [ %o0, %loop ], [ 0, %entry ]
+  %t0 = trunc i32 0 to i8
+  %o0 = or i8 %p0, %t0
+  %t1 = trunc i32 0 to i8
+  %o1 = or i8 %p1, %t1
+  %t2 = trunc i32 0 to i8
+  %o2 = or i8 %p2, %t2
+  %ld = load i32, ptr null, align 1
+  %sh = lshr i32 %ld, 0
+  %t3 = trunc i32 %sh to i8
+  %o3 = or i8 %p3, %t3
+  %z3 = zext i8 %o3 to i32
+  %s3 = shl i32 %z3, 0
+  %z2 = zext i8 %o2 to i32
+  %a1 = or i32 %s3, %z2
+  %z1 = zext i8 %p1 to i32
+  %s1 = shl i32 %z1, 0
+  %a2 = or i32 %a1, %s1
+  %z0 = zext i8 %o0 to i32
+  %a3 = or i32 %a2, %z0
+  store i32 %a3, ptr null, align 1
+  br label %loop
+}


        


More information about the llvm-commits mailing list