[llvm] [SLP] Reset BatchAA after vectorizing a tree (PR #227833)

Oscar Smith via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 30 18:36:53 PDT 2026


https://github.com/oscardssmith updated https://github.com/llvm/llvm-project/pull/227833

>From 0acce58decb7715539f6011e085076941e51f01b Mon Sep 17 00:00:00 2001
From: Oscar Smith <oscar.smith at juliahub.com>
Date: Thu, 1 Oct 2026 01:28:12 +0000
Subject: [PATCH 1/2] [SLP][NFC] Precommit test for stale BatchAA after
 vectorizing a tree

The current output sinks the store to %p8 below the masked gather that
loads from %p8, which is a miscompile.

Co-Authored-By: Claude Opus 5.5 <noreply at anthropic.com>
---
 .../X86/masked-gather-stale-capture.ll        | 101 ++++++++++++++++++
 1 file changed, 101 insertions(+)
 create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll

diff --git a/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll b/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll
new file mode 100644
index 0000000000000..5e12ff02175a4
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll
@@ -0,0 +1,101 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S < %s -mtriple=x86_64-unknown-linux-gnu -mcpu=znver4 | FileCheck %s
+
+; The store to %p8 must stay before the masked gather that loads from %p8.
+; Vectorizing the first tree creates the gather, whose pointer operand
+; captures the (previously uncaptured) alloca %buf. The capture information
+; cached by the alias analysis must not be reused when scheduling the second
+; tree (the stores to %out+96/%out+104), or the store is treated as
+; independent of the gather and sunk below it.
+;
+; FIXME: The store to %p8 is currently sunk below the gather (miscompile).
+
+define void @test(ptr %out, double %v32, double %v40, double %v48, double %v56, double %k0, double %k1, double %k13) {
+; CHECK-LABEL: define void @test(
+; CHECK-SAME: ptr [[OUT:%.*]], double [[V32:%.*]], double [[V40:%.*]], double [[V48:%.*]], double [[V56:%.*]], double [[K0:%.*]], double [[K1:%.*]], double [[K13:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[BUF:%.*]] = alloca [16 x double], align 8
+; CHECK-NEXT:    call void @g()
+; CHECK-NEXT:    [[P56:%.*]] = getelementptr i8, ptr [[BUF]], i64 56
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <5 x ptr> poison, ptr [[BUF]], i64 0
+; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <5 x ptr> [[TMP0]], <5 x ptr> poison, <5 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr i8, <5 x ptr> [[TMP1]], <5 x i64> <i64 32, i64 8, i64 40, i64 48, i64 56>
+; CHECK-NEXT:    [[P48:%.*]] = getelementptr i8, ptr [[BUF]], i64 48
+; CHECK-NEXT:    [[P40:%.*]] = getelementptr i8, ptr [[BUF]], i64 40
+; CHECK-NEXT:    [[P8:%.*]] = getelementptr i8, ptr [[BUF]], i64 8
+; CHECK-NEXT:    [[P32:%.*]] = getelementptr i8, ptr [[BUF]], i64 32
+; CHECK-NEXT:    store double [[V32]], ptr [[P32]], align 8
+; CHECK-NEXT:    store double [[V40]], ptr [[P40]], align 8
+; CHECK-NEXT:    store double [[V48]], ptr [[P48]], align 8
+; CHECK-NEXT:    store double [[V56]], ptr [[P56]], align 8
+; CHECK-NEXT:    [[TMP7:%.*]] = call <5 x double> @llvm.masked.gather.v5f64.v5p0(<5 x ptr> align 8 [[TMP2]], <5 x i1> splat (i1 true), <5 x double> poison)
+; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 0, i32 2, i32 3, i32 4>
+; CHECK-NEXT:    [[TMP13:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 2, i32 2, i32 2, i32 2>
+; CHECK-NEXT:    [[TMP14:%.*]] = fmul <8 x double> [[TMP12]], [[TMP13]]
+; CHECK-NEXT:    [[TMP15:%.*]] = extractelement <5 x double> [[TMP7]], i64 0
+; CHECK-NEXT:    [[U:%.*]] = fmul double [[TMP15]], [[K13]]
+; CHECK-NEXT:    store <8 x double> [[TMP14]], ptr [[OUT]], align 8
+; CHECK-NEXT:    [[O12:%.*]] = getelementptr i8, ptr [[OUT]], i64 96
+; CHECK-NEXT:    [[S:%.*]] = fadd double [[K0]], [[K1]]
+; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <2 x double> poison, double [[S]], i64 0
+; CHECK-NEXT:    [[TMP9:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP10:%.*]] = fmul <2 x double> [[TMP9]], <double 1.000000e+00, double 0.000000e+00>
+; CHECK-NEXT:    [[TMP11:%.*]] = fadd <2 x double> [[TMP10]], zeroinitializer
+; CHECK-NEXT:    store double [[S]], ptr [[P8]], align 8
+; CHECK-NEXT:    store <2 x double> [[TMP11]], ptr [[O12]], align 8
+; CHECK-NEXT:    ret void
+;
+entry:
+  %buf = alloca [16 x double], align 8
+  %s = fadd double %k0, %k1
+  %p8 = getelementptr i8, ptr %buf, i64 8
+  store double %s, ptr %p8, align 8
+  %p32 = getelementptr i8, ptr %buf, i64 32
+  %p40 = getelementptr i8, ptr %buf, i64 40
+  %p48 = getelementptr i8, ptr %buf, i64 48
+  %p56 = getelementptr i8, ptr %buf, i64 56
+  store double %v32, ptr %p32, align 8
+  store double %v40, ptr %p40, align 8
+  store double %v48, ptr %p48, align 8
+  store double %v56, ptr %p56, align 8
+  call void @g()
+  %r0 = fadd double %s, 0.000000e+00
+  %t = fmul double %s, 0.000000e+00
+  %r1 = fadd double %t, 0.000000e+00
+  %l32 = load double, ptr %p32, align 8
+  %l8 = load double, ptr %p8, align 8
+  %m0 = fmul double %l32, %l8
+  %l40 = load double, ptr %p40, align 8
+  %m1 = fmul double %l8, %l40
+  %l48 = load double, ptr %p48, align 8
+  %m2 = fmul double %l8, %l48
+  %l56 = load double, ptr %p56, align 8
+  %m3 = fmul double %l8, %l56
+  %m4 = fmul double %l32, %l40
+  %m5 = fmul double %l40, %l40
+  %m6 = fmul double %l48, %l40
+  %m7 = fmul double %l56, %l40
+  %u = fmul double %l32, %k13
+  store double %m0, ptr %out, align 8
+  %o1 = getelementptr i8, ptr %out, i64 8
+  store double %m1, ptr %o1, align 8
+  %o2 = getelementptr i8, ptr %out, i64 16
+  store double %m2, ptr %o2, align 8
+  %o3 = getelementptr i8, ptr %out, i64 24
+  store double %m3, ptr %o3, align 8
+  %o4 = getelementptr i8, ptr %out, i64 32
+  store double %m4, ptr %o4, align 8
+  %o5 = getelementptr i8, ptr %out, i64 40
+  store double %m5, ptr %o5, align 8
+  %o6 = getelementptr i8, ptr %out, i64 48
+  store double %m6, ptr %o6, align 8
+  %o7 = getelementptr i8, ptr %out, i64 56
+  store double %m7, ptr %o7, align 8
+  %o12 = getelementptr i8, ptr %out, i64 96
+  store double %r0, ptr %o12, align 8
+  %o13 = getelementptr i8, ptr %out, i64 104
+  store double %r1, ptr %o13, align 8
+  ret void
+}
+
+declare void @g()

>From f88f9af5ee5cfab1862d15db653ae118f5436576 Mon Sep 17 00:00:00 2001
From: Oscar Smith <oscar.smith at juliahub.com>
Date: Wed, 30 Sep 2026 18:56:13 +0000
Subject: [PATCH 2/2] [SLP] Reset BatchAA after vectorizing a tree

BoUpSLP kept a single BatchAAResults for the whole function, on the
assumption that SLP never invalidates the capture information it caches.
That does not hold: vectorizing a tree can create new captures, e.g. when a
pointer to a previously uncaptured alloca is inserted into a vector of
pointers feeding a masked gather. When a later tree was scheduled, BasicAA
still treated the alloca as uncaptured, and since the gather only receives
it through a vector-of-pointers argument, it concluded that the gather
cannot read the alloca. The resulting missing memory dependency allowed a
scalar store to be sunk below a gather that loads the stored value.

Rebuild BatchAA after each vectorized tree so that later alias queries see
up-to-date capture information. Results in SLP's own AliasCache for pairs
of original instructions remain valid, since the new vector code does not
change which memory those instructions access.

Co-Authored-By: Claude Opus 5.5 <noreply at anthropic.com>
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 23 +++++++++++++------
 .../X86/masked-gather-stale-capture.ll        | 16 ++++++-------
 2 files changed, 23 insertions(+), 16 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index dbd27a3a53820..98afeb8e383d3 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -431,8 +431,9 @@ class slpvectorizer::BoUpSLP {
           TargetLibraryInfo *TLi, AAResults *Aa, LoopInfo *Li,
           DominatorTree *Dt, AssumptionCache *AC, DemandedBits *DB,
           const DataLayout *DL, OptimizationRemarkEmitter *ORE)
-      : BatchAA(*Aa), F(Func), SE(Se), TTI(Tti), TLI(TLi), LI(Li), DT(Dt),
-        AC(AC), DB(DB), DL(DL), ORE(ORE), CostKind(getSLPCostKind(Func)),
+      : AA(Aa), BatchAA(std::in_place, *Aa), F(Func), SE(Se), TTI(Tti),
+        TLI(TLi), LI(Li), DT(Dt), AC(AC), DB(DB), DL(DL), ORE(ORE),
+        CostKind(getSLPCostKind(Func)),
         Builder(Se->getContext(), TargetFolder(*DL)) {
     CodeMetrics::collectEphemeralValues(F, AC, EphValues);
     // Use the vector register size specified by the target unless overridden
@@ -3688,7 +3689,7 @@ class slpvectorizer::BoUpSLP {
     auto Res = AliasCache.try_emplace(Key);
     if (!Res.second)
       return Res.first->second;
-    bool Aliased = isModOrRefSet(BatchAA.getModRefInfo(Inst2, Loc1));
+    bool Aliased = isModOrRefSet(BatchAA->getModRefInfo(Inst2, Loc1));
     // Store the result in the cache.
     Res.first->getSecond() = Aliased;
     return Aliased;
@@ -3784,10 +3785,12 @@ class slpvectorizer::BoUpSLP {
   /// TODO: consider moving this to the AliasAnalysis itself.
   SmallDenseMap<AliasCacheKey, bool> AliasCache;
 
-  // Cache for pointerMayBeCaptured calls inside AA.  This is preserved
-  // globally through SLP because we don't perform any action which
-  // invalidates capture results.
-  BatchAAResults BatchAA;
+  AAResults *AA;
+
+  // Cache for pointerMayBeCaptured calls inside AA. Vectorizing a tree can
+  // create new captures (e.g. a pointer inserted into a vector of pointers
+  // feeding a masked gather), so this is reset after each vectorized tree.
+  std::optional<BatchAAResults> BatchAA;
 
   /// Temporary store for deleted instructions. Instructions will be deleted
   /// eventually when the BoUpSLP is destructed.  The deferral is required to
@@ -26769,6 +26772,12 @@ BoUpSLP::vectorizeTree(const ExtraValueToDebugLocsMap &ExternallyUsedValues,
   // - instructions are not deleted until later.
   removeInstructionsAndOperands(ArrayRef(RemovedInsts), VectorValuesAndScales);
 
+  // The new vector code may capture pointers that were previously not captured
+  // (e.g. the pointer operands of a masked gather), so the capture results
+  // cached in BatchAA are stale. Alias queries against the new instructions
+  // would otherwise treat such pointers as inaccessible to them.
+  BatchAA.emplace(*AA);
+
   Builder.ClearInsertionPoint();
   InstrElementSize.clear();
 
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll b/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll
index 5e12ff02175a4..39504d3579cff 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll
@@ -7,8 +7,6 @@
 ; cached by the alias analysis must not be reused when scheduling the second
 ; tree (the stores to %out+96/%out+104), or the store is treated as
 ; independent of the gather and sunk below it.
-;
-; FIXME: The store to %p8 is currently sunk below the gather (miscompile).
 
 define void @test(ptr %out, double %v32, double %v40, double %v48, double %v56, double %k0, double %k1, double %k13) {
 ; CHECK-LABEL: define void @test(
@@ -28,13 +26,6 @@ define void @test(ptr %out, double %v32, double %v40, double %v48, double %v56,
 ; CHECK-NEXT:    store double [[V40]], ptr [[P40]], align 8
 ; CHECK-NEXT:    store double [[V48]], ptr [[P48]], align 8
 ; CHECK-NEXT:    store double [[V56]], ptr [[P56]], align 8
-; CHECK-NEXT:    [[TMP7:%.*]] = call <5 x double> @llvm.masked.gather.v5f64.v5p0(<5 x ptr> align 8 [[TMP2]], <5 x i1> splat (i1 true), <5 x double> poison)
-; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 0, i32 2, i32 3, i32 4>
-; CHECK-NEXT:    [[TMP13:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 2, i32 2, i32 2, i32 2>
-; CHECK-NEXT:    [[TMP14:%.*]] = fmul <8 x double> [[TMP12]], [[TMP13]]
-; CHECK-NEXT:    [[TMP15:%.*]] = extractelement <5 x double> [[TMP7]], i64 0
-; CHECK-NEXT:    [[U:%.*]] = fmul double [[TMP15]], [[K13]]
-; CHECK-NEXT:    store <8 x double> [[TMP14]], ptr [[OUT]], align 8
 ; CHECK-NEXT:    [[O12:%.*]] = getelementptr i8, ptr [[OUT]], i64 96
 ; CHECK-NEXT:    [[S:%.*]] = fadd double [[K0]], [[K1]]
 ; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <2 x double> poison, double [[S]], i64 0
@@ -42,6 +33,13 @@ define void @test(ptr %out, double %v32, double %v40, double %v48, double %v56,
 ; CHECK-NEXT:    [[TMP10:%.*]] = fmul <2 x double> [[TMP9]], <double 1.000000e+00, double 0.000000e+00>
 ; CHECK-NEXT:    [[TMP11:%.*]] = fadd <2 x double> [[TMP10]], zeroinitializer
 ; CHECK-NEXT:    store double [[S]], ptr [[P8]], align 8
+; CHECK-NEXT:    [[TMP7:%.*]] = call <5 x double> @llvm.masked.gather.v5f64.v5p0(<5 x ptr> align 8 [[TMP2]], <5 x i1> splat (i1 true), <5 x double> poison)
+; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 0, i32 2, i32 3, i32 4>
+; CHECK-NEXT:    [[TMP13:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 2, i32 2, i32 2, i32 2>
+; CHECK-NEXT:    [[TMP14:%.*]] = fmul <8 x double> [[TMP12]], [[TMP13]]
+; CHECK-NEXT:    [[TMP15:%.*]] = extractelement <5 x double> [[TMP7]], i64 0
+; CHECK-NEXT:    [[U:%.*]] = fmul double [[TMP15]], [[K13]]
+; CHECK-NEXT:    store <8 x double> [[TMP14]], ptr [[OUT]], align 8
 ; CHECK-NEXT:    store <2 x double> [[TMP11]], ptr [[O12]], align 8
 ; CHECK-NEXT:    ret void
 ;



More information about the llvm-commits mailing list