[llvm] [SLP] Reset BatchAA after vectorizing a tree (PR #227833)
Oscar Smith via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 18:36:53 PDT 2026
https://github.com/oscardssmith updated https://github.com/llvm/llvm-project/pull/227833
>From 0acce58decb7715539f6011e085076941e51f01b Mon Sep 17 00:00:00 2001
From: Oscar Smith <oscar.smith at juliahub.com>
Date: Thu, 1 Oct 2026 01:28:12 +0000
Subject: [PATCH 1/2] [SLP][NFC] Precommit test for stale BatchAA after
vectorizing a tree
The current output sinks the store to %p8 below the masked gather that
loads from %p8, which is a miscompile.
Co-Authored-By: Claude Opus 5.5 <noreply at anthropic.com>
---
.../X86/masked-gather-stale-capture.ll | 101 ++++++++++++++++++
1 file changed, 101 insertions(+)
create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll b/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll
new file mode 100644
index 0000000000000..5e12ff02175a4
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll
@@ -0,0 +1,101 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S < %s -mtriple=x86_64-unknown-linux-gnu -mcpu=znver4 | FileCheck %s
+
+; The store to %p8 must stay before the masked gather that loads from %p8.
+; Vectorizing the first tree creates the gather, whose pointer operand
+; captures the (previously uncaptured) alloca %buf. The capture information
+; cached by the alias analysis must not be reused when scheduling the second
+; tree (the stores to %out+96/%out+104), or the store is treated as
+; independent of the gather and sunk below it.
+;
+; FIXME: The store to %p8 is currently sunk below the gather (miscompile).
+
+define void @test(ptr %out, double %v32, double %v40, double %v48, double %v56, double %k0, double %k1, double %k13) {
+; CHECK-LABEL: define void @test(
+; CHECK-SAME: ptr [[OUT:%.*]], double [[V32:%.*]], double [[V40:%.*]], double [[V48:%.*]], double [[V56:%.*]], double [[K0:%.*]], double [[K1:%.*]], double [[K13:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[BUF:%.*]] = alloca [16 x double], align 8
+; CHECK-NEXT: call void @g()
+; CHECK-NEXT: [[P56:%.*]] = getelementptr i8, ptr [[BUF]], i64 56
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <5 x ptr> poison, ptr [[BUF]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <5 x ptr> [[TMP0]], <5 x ptr> poison, <5 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr i8, <5 x ptr> [[TMP1]], <5 x i64> <i64 32, i64 8, i64 40, i64 48, i64 56>
+; CHECK-NEXT: [[P48:%.*]] = getelementptr i8, ptr [[BUF]], i64 48
+; CHECK-NEXT: [[P40:%.*]] = getelementptr i8, ptr [[BUF]], i64 40
+; CHECK-NEXT: [[P8:%.*]] = getelementptr i8, ptr [[BUF]], i64 8
+; CHECK-NEXT: [[P32:%.*]] = getelementptr i8, ptr [[BUF]], i64 32
+; CHECK-NEXT: store double [[V32]], ptr [[P32]], align 8
+; CHECK-NEXT: store double [[V40]], ptr [[P40]], align 8
+; CHECK-NEXT: store double [[V48]], ptr [[P48]], align 8
+; CHECK-NEXT: store double [[V56]], ptr [[P56]], align 8
+; CHECK-NEXT: [[TMP7:%.*]] = call <5 x double> @llvm.masked.gather.v5f64.v5p0(<5 x ptr> align 8 [[TMP2]], <5 x i1> splat (i1 true), <5 x double> poison)
+; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 0, i32 2, i32 3, i32 4>
+; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 2, i32 2, i32 2, i32 2>
+; CHECK-NEXT: [[TMP14:%.*]] = fmul <8 x double> [[TMP12]], [[TMP13]]
+; CHECK-NEXT: [[TMP15:%.*]] = extractelement <5 x double> [[TMP7]], i64 0
+; CHECK-NEXT: [[U:%.*]] = fmul double [[TMP15]], [[K13]]
+; CHECK-NEXT: store <8 x double> [[TMP14]], ptr [[OUT]], align 8
+; CHECK-NEXT: [[O12:%.*]] = getelementptr i8, ptr [[OUT]], i64 96
+; CHECK-NEXT: [[S:%.*]] = fadd double [[K0]], [[K1]]
+; CHECK-NEXT: [[TMP8:%.*]] = insertelement <2 x double> poison, double [[S]], i64 0
+; CHECK-NEXT: [[TMP9:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP10:%.*]] = fmul <2 x double> [[TMP9]], <double 1.000000e+00, double 0.000000e+00>
+; CHECK-NEXT: [[TMP11:%.*]] = fadd <2 x double> [[TMP10]], zeroinitializer
+; CHECK-NEXT: store double [[S]], ptr [[P8]], align 8
+; CHECK-NEXT: store <2 x double> [[TMP11]], ptr [[O12]], align 8
+; CHECK-NEXT: ret void
+;
+entry:
+ %buf = alloca [16 x double], align 8
+ %s = fadd double %k0, %k1
+ %p8 = getelementptr i8, ptr %buf, i64 8
+ store double %s, ptr %p8, align 8
+ %p32 = getelementptr i8, ptr %buf, i64 32
+ %p40 = getelementptr i8, ptr %buf, i64 40
+ %p48 = getelementptr i8, ptr %buf, i64 48
+ %p56 = getelementptr i8, ptr %buf, i64 56
+ store double %v32, ptr %p32, align 8
+ store double %v40, ptr %p40, align 8
+ store double %v48, ptr %p48, align 8
+ store double %v56, ptr %p56, align 8
+ call void @g()
+ %r0 = fadd double %s, 0.000000e+00
+ %t = fmul double %s, 0.000000e+00
+ %r1 = fadd double %t, 0.000000e+00
+ %l32 = load double, ptr %p32, align 8
+ %l8 = load double, ptr %p8, align 8
+ %m0 = fmul double %l32, %l8
+ %l40 = load double, ptr %p40, align 8
+ %m1 = fmul double %l8, %l40
+ %l48 = load double, ptr %p48, align 8
+ %m2 = fmul double %l8, %l48
+ %l56 = load double, ptr %p56, align 8
+ %m3 = fmul double %l8, %l56
+ %m4 = fmul double %l32, %l40
+ %m5 = fmul double %l40, %l40
+ %m6 = fmul double %l48, %l40
+ %m7 = fmul double %l56, %l40
+ %u = fmul double %l32, %k13
+ store double %m0, ptr %out, align 8
+ %o1 = getelementptr i8, ptr %out, i64 8
+ store double %m1, ptr %o1, align 8
+ %o2 = getelementptr i8, ptr %out, i64 16
+ store double %m2, ptr %o2, align 8
+ %o3 = getelementptr i8, ptr %out, i64 24
+ store double %m3, ptr %o3, align 8
+ %o4 = getelementptr i8, ptr %out, i64 32
+ store double %m4, ptr %o4, align 8
+ %o5 = getelementptr i8, ptr %out, i64 40
+ store double %m5, ptr %o5, align 8
+ %o6 = getelementptr i8, ptr %out, i64 48
+ store double %m6, ptr %o6, align 8
+ %o7 = getelementptr i8, ptr %out, i64 56
+ store double %m7, ptr %o7, align 8
+ %o12 = getelementptr i8, ptr %out, i64 96
+ store double %r0, ptr %o12, align 8
+ %o13 = getelementptr i8, ptr %out, i64 104
+ store double %r1, ptr %o13, align 8
+ ret void
+}
+
+declare void @g()
>From f88f9af5ee5cfab1862d15db653ae118f5436576 Mon Sep 17 00:00:00 2001
From: Oscar Smith <oscar.smith at juliahub.com>
Date: Wed, 30 Sep 2026 18:56:13 +0000
Subject: [PATCH 2/2] [SLP] Reset BatchAA after vectorizing a tree
BoUpSLP kept a single BatchAAResults for the whole function, on the
assumption that SLP never invalidates the capture information it caches.
That does not hold: vectorizing a tree can create new captures, e.g. when a
pointer to a previously uncaptured alloca is inserted into a vector of
pointers feeding a masked gather. When a later tree was scheduled, BasicAA
still treated the alloca as uncaptured, and since the gather only receives
it through a vector-of-pointers argument, it concluded that the gather
cannot read the alloca. The resulting missing memory dependency allowed a
scalar store to be sunk below a gather that loads the stored value.
Rebuild BatchAA after each vectorized tree so that later alias queries see
up-to-date capture information. Results in SLP's own AliasCache for pairs
of original instructions remain valid, since the new vector code does not
change which memory those instructions access.
Co-Authored-By: Claude Opus 5.5 <noreply at anthropic.com>
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 23 +++++++++++++------
.../X86/masked-gather-stale-capture.ll | 16 ++++++-------
2 files changed, 23 insertions(+), 16 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index dbd27a3a53820..98afeb8e383d3 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -431,8 +431,9 @@ class slpvectorizer::BoUpSLP {
TargetLibraryInfo *TLi, AAResults *Aa, LoopInfo *Li,
DominatorTree *Dt, AssumptionCache *AC, DemandedBits *DB,
const DataLayout *DL, OptimizationRemarkEmitter *ORE)
- : BatchAA(*Aa), F(Func), SE(Se), TTI(Tti), TLI(TLi), LI(Li), DT(Dt),
- AC(AC), DB(DB), DL(DL), ORE(ORE), CostKind(getSLPCostKind(Func)),
+ : AA(Aa), BatchAA(std::in_place, *Aa), F(Func), SE(Se), TTI(Tti),
+ TLI(TLi), LI(Li), DT(Dt), AC(AC), DB(DB), DL(DL), ORE(ORE),
+ CostKind(getSLPCostKind(Func)),
Builder(Se->getContext(), TargetFolder(*DL)) {
CodeMetrics::collectEphemeralValues(F, AC, EphValues);
// Use the vector register size specified by the target unless overridden
@@ -3688,7 +3689,7 @@ class slpvectorizer::BoUpSLP {
auto Res = AliasCache.try_emplace(Key);
if (!Res.second)
return Res.first->second;
- bool Aliased = isModOrRefSet(BatchAA.getModRefInfo(Inst2, Loc1));
+ bool Aliased = isModOrRefSet(BatchAA->getModRefInfo(Inst2, Loc1));
// Store the result in the cache.
Res.first->getSecond() = Aliased;
return Aliased;
@@ -3784,10 +3785,12 @@ class slpvectorizer::BoUpSLP {
/// TODO: consider moving this to the AliasAnalysis itself.
SmallDenseMap<AliasCacheKey, bool> AliasCache;
- // Cache for pointerMayBeCaptured calls inside AA. This is preserved
- // globally through SLP because we don't perform any action which
- // invalidates capture results.
- BatchAAResults BatchAA;
+ AAResults *AA;
+
+ // Cache for pointerMayBeCaptured calls inside AA. Vectorizing a tree can
+ // create new captures (e.g. a pointer inserted into a vector of pointers
+ // feeding a masked gather), so this is reset after each vectorized tree.
+ std::optional<BatchAAResults> BatchAA;
/// Temporary store for deleted instructions. Instructions will be deleted
/// eventually when the BoUpSLP is destructed. The deferral is required to
@@ -26769,6 +26772,12 @@ BoUpSLP::vectorizeTree(const ExtraValueToDebugLocsMap &ExternallyUsedValues,
// - instructions are not deleted until later.
removeInstructionsAndOperands(ArrayRef(RemovedInsts), VectorValuesAndScales);
+ // The new vector code may capture pointers that were previously not captured
+ // (e.g. the pointer operands of a masked gather), so the capture results
+ // cached in BatchAA are stale. Alias queries against the new instructions
+ // would otherwise treat such pointers as inaccessible to them.
+ BatchAA.emplace(*AA);
+
Builder.ClearInsertionPoint();
InstrElementSize.clear();
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll b/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll
index 5e12ff02175a4..39504d3579cff 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/masked-gather-stale-capture.ll
@@ -7,8 +7,6 @@
; cached by the alias analysis must not be reused when scheduling the second
; tree (the stores to %out+96/%out+104), or the store is treated as
; independent of the gather and sunk below it.
-;
-; FIXME: The store to %p8 is currently sunk below the gather (miscompile).
define void @test(ptr %out, double %v32, double %v40, double %v48, double %v56, double %k0, double %k1, double %k13) {
; CHECK-LABEL: define void @test(
@@ -28,13 +26,6 @@ define void @test(ptr %out, double %v32, double %v40, double %v48, double %v56,
; CHECK-NEXT: store double [[V40]], ptr [[P40]], align 8
; CHECK-NEXT: store double [[V48]], ptr [[P48]], align 8
; CHECK-NEXT: store double [[V56]], ptr [[P56]], align 8
-; CHECK-NEXT: [[TMP7:%.*]] = call <5 x double> @llvm.masked.gather.v5f64.v5p0(<5 x ptr> align 8 [[TMP2]], <5 x i1> splat (i1 true), <5 x double> poison)
-; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 0, i32 2, i32 3, i32 4>
-; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 2, i32 2, i32 2, i32 2>
-; CHECK-NEXT: [[TMP14:%.*]] = fmul <8 x double> [[TMP12]], [[TMP13]]
-; CHECK-NEXT: [[TMP15:%.*]] = extractelement <5 x double> [[TMP7]], i64 0
-; CHECK-NEXT: [[U:%.*]] = fmul double [[TMP15]], [[K13]]
-; CHECK-NEXT: store <8 x double> [[TMP14]], ptr [[OUT]], align 8
; CHECK-NEXT: [[O12:%.*]] = getelementptr i8, ptr [[OUT]], i64 96
; CHECK-NEXT: [[S:%.*]] = fadd double [[K0]], [[K1]]
; CHECK-NEXT: [[TMP8:%.*]] = insertelement <2 x double> poison, double [[S]], i64 0
@@ -42,6 +33,13 @@ define void @test(ptr %out, double %v32, double %v40, double %v48, double %v56,
; CHECK-NEXT: [[TMP10:%.*]] = fmul <2 x double> [[TMP9]], <double 1.000000e+00, double 0.000000e+00>
; CHECK-NEXT: [[TMP11:%.*]] = fadd <2 x double> [[TMP10]], zeroinitializer
; CHECK-NEXT: store double [[S]], ptr [[P8]], align 8
+; CHECK-NEXT: [[TMP7:%.*]] = call <5 x double> @llvm.masked.gather.v5f64.v5p0(<5 x ptr> align 8 [[TMP2]], <5 x i1> splat (i1 true), <5 x double> poison)
+; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 0, i32 1, i32 1, i32 1, i32 0, i32 2, i32 3, i32 4>
+; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <5 x double> [[TMP7]], <5 x double> poison, <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 2, i32 2, i32 2, i32 2>
+; CHECK-NEXT: [[TMP14:%.*]] = fmul <8 x double> [[TMP12]], [[TMP13]]
+; CHECK-NEXT: [[TMP15:%.*]] = extractelement <5 x double> [[TMP7]], i64 0
+; CHECK-NEXT: [[U:%.*]] = fmul double [[TMP15]], [[K13]]
+; CHECK-NEXT: store <8 x double> [[TMP14]], ptr [[OUT]], align 8
; CHECK-NEXT: store <2 x double> [[TMP11]], ptr [[O12]], align 8
; CHECK-NEXT: ret void
;
More information about the llvm-commits
mailing list