[llvm-branch-commits] [llvm] [LICM] Drop per-iteration AA tags (PR #223530)
Zach Goldthorpe via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Mon Sep 21 10:53:09 PDT 2026
https://github.com/zGoldthorpe updated https://github.com/llvm/llvm-project/pull/223530
>From fd1b8bb163169eb542be827926dc7023e4037e7b Mon Sep 17 00:00:00 2001
From: Zach Goldthorpe <Zach.Goldthorpe at amd.com>
Date: Mon, 14 Sep 2026 16:10:54 -0500
Subject: [PATCH 1/2] [LICM] Drop per-iteration AA tags
---
llvm/lib/Transforms/Scalar/LICM.cpp | 48 ++++++++++++++++++-
.../Transforms/LICM/scalar-promote-aa-tags.ll | 18 +++----
2 files changed, 53 insertions(+), 13 deletions(-)
diff --git a/llvm/lib/Transforms/Scalar/LICM.cpp b/llvm/lib/Transforms/Scalar/LICM.cpp
index cda59610ff4fc5..36d32241e985c3 100644
--- a/llvm/lib/Transforms/Scalar/LICM.cpp
+++ b/llvm/lib/Transforms/Scalar/LICM.cpp
@@ -73,6 +73,7 @@
#include "llvm/IR/IntrinsicInst.h"
#include "llvm/IR/LLVMContext.h"
#include "llvm/IR/Metadata.h"
+#include "llvm/IR/Module.h"
#include "llvm/IR/PatternMatch.h"
#include "llvm/IR/PredIteratorCache.h"
#include "llvm/InitializePasses.h"
@@ -2348,6 +2349,33 @@ static bool isPotentiallyPromotable(const Instruction *I, const Loop *L) {
return false;
}
+/// Returns whether \p N has any operand from the set \p Operands.
+static bool
+hasAnyOperandsFrom(const MDNode *N,
+ const SmallPtrSetImpl<const MDNode *> &Operands) {
+ return N && llvm::any_of(N->operands(), [&](const MDOperand &Op) {
+ return Operands.contains(cast<MDNode>(Op.get()));
+ });
+}
+
+/// Returns the alias scopes declared via llvm.experimental.noalias.scope.decl
+/// to be local to the loop \p L.
+static SmallPtrSet<const MDNode *, 4>
+collectLoopLocalAliasScopes(const Loop *L) {
+ Function *DeclFn = L->getHeader()->getModule()->getFunction(
+ Intrinsic::getName(Intrinsic::experimental_noalias_scope_decl));
+ if (!DeclFn || DeclFn->use_empty())
+ return {};
+
+ SmallPtrSet<const MDNode *, 4> LoopLocalScopes;
+ for (const BasicBlock *BB : L->blocks())
+ for (const Instruction &I : *BB)
+ if (const auto *Decl = dyn_cast<NoAliasScopeDeclInst>(&I))
+ for (const MDOperand &Op : Decl->getScopeList()->operands())
+ LoopLocalScopes.insert(cast<MDNode>(Op.get()));
+ return LoopLocalScopes;
+}
+
/// Returns the potentially promotable stores with AA tags that are valid along
/// all non-unwinding execution paths of the loop \p L, which allows for the AA
/// tags to be used when deciding promotions.
@@ -2361,13 +2389,31 @@ collectStoresWithInvariantAATags(MemorySSA *MSSA, DominatorTree *DT, Loop *L) {
StoresByLoc[MemoryLocation::get(SI)].push_back(SI);
});
+ // A scope declared inside the loop denotes a different scope on each
+ // iteration, so it cannot support the cross-iteration check below.
+ std::optional<SmallPtrSet<const MDNode *, 4>> LoopLocalAliasScopes;
+ auto HasLoopLocalAliasScope = [&](const AAMDNodes &AATags) {
+ if (!AATags.Scope && !AATags.NoAlias)
+ return false;
+ if (!LoopLocalAliasScopes)
+ LoopLocalAliasScopes = collectLoopLocalAliasScopes(L);
+ return hasAnyOperandsFrom(AATags.Scope, *LoopLocalAliasScopes) ||
+ hasAnyOperandsFrom(AATags.NoAlias, *LoopLocalAliasScopes);
+ };
+
// This only looks at explicit exiting blocks. If we ever start sinking
// stores into unwind edges, this will break.
SmallVector<BasicBlock *, 4> ExitingBlocks;
L->getExitingBlocks(ExitingBlocks);
SmallPtrSet<const StoreInst *, 8> StoresWithInvariantAATags;
- for (const auto &Stores : llvm::make_second_range(StoresByLoc)) {
+ for (const auto &Pair : StoresByLoc) {
+ const MemoryLocation &Loc = Pair.first;
+ const SmallVector<const StoreInst *, 1> &Stores = Pair.second;
+
+ if (HasLoopLocalAliasScope(Loc.AATags))
+ continue;
+
// Without exiting blocks the loop is never left, and promotion has no
// exit block to insert a store into either.
if (llvm::all_of(ExitingBlocks, [&](BasicBlock *ExitingBB) {
diff --git a/llvm/test/Transforms/LICM/scalar-promote-aa-tags.ll b/llvm/test/Transforms/LICM/scalar-promote-aa-tags.ll
index 46e1081015be50..8802abf350c52a 100644
--- a/llvm/test/Transforms/LICM/scalar-promote-aa-tags.ll
+++ b/llvm/test/Transforms/LICM/scalar-promote-aa-tags.ll
@@ -414,11 +414,9 @@ define i32 @not_promotable.per_iteration_noalias_scope(i64 %idx, i1 %c, i1 %c2)
; CHECK-SAME: i64 [[IDX:%.*]], i1 [[C:%.*]], i1 [[C2:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[PTR:%.*]] = alloca [4 x i32], align 4
-; CHECK-NEXT: [[PTR_PROMOTED:%.*]] = load i32, ptr [[PTR]], align 4
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[V_INC1:%.*]] = phi i32 [ [[PTR_PROMOTED]], %[[ENTRY]] ], [ [[V_INC2:%.*]], %[[LATCH:.*]] ]
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[IDX]], %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[IDX]], %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
; CHECK-NEXT: call void @llvm.experimental.noalias.scope.decl(metadata [[META8]])
; CHECK-NEXT: [[FPTR:%.*]] = getelementptr i32, ptr [[PTR]], i64 [[IV]]
; CHECK-NEXT: br i1 [[C]], label %[[IF:.*]], label %[[ELSE:.*]]
@@ -426,15 +424,14 @@ define i32 @not_promotable.per_iteration_noalias_scope(i64 %idx, i1 %c, i1 %c2)
; CHECK-NEXT: store i32 42, ptr [[FPTR]], align 4, !alias.scope [[META8]]
; CHECK-NEXT: br label %[[LATCH]]
; CHECK: [[ELSE]]:
+; CHECK-NEXT: [[V_INC1:%.*]] = load i32, ptr [[PTR]], align 4, !noalias [[META8]]
; CHECK-NEXT: [[V_INC:%.*]] = add i32 [[V_INC1]], 1
+; CHECK-NEXT: store i32 [[V_INC]], ptr [[PTR]], align 4, !noalias [[META8]]
; CHECK-NEXT: br i1 [[C2]], label %[[EXIT:.*]], label %[[LATCH]]
; CHECK: [[LATCH]]:
-; CHECK-NEXT: [[V_INC2]] = phi i32 [ [[V_INC]], %[[ELSE]] ], [ [[V_INC1]], %[[IF]] ]
; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
; CHECK-NEXT: br label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[V_INC_LCSSA:%.*]] = phi i32 [ [[V_INC]], %[[ELSE]] ]
-; CHECK-NEXT: store i32 [[V_INC_LCSSA]], ptr [[PTR]], align 4
; CHECK-NEXT: [[RES:%.*]] = load i32, ptr [[PTR]], align 4
; CHECK-NEXT: ret i32 [[RES]]
;
@@ -472,11 +469,9 @@ define i32 @not_promotable.per_iteration_alias_scope(i64 %idx, i1 %c, i1 %c2) {
; CHECK-SAME: i64 [[IDX:%.*]], i1 [[C:%.*]], i1 [[C2:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[PTR:%.*]] = alloca [4 x i32], align 4
-; CHECK-NEXT: [[PTR_PROMOTED:%.*]] = load i32, ptr [[PTR]], align 4
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[V:%.*]] = phi i32 [ [[PTR_PROMOTED]], %[[ENTRY]] ], [ [[V_INC1:%.*]], %[[LATCH:.*]] ]
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[IDX]], %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[IDX]], %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
; CHECK-NEXT: call void @llvm.experimental.noalias.scope.decl(metadata [[META8]])
; CHECK-NEXT: [[FPTR:%.*]] = getelementptr i32, ptr [[PTR]], i64 [[IV]]
; CHECK-NEXT: br i1 [[C]], label %[[IF:.*]], label %[[ELSE:.*]]
@@ -484,15 +479,14 @@ define i32 @not_promotable.per_iteration_alias_scope(i64 %idx, i1 %c, i1 %c2) {
; CHECK-NEXT: store i32 42, ptr [[FPTR]], align 4, !noalias [[META8]]
; CHECK-NEXT: br label %[[LATCH]]
; CHECK: [[ELSE]]:
+; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[PTR]], align 4, !alias.scope [[META8]]
; CHECK-NEXT: [[V_INC:%.*]] = add i32 [[V]], 1
+; CHECK-NEXT: store i32 [[V_INC]], ptr [[PTR]], align 4, !alias.scope [[META8]]
; CHECK-NEXT: br i1 [[C2]], label %[[EXIT:.*]], label %[[LATCH]]
; CHECK: [[LATCH]]:
-; CHECK-NEXT: [[V_INC1]] = phi i32 [ [[V_INC]], %[[ELSE]] ], [ [[V]], %[[IF]] ]
; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
; CHECK-NEXT: br label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[V_INC_LCSSA:%.*]] = phi i32 [ [[V_INC]], %[[ELSE]] ]
-; CHECK-NEXT: store i32 [[V_INC_LCSSA]], ptr [[PTR]], align 4
; CHECK-NEXT: [[RES:%.*]] = load i32, ptr [[PTR]], align 4
; CHECK-NEXT: ret i32 [[RES]]
;
>From 31ec5777804a2a0b5d9f2c34ac36580e0ff15f4d Mon Sep 17 00:00:00 2001
From: Zach Goldthorpe <Zach.Goldthorpe at amd.com>
Date: Tue, 15 Sep 2026 10:37:33 -0500
Subject: [PATCH 2/2] Collect per-iteration alias scopes in advance
---
llvm/lib/Transforms/Scalar/LICM.cpp | 87 ++++++++++++-----------------
1 file changed, 37 insertions(+), 50 deletions(-)
diff --git a/llvm/lib/Transforms/Scalar/LICM.cpp b/llvm/lib/Transforms/Scalar/LICM.cpp
index 36d32241e985c3..3dece37cc2c5fa 100644
--- a/llvm/lib/Transforms/Scalar/LICM.cpp
+++ b/llvm/lib/Transforms/Scalar/LICM.cpp
@@ -224,10 +224,10 @@ static void foreachMemoryAccess(MemorySSA *MSSA, Loop *L,
function_ref<void(Instruction *)> Fn);
using PointersAndHasReadsOutsideSet =
std::pair<SmallSetVector<Value *, 8>, bool>;
-static SmallVector<PointersAndHasReadsOutsideSet, 0>
-collectPromotionCandidates(MemorySSA *MSSA, AliasAnalysis *AA,
- DominatorTree *DT, ICFLoopSafetyInfo *SafetyInfo,
- Loop *L);
+static SmallVector<PointersAndHasReadsOutsideSet, 0> collectPromotionCandidates(
+ MemorySSA *MSSA, AliasAnalysis *AA, DominatorTree *DT,
+ ICFLoopSafetyInfo *SafetyInfo,
+ const SmallPtrSetImpl<const MDNode *> &LoopLocalAliasScopes, Loop *L);
namespace {
struct LoopInvariantCodeMotion {
@@ -445,11 +445,22 @@ bool LoopInvariantCodeMotion::runOnLoop(Loop *L, AAResults *AA, LoopInfo *LI,
// there is currently no general solution for this. Similar issues could also
// potentially happen in other passes where instructions are being moved
// across that edge.
- bool HasCoroSuspendInst = llvm::any_of(L->getBlocks(), [](BasicBlock *BB) {
- using namespace PatternMatch;
- return any_of(make_pointer_range(*BB),
- match_fn(m_Intrinsic<Intrinsic::coro_suspend>()));
- });
+ bool HasCoroSuspendInst = false;
+
+ // AA metadata declared to be local to each iteration cannot be used to infer
+ // alias information when promoting stores.
+ SmallPtrSet<const MDNode *, 4> LoopLocalAliasScopes;
+
+ for (BasicBlock *BB : L->getBlocks()) {
+ for (Instruction &I : *BB) {
+ using namespace PatternMatch;
+ HasCoroSuspendInst |= match(&I, m_Intrinsic<Intrinsic::coro_suspend>());
+
+ if (auto *Decl = dyn_cast<NoAliasScopeDeclInst>(&I))
+ for (const MDOperand &Op : Decl->getScopeList()->operands())
+ LoopLocalAliasScopes.insert(cast<MDNode>(Op.get()));
+ }
+ }
MemorySSAUpdater MSSAU(MSSA);
SinkAndHoistLICMFlags Flags(LicmMssaOptCap, LicmMssaNoAccForPromotionCap,
@@ -520,7 +531,8 @@ bool LoopInvariantCodeMotion::runOnLoop(Loop *L, AAResults *AA, LoopInfo *LI,
do {
LocalPromoted = false;
for (auto [PointerMustAliases, HasReadsOutsideSet] :
- collectPromotionCandidates(MSSA, AA, DT, &SafetyInfo, L)) {
+ collectPromotionCandidates(MSSA, AA, DT, &SafetyInfo,
+ LoopLocalAliasScopes, L)) {
LocalPromoted |= promoteLoopAccessesToScalars(
PointerMustAliases, ExitBlocks, InsertPts, MSSAInsertPts, PIC, LI,
DT, AC, TLI, TTI, L, MSSAU, &SafetyInfo, ORE,
@@ -2351,36 +2363,19 @@ static bool isPotentiallyPromotable(const Instruction *I, const Loop *L) {
/// Returns whether \p N has any operand from the set \p Operands.
static bool
-hasAnyOperandsFrom(const MDNode *N,
- const SmallPtrSetImpl<const MDNode *> &Operands) {
+hasAnyMDOperandsFrom(const MDNode *N,
+ const SmallPtrSetImpl<const MDNode *> &Operands) {
return N && llvm::any_of(N->operands(), [&](const MDOperand &Op) {
return Operands.contains(cast<MDNode>(Op.get()));
});
}
-/// Returns the alias scopes declared via llvm.experimental.noalias.scope.decl
-/// to be local to the loop \p L.
-static SmallPtrSet<const MDNode *, 4>
-collectLoopLocalAliasScopes(const Loop *L) {
- Function *DeclFn = L->getHeader()->getModule()->getFunction(
- Intrinsic::getName(Intrinsic::experimental_noalias_scope_decl));
- if (!DeclFn || DeclFn->use_empty())
- return {};
-
- SmallPtrSet<const MDNode *, 4> LoopLocalScopes;
- for (const BasicBlock *BB : L->blocks())
- for (const Instruction &I : *BB)
- if (const auto *Decl = dyn_cast<NoAliasScopeDeclInst>(&I))
- for (const MDOperand &Op : Decl->getScopeList()->operands())
- LoopLocalScopes.insert(cast<MDNode>(Op.get()));
- return LoopLocalScopes;
-}
-
/// Returns the potentially promotable stores with AA tags that are valid along
/// all non-unwinding execution paths of the loop \p L, which allows for the AA
/// tags to be used when deciding promotions.
-static SmallPtrSet<const StoreInst *, 8>
-collectStoresWithInvariantAATags(MemorySSA *MSSA, DominatorTree *DT, Loop *L) {
+static SmallPtrSet<const StoreInst *, 8> collectStoresWithInvariantAATags(
+ MemorySSA *MSSA, DominatorTree *DT,
+ const SmallPtrSetImpl<const MDNode *> &LoopLocalAliasScopes, Loop *L) {
SmallDenseMap<MemoryLocation, SmallVector<const StoreInst *, 1>, 4>
StoresByLoc;
foreachMemoryAccess(MSSA, L, [&](Instruction *I) {
@@ -2389,18 +2384,6 @@ collectStoresWithInvariantAATags(MemorySSA *MSSA, DominatorTree *DT, Loop *L) {
StoresByLoc[MemoryLocation::get(SI)].push_back(SI);
});
- // A scope declared inside the loop denotes a different scope on each
- // iteration, so it cannot support the cross-iteration check below.
- std::optional<SmallPtrSet<const MDNode *, 4>> LoopLocalAliasScopes;
- auto HasLoopLocalAliasScope = [&](const AAMDNodes &AATags) {
- if (!AATags.Scope && !AATags.NoAlias)
- return false;
- if (!LoopLocalAliasScopes)
- LoopLocalAliasScopes = collectLoopLocalAliasScopes(L);
- return hasAnyOperandsFrom(AATags.Scope, *LoopLocalAliasScopes) ||
- hasAnyOperandsFrom(AATags.NoAlias, *LoopLocalAliasScopes);
- };
-
// This only looks at explicit exiting blocks. If we ever start sinking
// stores into unwind edges, this will break.
SmallVector<BasicBlock *, 4> ExitingBlocks;
@@ -2411,7 +2394,10 @@ collectStoresWithInvariantAATags(MemorySSA *MSSA, DominatorTree *DT, Loop *L) {
const MemoryLocation &Loc = Pair.first;
const SmallVector<const StoreInst *, 1> &Stores = Pair.second;
- if (HasLoopLocalAliasScope(Loc.AATags))
+ // A scope declared inside the loop denotes a different scope on each
+ // iteration, and thus should not be preserved.
+ if (hasAnyMDOperandsFrom(Loc.AATags.Scope, LoopLocalAliasScopes) ||
+ hasAnyMDOperandsFrom(Loc.AATags.NoAlias, LoopLocalAliasScopes))
continue;
// Without exiting blocks the loop is never left, and promotion has no
@@ -2428,10 +2414,10 @@ collectStoresWithInvariantAATags(MemorySSA *MSSA, DominatorTree *DT, Loop *L) {
// The bool indicates whether there might be reads outside the set, in which
// case only loads may be promoted.
-static SmallVector<PointersAndHasReadsOutsideSet, 0>
-collectPromotionCandidates(MemorySSA *MSSA, AliasAnalysis *AA,
- DominatorTree *DT, ICFLoopSafetyInfo *SafetyInfo,
- Loop *L) {
+static SmallVector<PointersAndHasReadsOutsideSet, 0> collectPromotionCandidates(
+ MemorySSA *MSSA, AliasAnalysis *AA, DominatorTree *DT,
+ ICFLoopSafetyInfo *SafetyInfo,
+ const SmallPtrSetImpl<const MDNode *> &LoopLocalAliasScopes, Loop *L) {
BatchAAResults BatchAA(*AA);
AliasSetTracker AST(BatchAA);
@@ -2440,7 +2426,8 @@ collectPromotionCandidates(MemorySSA *MSSA, AliasAnalysis *AA,
std::optional<SmallPtrSet<const StoreInst *, 8>> StoresWithInvariantAATags;
auto HasInvariantAATags = [&](const StoreInst *SI) {
if (!StoresWithInvariantAATags)
- StoresWithInvariantAATags = collectStoresWithInvariantAATags(MSSA, DT, L);
+ StoresWithInvariantAATags =
+ collectStoresWithInvariantAATags(MSSA, DT, LoopLocalAliasScopes, L);
return StoresWithInvariantAATags->contains(SI);
};
More information about the llvm-branch-commits
mailing list